Compare commits
64
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4d01bb3ecf | ||
|
|
332c63e440 | ||
|
|
b306c210c6 | ||
|
|
b9015e1c2e | ||
|
|
26c5008136 | ||
|
|
74d6b90d69 | ||
|
|
7b11c62f53 | ||
|
|
8eed54b3da | ||
|
|
4be16dcdf5 | ||
|
|
ded9cabdf4 | ||
|
|
cf6bc6987e | ||
|
|
ff7b1452b1 | ||
|
|
517c1cea25 | ||
|
|
a3e3010e0f | ||
|
|
057c4eb545 | ||
|
|
f720381d43 | ||
|
|
2c9042d62c | ||
|
|
82ef289d3a | ||
|
|
7246b0e002 | ||
|
|
8cfd40aac8 | ||
|
|
5382c9a8e4 | ||
|
|
53de91b2df | ||
|
|
8a36af7c7d | ||
|
|
e9789ce3f4 | ||
|
|
837231c068 | ||
|
|
95025be1bc | ||
|
|
03a964bb2d | ||
|
|
123a16e346 | ||
|
|
9701812bee | ||
|
|
29ac03468e | ||
|
|
bfb7701db1 | ||
|
|
e8b6ff5d7c | ||
|
|
1456907000 | ||
|
|
ec1c521187 | ||
|
|
a7744c24bd | ||
|
|
522e6f2ae8 | ||
|
|
81d4bd81e4 | ||
|
|
687678a2ea | ||
|
|
b0f9071bf5 | ||
|
|
81e2673923 | ||
|
|
75e9fd771b | ||
|
|
863926abd6 | ||
|
|
241e7256f6 | ||
|
|
6556b85abf | ||
|
|
289cabe993 | ||
|
|
d6cf7cfa44 | ||
|
|
4cc2f0eba5 | ||
|
|
97dfaa7526 | ||
|
|
66aa4dbc8b | ||
|
|
dce5d31462 | ||
|
|
9dc3987e02 | ||
|
|
9b238a525a | ||
|
|
ad82aac663 | ||
|
|
0629f5e2df | ||
|
|
ecb203dcf5 | ||
|
|
6c672567f3 | ||
|
|
cc6e416c59 | ||
|
|
c66a47973a | ||
|
|
9629897202 | ||
|
|
5399a8a724 | ||
|
|
de5d9f358e | ||
|
|
ca3fdce0e0 | ||
|
|
fc2d92eabd | ||
|
|
39d2e80145 |
@@ -325,13 +325,27 @@ jobs:
|
|||||||
chomp $id;
|
chomp $id;
|
||||||
my @files = grep { -f $_ } glob(q{dist/*/*});
|
my @files = grep { -f $_ } glob(q{dist/*/*});
|
||||||
@files or die qq{ERROR: no assets under dist/\n};
|
@files or die qq{ERROR: no assets under dist/\n};
|
||||||
|
# A file that arrived empty from the artifact step would be uploaded as an
|
||||||
|
# empty attachment, every status would still be 201, and the run would go
|
||||||
|
# green over a release nobody can install. Refuse it here, before the
|
||||||
|
# upload, and verify what was stored afterwards.
|
||||||
|
my %size;
|
||||||
|
for my $path (@files) {
|
||||||
|
my $n = -s $path // 0;
|
||||||
|
(my $name = $path) =~ s{.*/}{};
|
||||||
|
$n > 0 or die qq{ERROR: $path is empty, so there is nothing to upload\n};
|
||||||
|
$size{$name} = $n;
|
||||||
|
}
|
||||||
my $bad = 0;
|
my $bad = 0;
|
||||||
for my $path (@files) {
|
for my $path (@files) {
|
||||||
(my $name = $path) =~ s{.*/}{};
|
(my $name = $path) =~ s{.*/}{};
|
||||||
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
|
my @cmd = (q{curl}, q{-sS}, q{-o}, q{/dev/null}, q{-w}, q{%{http_code}},
|
||||||
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}},
|
||||||
q{-H}, q{Content-Type: application/octet-stream},
|
q{-H}, q{Content-Type: application/octet-stream},
|
||||||
q{-X}, q{POST}, q{--data-binary}, qq{@$path},
|
# The @ must not sit inside a qq{} string: there it starts an
|
||||||
|
# array interpolation and the upload body collapses to empty,
|
||||||
|
# which Gitea stores as a 201-created zero-byte attachment.
|
||||||
|
q{-X}, q{POST}, q{--data-binary}, q{@} . $path,
|
||||||
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
|
qq{$ENV{GITEA_SERVER_URL}/api/v1/repos/$ENV{GITEA_REPOSITORY}/releases/$id/assets?name=$name});
|
||||||
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
open(my $curl, q{-|}, @cmd) or die qq{curl: $!};
|
||||||
my $code = <$curl>;
|
my $code = <$curl>;
|
||||||
@@ -346,5 +360,31 @@ jobs:
|
|||||||
printf qq{%s: HTTP %s\n}, $name, $code;
|
printf qq{%s: HTTP %s\n}, $name, $code;
|
||||||
$bad = 1 if $code ne q{201};
|
$bad = 1 if $code ne q{201};
|
||||||
}
|
}
|
||||||
|
# Read every asset back through the release download route and require the
|
||||||
|
# served length to be the file that was sent: stored but empty is a broken
|
||||||
|
# release however green the run looks.
|
||||||
|
open(my $v, q{<}, q{version-no-v.txt}) or die qq{version-no-v.txt: $!};
|
||||||
|
my $v = <$v>;
|
||||||
|
close($v);
|
||||||
|
chomp $v;
|
||||||
|
for my $name (sort keys %size) {
|
||||||
|
my $url = qq{$ENV{GITEA_SERVER_URL}/$ENV{GITEA_REPOSITORY}/releases/download/v$v/$name};
|
||||||
|
my @head = (q{curl}, q{-sS}, q{-I}, q{-H}, qq{Authorization: token $ENV{GITEA_TOKEN}}, $url);
|
||||||
|
open(my $h, q{-|}, @head) or die qq{curl: $!};
|
||||||
|
my $len;
|
||||||
|
my $status;
|
||||||
|
while (my $l = <$h>) {
|
||||||
|
$status = $1 if $l =~ m{^HTTP/\S+\s+(\d+)};
|
||||||
|
$len = $1 if $l =~ m{^content-length:\s*(\d+)}i;
|
||||||
|
}
|
||||||
|
my $ok = close($h);
|
||||||
|
$len = defined $len ? $len : 0;
|
||||||
|
if (!$ok || $status != 200 || $len != $size{$name}) {
|
||||||
|
printf qq{ERROR: %s serves %s bytes, expected %d\n}, $name, $len, $size{$name};
|
||||||
|
$bad = 1;
|
||||||
|
next;
|
||||||
|
}
|
||||||
|
printf qq{%s: serves %d bytes\n}, $name, $len;
|
||||||
|
}
|
||||||
exit($bad ? 1 : 0);
|
exit($bad ? 1 : 0);
|
||||||
'
|
'
|
||||||
|
|||||||
@@ -56,6 +56,28 @@ jobs:
|
|||||||
- name: Build
|
- name: Build
|
||||||
run: go build ./...
|
run: go build ./...
|
||||||
|
|
||||||
|
- name: FreeBSD build (amd64)
|
||||||
|
# The debugger's ptrace surface and the JIT substrate are the two
|
||||||
|
# FreeBSD-portable layers the tree carries; the forge has no FreeBSD
|
||||||
|
# runner, so a push can only compile-gate them. Running the ptrace
|
||||||
|
# suite needs real FreeBSD hardware.
|
||||||
|
env:
|
||||||
|
GOOS: freebsd
|
||||||
|
GOARCH: amd64
|
||||||
|
run: go build ./...
|
||||||
|
|
||||||
|
- name: FreeBSD build (arm64)
|
||||||
|
env:
|
||||||
|
GOOS: freebsd
|
||||||
|
GOARCH: arm64
|
||||||
|
run: go build ./...
|
||||||
|
|
||||||
|
- name: FreeBSD build (riscv64)
|
||||||
|
env:
|
||||||
|
GOOS: freebsd
|
||||||
|
GOARCH: riscv64
|
||||||
|
run: go build ./...
|
||||||
|
|
||||||
- name: Format
|
- name: Format
|
||||||
run: |
|
run: |
|
||||||
perl -e '
|
perl -e '
|
||||||
|
|||||||
+198
-2
@@ -1,6 +1,6 @@
|
|||||||
# Changelog
|
# Changelog
|
||||||
|
|
||||||
All notable changes to gasm-devkit are documented here.
|
All notable changes to gasm-sdk are documented here.
|
||||||
|
|
||||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
||||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||||
@@ -9,7 +9,198 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|
||||||
-
|
- **The FreeBSD port of the debugger.** `gasm debug` runs on FreeBSD on
|
||||||
|
amd64, arm64 and riscv64: the same interactive surface as on Linux —
|
||||||
|
breakpoints, hardware watchpoints (x86 debug registers, the arm64 debug
|
||||||
|
register file), single-stepping, register and memory access — behind the
|
||||||
|
kernel's own ptrace requests, with tracee memory through `PT_IO` and
|
||||||
|
stop reports through `PT_LWPINFO`. The JIT substrate maps executable
|
||||||
|
memory through `golang.org/x/sys/unix`, so `verify` builds on FreeBSD
|
||||||
|
too. The pipeline compile-gates all three architectures; live
|
||||||
|
validation awaits a FreeBSD machine.
|
||||||
|
- **Workspace-wide navigation in the language server.** `gasm lsp` indexes
|
||||||
|
the `.s` files under the workspace root beyond the documents the editor
|
||||||
|
has open, so go-to-definition, find references and workspace symbol search
|
||||||
|
reach files that were never opened. An open buffer always shadows its
|
||||||
|
disk copy, and watched-file events together with a per-query freshness
|
||||||
|
check keep the index current.
|
||||||
|
- **Quick fixes for the textflag include and the argument area.** The
|
||||||
|
`missing-textflag-include` warning offers to add the include after the
|
||||||
|
last one in the file, and the `abi-argsize` warning offers to set the
|
||||||
|
TEXT argument area to the size the `// func` signature implies, computed
|
||||||
|
by the new `lint.ExpectedArgSize`.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **The corpus audit assembles like the build.** A file's `//go:build`
|
||||||
|
constraint decides which target architectures attempt it: cpu_x86.s is
|
||||||
|
an x86 build alone, and the msan and goexperiment.runtimesecret trees
|
||||||
|
are compiled by no supported build, so they leave the measured set
|
||||||
|
instead of failing it. The headline now reads "assemble for every
|
||||||
|
applicable target": every real-code GOROOT assembly file, the tree
|
||||||
|
without testdata, assembles for all four architectures (250 of 250,
|
||||||
|
100 %); over the whole tree including testdata the measure is 271 of
|
||||||
|
322 (84.2 %).
|
||||||
|
- **The module moves to `sourcedock.dev/petrbalvin/gasm-sdk`.** The
|
||||||
|
repository and the module rename together with the product, now the
|
||||||
|
GAsm Software Development Kit. Fresh installs become
|
||||||
|
`go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest`, and
|
||||||
|
installs pinned to the old `gasm-sdk` path stop resolving once the
|
||||||
|
repository takes the new name: reinstall from the new path. The
|
||||||
|
binary stays `gasm`.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **Rename edits land in their own documents.** A rename collected the
|
||||||
|
ranges of every reference across the open documents but applied them all
|
||||||
|
to the document that started it, so renaming a symbol used in a second
|
||||||
|
file moved that file's text into the first. Each edit now applies to the
|
||||||
|
document it was collected in.
|
||||||
|
- **Negative numeric PC-relative jumps.** `JMP -3(PC)`, the shape the
|
||||||
|
runtime's exit loops write (sys_linux_amd64.s, sys_netbsd_amd64.s),
|
||||||
|
resolved to nothing: only the forward forms counted. A negative count
|
||||||
|
now walks the same instruction statements backwards, labels excluded,
|
||||||
|
byte-identical with the toolchain.
|
||||||
|
- **The arm64 move-wide family reads its immediate as an unsigned
|
||||||
|
pattern.** `MOVK $(40000<<48)` folds to a negative int64 and was
|
||||||
|
rejected; the toolchain picks the 16-bit lane from the 64-bit bit
|
||||||
|
pattern, so the encoder now does the same, and a zero immediate is
|
||||||
|
rejected where the toolchain rejects it.
|
||||||
|
|
||||||
|
## [0.35.0] - 2026-09-22
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **The go_asm.h generator.** `gasm asm` generates the package's go_asm.h
|
||||||
|
itself when an assembly file includes it: the Go files beside the source
|
||||||
|
are type-checked for the target architecture and the constants and field
|
||||||
|
offsets become assembler defines, so package-context files assemble with
|
||||||
|
no compiler and no `go build` in the loop. `-GOOS` selects the
|
||||||
|
type-checking GOOS for GOOS-specific files, and the corpus audit derives
|
||||||
|
the GOOS from the file name.
|
||||||
|
- **ELF data relocations on arm64, riscv64 and loong64.** `gasm asm
|
||||||
|
--format elf` emits `.rela.data` for symbol-valued DATA initialisers on
|
||||||
|
every architecture (amd64 carried them already), so standalone ELF
|
||||||
|
objects link on all four targets.
|
||||||
|
- **Corpus failure listing.** `gasm audit-instructions --corpus --list`
|
||||||
|
prints every failing file with its failure reason, per architecture,
|
||||||
|
instead of one representative file per reason.
|
||||||
|
- **DATA with symbol values and relaxed symbol spellings.** DATA
|
||||||
|
initialisers accept `$symbol(SB)` values, laid down as an absolute
|
||||||
|
relocation at the data field (GOOBJ on all four architectures and ELF
|
||||||
|
on all four as of this release), and U+2215 is accepted inside symbol
|
||||||
|
package paths.
|
||||||
|
- **Macro expansion and include splicing.** `gasm asm`, `gasm diff` and
|
||||||
|
`gasm audit-instructions` now preprocess assembly the way the
|
||||||
|
toolchain does: object and parameterised `#define` macros expand at
|
||||||
|
the point of use, `#undef` and the `#ifdef`/`#ifndef`/`#else`/
|
||||||
|
`#endif` family select branches, `#include` splices headers resolved
|
||||||
|
through the source directory and the new repeatable `-I` flag, `;`
|
||||||
|
separates statements, and constant expressions left in operands
|
||||||
|
(`$(32-7)`, `$~63`, `(index*4)(base)`) fold at parse. Expansion
|
||||||
|
happens only on the assembly path: `gasm lint`, `gasm fmt` and the
|
||||||
|
language server keep reading the raw file.
|
||||||
|
- **Encoder coverage: the instruction families GOROOT's real code
|
||||||
|
uses.** The encoder now covers the
|
||||||
|
instruction families GOROOT's real code uses that gasm lacked,
|
||||||
|
byte-verified against `go tool asm`: on amd64 the carry ALU, the
|
||||||
|
atomics (CMPXCHG, XADD, XCHG), AES-NI, SHA-1/256, PCLMULQDQ, CRC32,
|
||||||
|
GFNI, ADX, BMI, the string primitives, the system set (CPUID, RDTSC,
|
||||||
|
SYSCALL, fences, MXCSR) and the SSE/AVX/EVEX gaps; on arm64 the pair
|
||||||
|
loads and stores (LDP/STP), acquire/release and LSE atomics, AES and
|
||||||
|
SHA, the system operations, the bit ops and the NEON slice including
|
||||||
|
structure loads and the literal-pool moves; on riscv64 the RV64A AMO
|
||||||
|
family with aq/rl ordering, the Zbb pseudos with their RVC
|
||||||
|
compressions, the FMA forms and the RVV slice with `vsetvli`/
|
||||||
|
`vsetivli`; on loong64 the AM atomics with acquire/release forms, the
|
||||||
|
LSX/LASX slice, the `VMOVQ`/`XVMOVQ` transfer family and FSEL.
|
||||||
|
Also fixed on the way: arm64 `CASD`/`CASW` lacked an opcode bit, and
|
||||||
|
riscv64 `VSETVLI` with an immediate length now canonicalises to
|
||||||
|
`vsetivli` as the toolchain does.
|
||||||
|
- **Encoder coverage: quad-register AVX-512 and floating-point
|
||||||
|
immediates.** The encoder gains the
|
||||||
|
quad-register AVX-512 families (4FMAPS, 4FNMADD, 4VNNIW, VP4DPWSSD,
|
||||||
|
VP4DPWSSDS) with the register list riding the inverted V'VVVV field,
|
||||||
|
floating-point immediates on the SSE scalar moves and arithmetic
|
||||||
|
(the constant lands in a synthesised read-only pool, a positive zero
|
||||||
|
collapses to XORPS exactly as the toolchain does), accept-and-ignore
|
||||||
|
FUNCDATA and PCDATA, three-operand double shifts, static-symbol
|
||||||
|
operands for the legacy SSE moves, and the pooled 64-bit immediate
|
||||||
|
materialisation on riscv64. The parser carries bracketed register
|
||||||
|
ranges, index-only VSIB memory operands and bare trailing immediates;
|
||||||
|
macro substitution reaches parameters used with element suffixes
|
||||||
|
(`A.S4`), and `;` separates statements in plain files.
|
||||||
|
- **Per-architecture reference pages.** [docs/asm/](docs/asm/README.md)
|
||||||
|
gains AMD64, ARM64, RISCV64 and LOONG64: the register files and the
|
||||||
|
roles the ABI fixes, addressing, operand order with every special form,
|
||||||
|
constants and materialisation, alignment, fences and the relocations
|
||||||
|
each target emits. An instruction inventory appendix per architecture
|
||||||
|
is generated from the toolchain's own tables by `just gen`, and the
|
||||||
|
regenerated tables recognise 147 more mnemonics than the previous
|
||||||
|
release carried (arm64 107, riscv64 31, loong64 9).
|
||||||
|
- **The Plan 9 assembly language reference.** [docs/asm/](docs/asm/README.md)
|
||||||
|
opens the complete language reference with its common core: the lexicon,
|
||||||
|
statement structure and constant expressions, the operand grammar with
|
||||||
|
the pseudo-registers and symbol naming, the directives and the function
|
||||||
|
flag vocabulary, preprocessing with `#define` and `#include`, and the
|
||||||
|
Go-embedded layer (ABI0, prototypes, `go_asm.h`, `funcdata.h` and the
|
||||||
|
runtime contract). Every claim is verified against `go tool asm` of
|
||||||
|
Go 1.27.1 and gasm's differential tests; the per-architecture pages and
|
||||||
|
generated instruction appendices follow.
|
||||||
|
- **GOOBJ format specification.** [docs/GOOBJ.md](docs/GOOBJ.md)
|
||||||
|
documents the Go object file format in full: both containers, the 96
|
||||||
|
byte header and all 19 blocks, every structure with its byte
|
||||||
|
offsets, symbol kinds and flag bits, all 106 relocation types with
|
||||||
|
the weak variants, aux symbols, the FuncInfo payload, the pc-value
|
||||||
|
table encoding, the content hashes and the builtin table, all
|
||||||
|
verified byte for byte against objects produced by Go 1.27.1's own
|
||||||
|
tools.
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- **The corpus audit measures like a build.** Files named for a Go port
|
||||||
|
gasm does not target (arm, 386, s390x, ...) are never attempted, because
|
||||||
|
no supported build compiles them; the GOOS comes from the file name; and
|
||||||
|
each target's go_asm.h is generated on the fly. The headline is reported
|
||||||
|
over attemptable files: 291 of 353 on the full corpus (82.4 %) assemble
|
||||||
|
for every target architecture and 295 of 303 on real code (97.4 %),
|
||||||
|
against 108 of 627 over all files (17.2 %) that the previous release
|
||||||
|
measured.
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **The operand forms GOROOT writes.** Numeric PC-relative jumps
|
||||||
|
(`JEQ 2(PC)`, the park loop `JMP 0(PC)`) resolve with the toolchain's
|
||||||
|
own instruction counting and fold jump-to-jump chains exactly as its
|
||||||
|
branch optimiser does; symbol immediates (`MOVQ $sym(SB), AX`)
|
||||||
|
assemble to the toolchain's RIP-relative LEA with an R_PCREL
|
||||||
|
relocation; negated constant expressions in operands (`ADJSP
|
||||||
|
$-(REGS - 8)`, the shape the cgo ABI macros write) fold; the immediate
|
||||||
|
multiply (`IMULQ $1000000000, AX`) encodes with the toolchain's
|
||||||
|
0x69/0x6B selection; the TLS access pair assembles as the toolchain's
|
||||||
|
one-instruction form (the bare `MOVQ TLS, r` load nops out and
|
||||||
|
`off(r)(TLS*1)` folds to the segment-prefixed absolute whose disp32
|
||||||
|
carries the R_TLSLE relocation, per-GOOS); arm64 accepts the
|
||||||
|
bare-register indirect branch (`BL R9` beside `BL (R9)`, both BLR) and
|
||||||
|
the zero-immediate store (`MOVD $0, mem` through the zero register,
|
||||||
|
rejecting non-zero immediates as the toolchain does); `PCALIGN` now
|
||||||
|
aligns on amd64, padding with the toolchain's greedy
|
||||||
|
single-instruction NOPs; the segment-absolute forms (`MOVQ 0x30(GS),
|
||||||
|
AX` and the store direction) and the absolute crash-store
|
||||||
|
(`MOVL $0xf1, 0xf1`) encode; and `gasm asm` predefines the
|
||||||
|
`GOARCH_<arch>` and `GOOS_<goos>` macros the go command passes to
|
||||||
|
`go tool asm`, so GOROOT headers' `#ifdef GOARCH_amd64` platform
|
||||||
|
blocks (`go_tls.h`'s `get_tls` and friends) select as intended. The
|
||||||
|
GOROOT corpus measure moves to 291 of 353 files assembling for every
|
||||||
|
target architecture (82.4 %), 97.4 % of the real-code corpus, from
|
||||||
|
70.8 % and 82.2 %.
|
||||||
|
- **Tool corrections across the pipeline.** The formatter keeps square
|
||||||
|
brackets in SIMD operands, statement separators and canonical macro
|
||||||
|
bodies; the linter drops false positives on shift counts, SETcc
|
||||||
|
spellings and ABIInternal references; the lexer treats a trailing
|
||||||
|
carriage return as a line end so comment text stays idempotent; and
|
||||||
|
arm64 rejects bare BTI with a diagnostic while accepting the full
|
||||||
|
family.
|
||||||
|
|
||||||
## [0.34.0] - 2026-09-20
|
## [0.34.0] - 2026-09-20
|
||||||
|
|
||||||
@@ -106,6 +297,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
- **The corpus audit attempts fewer files that no build would compile.**
|
||||||
|
Files named for Go ports gasm does not target (arm, 386, s390x, ...)
|
||||||
|
are reported as other-port and never attempted, the headline rate is
|
||||||
|
computed over attemptable files, and the audit searches the
|
||||||
|
toolchain's shipped headers (funcdata.h and friends) automatically.
|
||||||
- **riscv64 JALR silently jumped to the wrong register.** The trampoline
|
- **riscv64 JALR silently jumped to the wrong register.** The trampoline
|
||||||
form `JALR X0, 0(X5)` read the memory operand's base as the destination,
|
form `JALR X0, 0(X5)` read the memory operand's base as the destination,
|
||||||
encoding a jump to X0 with no diagnostic; the destination is the first
|
encoding a jump to X0 with no diagnostic; the destination is the first
|
||||||
|
|||||||
+4
-4
@@ -1,6 +1,6 @@
|
|||||||
# Contributing
|
# Contributing
|
||||||
|
|
||||||
Contributions to **gasm-devkit** are governed by the Contributor terms
|
Contributions to **gasm-sdk** are governed by the Contributor terms
|
||||||
below; submitting one means you accept them.
|
below; submitting one means you accept them.
|
||||||
|
|
||||||
## Contributor terms
|
## Contributor terms
|
||||||
@@ -29,8 +29,8 @@ compiler (gcc), because `just gates` includes `just race` and the race
|
|||||||
detector needs cgo.
|
detector needs cgo.
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
|
git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
|
||||||
cd gasm-devkit
|
cd gasm-sdk
|
||||||
just build
|
just build
|
||||||
just gates
|
just gates
|
||||||
```
|
```
|
||||||
@@ -126,7 +126,7 @@ tag, where it would double the time and the memory a shared runner cannot spare.
|
|||||||
|
|
||||||
## Reporting bugs
|
## Reporting bugs
|
||||||
|
|
||||||
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
|
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
|
||||||
version, the operating system and architecture, the exact command, the full output,
|
version, the operating system and architecture, the exact command, the full output,
|
||||||
and the expected against the actual behaviour.
|
and the expected against the actual behaviour.
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# Plan 9 assembly tooling, inside and outside Go
|
# GAsm: Software Development Kit for Plan 9 Assembly
|
||||||
|
|
||||||
> **Warning: this is an experiment.** gasm-devkit is under active
|
> **Warning: this is an experiment.** gasm-sdk is under active
|
||||||
> development and is not stable. The version is 0.x.x: commands, flags,
|
> development and is not stable. The version is 0.x.x: commands, flags,
|
||||||
> output formats and behaviour can change without warning at any time.
|
> output formats and behaviour can change without warning at any time.
|
||||||
> A 1.0.0 release is light years away. Nothing in this document is a
|
> A 1.0.0 release is light years away. Nothing in this document is a
|
||||||
@@ -13,7 +13,7 @@
|
|||||||
there is no formatter, no linter and no debugger for `.s` files, and no
|
there is no formatter, no linter and no debugger for `.s` files, and no
|
||||||
assembler that works without a Go installation. Developers write
|
assembler that works without a Go installation. Developers write
|
||||||
assembly blind, validate it by benchmark, and debug it by print
|
assembly blind, validate it by benchmark, and debug it by print
|
||||||
statement. gasm-devkit is the missing toolkit: a single, self-contained
|
statement. gasm-sdk is the missing toolkit: a single, self-contained
|
||||||
binary, `gasm`, that serves both purposes.
|
binary, `gasm`, that serves both purposes.
|
||||||
|
|
||||||
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
|
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
|
||||||
@@ -58,7 +58,7 @@ Plan 9 (Go): MOVQ AX, total-16(SP)
|
|||||||
|
|
||||||
The same lines, but only one of them tells you what the number is for.
|
The same lines, but only one of them tells you what the number is for.
|
||||||
The syntax is uppercase, regular and boring, which is the highest
|
The syntax is uppercase, regular and boring, which is the highest
|
||||||
compliment a language for machine code can earn. gasm-devkit exists
|
compliment a language for machine code can earn. gasm-sdk exists
|
||||||
to give that syntax the tooling it deserves.
|
to give that syntax the tooling it deserves.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
@@ -81,7 +81,10 @@ to give that syntax the tooling it deserves.
|
|||||||
GOOBJ format, which needs the installed toolchain and which `go build`
|
GOOBJ format, which needs the installed toolchain and which `go build`
|
||||||
consumes in place of the toolchain's output. Framed functions get the
|
consumes in place of the toolchain's output. Framed functions get the
|
||||||
stack-split guard and the morestack block, byte-identical to the
|
stack-split guard and the morestack block, byte-identical to the
|
||||||
toolchain's, so split functions link too.
|
toolchain's, so split functions link too. The assembler preprocesses
|
||||||
|
like the toolchain (`#define`, `#include` with `-I`, `#ifdef`), generates
|
||||||
|
`go_asm.h` from the package's Go files, and carries `PCALIGN`, the
|
||||||
|
`LOCK`/`REP` prefixes and the literal-data pseudo-ops.
|
||||||
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
|
- **Disassembler.** `gasm dis` lists a `.s` file's functions at their real
|
||||||
offsets after assembling, or disassembles raw bytes from a file or stdin.
|
offsets after assembling, or disassembles raw bytes from a file or stdin.
|
||||||
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
|
- **Dynamic verification.** `gasm verify` JIT-loads assembled functions into
|
||||||
@@ -91,13 +94,16 @@ to give that syntax the tooling it deserves.
|
|||||||
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
|
||||||
breakpoints (optionally conditional), hardware watchpoints, register and
|
breakpoints (optionally conditional), hardware watchpoints, register and
|
||||||
memory inspection, and headless script runs that report instruction and
|
memory inspection, and headless script runs that report instruction and
|
||||||
label coverage.
|
label coverage; it runs on Linux (all four architectures) and FreeBSD
|
||||||
|
(amd64, arm64, riscv64).
|
||||||
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
|
||||||
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
push and pull diagnostics, semantic-token highlighting, go-to-definition,
|
||||||
find references, rename, formatting, inlay hints, code actions, signature
|
find references, rename, formatting, inlay hints, code actions, signature
|
||||||
help, document highlights, workspace symbol search, #include document
|
help, document highlights, workspace symbol search, #include document
|
||||||
links and folding ranges over stdio; definition, references and rename
|
links and folding ranges over stdio; definition, references and rename
|
||||||
work across every open document.
|
work across every open document and the indexed workspace files beyond
|
||||||
|
them, and the quick fixes add a missing textflag.h include and set the
|
||||||
|
argument area from the // func signature.
|
||||||
- **Comparators and audits.** `gasm diff` compares the machine code of two
|
- **Comparators and audits.** `gasm diff` compares the machine code of two
|
||||||
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
|
||||||
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
`gasm audit-instructions` diffs the encoder against the installed toolchain,
|
||||||
@@ -110,9 +116,9 @@ Four architectures, the four that matter in practice:
|
|||||||
| Architecture | GOARCH | File suffix | Instructions recognised |
|
| Architecture | GOARCH | File suffix | Instructions recognised |
|
||||||
|--------------|-------------|--------------|---------------------------------------------|
|
|--------------|-------------|--------------|---------------------------------------------|
|
||||||
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
| AMD64 | `amd64` | `_amd64.s` | 1600 + common opcodes + traditional aliases |
|
||||||
| ARM64 | `arm64` | `_arm64.s` | 538 + common opcodes |
|
| ARM64 | `arm64` | `_arm64.s` | 645 + common opcodes |
|
||||||
| RISC-V | `riscv64` | `_riscv64.s` | 961 + common opcodes |
|
| RISC-V | `riscv64` | `_riscv64.s` | 992 + common opcodes |
|
||||||
| LoongArch | `loong64` | `_loong64.s` | 799 + common opcodes |
|
| LoongArch | `loong64` | `_loong64.s` | 808 + common opcodes |
|
||||||
|
|
||||||
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
"Common opcodes" are the instructions shared by every architecture (`RET`,
|
||||||
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
`JMP`, `NOP`, `CALL`, `TEXT`, `FUNCDATA`, `PCDATA`, ...). AMD64 additionally
|
||||||
@@ -124,9 +130,13 @@ can emit today is narrower, and a recognised but unencodable instruction is
|
|||||||
reported as an explicit error, never as a wrong byte.
|
reported as an explicit error, never as a wrong byte.
|
||||||
|
|
||||||
The same measurement runs over GOROOT's whole assembly corpus:
|
The same measurement runs over GOROOT's whole assembly corpus:
|
||||||
`gasm audit-instructions --corpus` reports 127 of 627 files (20.3 %)
|
`gasm audit-instructions --corpus` reports every real-code GOROOT assembly
|
||||||
assembling for every target architecture today, with the top failure
|
file (the tree without testdata) assembling for every target its build
|
||||||
reasons per architecture; the number moves with every release.
|
admits: 250 of 250, 100 %. Over the whole tree including testdata the
|
||||||
|
measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
|
||||||
|
are counted but never attempted, and `//go:build` constraints decide which
|
||||||
|
targets attempt a file at all, exactly as the build does. The number moves
|
||||||
|
with every release.
|
||||||
|
|
||||||
### Validation status
|
### Validation status
|
||||||
|
|
||||||
@@ -141,7 +151,7 @@ actually been executed.
|
|||||||
|---|---|---|
|
|---|---|---|
|
||||||
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
|
||||||
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
|
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
|
||||||
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
|
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64, riscv64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
|
||||||
|
|
||||||
Consequences, stated plainly. An emulator is a model of a CPU, not the
|
Consequences, stated plainly. An emulator is a model of a CPU, not the
|
||||||
CPU: instruction semantics are implemented in software and can differ
|
CPU: instruction semantics are implemented in software and can differ
|
||||||
@@ -156,6 +166,39 @@ been compiled and read, never executed. Its architecture-neutral units
|
|||||||
run under `go test ./...`, which the race workflow and a manual run
|
run under `go test ./...`, which the race workflow and a manual run
|
||||||
perform; the default `just test` gate does not sweep `./debug/...`.
|
perform; the default `just test` gate does not sweep `./debug/...`.
|
||||||
|
|
||||||
|
## The documentation goal
|
||||||
|
|
||||||
|
The toolkit is the primary goal. The secondary one is documentation: a
|
||||||
|
specification of the Plan 9 assembly language and of the GOOBJ object
|
||||||
|
format that is 100 % complete, detailed enough to implement against,
|
||||||
|
and written to a professional standard. These are the two subjects this
|
||||||
|
project works with every day, and they are the two for which no usable
|
||||||
|
documentation exists.
|
||||||
|
|
||||||
|
Go documents the language on a single page, "A Quick Guide to Go's
|
||||||
|
Assembler", which carries no section for loong64, one of the four
|
||||||
|
architectures gasm supports, and covers a fraction of what each
|
||||||
|
assembler accepts. What exists beyond it lives as comments inside the
|
||||||
|
toolchain's internal source: per-architecture reference manuals for
|
||||||
|
arm64, ppc64, riscv64 and loong64, written for the toolchain's own
|
||||||
|
maintainers rather than for an outside reader, and none at all for
|
||||||
|
amd64. GOOBJ fares worst of all. The format that `go build` consumes
|
||||||
|
has no specification anywhere: it is described by a comment in an
|
||||||
|
internal package, it is not a stable interface, and it can change with
|
||||||
|
any toolchain release.
|
||||||
|
|
||||||
|
The gap is therefore filled the only way it can be filled: by reverse
|
||||||
|
engineering the toolchain itself, the same work the encoders already
|
||||||
|
perform. Most of the documentation can come from nowhere else, and it
|
||||||
|
is written as that knowledge is produced during development. It is
|
||||||
|
verified the way the code is verified: an encoding documented here is
|
||||||
|
one that differential tests against `go tool asm` confirm
|
||||||
|
byte-for-byte, and a format field documented here is one the linker
|
||||||
|
demonstrably reads. The work has begun: [docs/GOOBJ.md](docs/GOOBJ.md)
|
||||||
|
specifies the object file format completely, and
|
||||||
|
[docs/asm/README.md](docs/asm/README.md) opens the language reference
|
||||||
|
with its common core. The per-architecture pages follow.
|
||||||
|
|
||||||
## Direction
|
## Direction
|
||||||
|
|
||||||
The plan, in the order it is being worked:
|
The plan, in the order it is being worked:
|
||||||
@@ -178,9 +221,11 @@ The plan, in the order it is being worked:
|
|||||||
toolchain itself does not support; through ELF, Plan 9 assembly becomes
|
toolchain itself does not support; through ELF, Plan 9 assembly becomes
|
||||||
usable outside Go entirely.
|
usable outside Go entirely.
|
||||||
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
|
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
|
||||||
architectures and is where the binary builds. FreeBSD follows: the
|
architectures and is where the binary builds. FreeBSD follows on amd64,
|
||||||
JIT's executable-memory mapping and the ptrace debugger layer are the
|
arm64 and riscv64: the JIT's executable-memory mapping and the ptrace
|
||||||
two pieces of porting work. Other unix systems may follow those two.
|
debugger layer are ported (the debugger's live validation awaits a
|
||||||
|
FreeBSD machine, as the validation status states). Other unix systems
|
||||||
|
may follow those two.
|
||||||
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
|
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
|
||||||
No others are planned.
|
No others are planned.
|
||||||
|
|
||||||
@@ -188,11 +233,11 @@ The plan, in the order it is being worked:
|
|||||||
|
|
||||||
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
|
||||||
linux/loong64 are on the
|
linux/loong64 are on the
|
||||||
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
|
[releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
|
||||||
From source (Go 1.27.1):
|
From source (Go 1.27.1):
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
|
go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
|
||||||
```
|
```
|
||||||
|
|
||||||
Or from a repository checkout:
|
Or from a repository checkout:
|
||||||
@@ -281,6 +326,8 @@ recipe.
|
|||||||
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
|
~/.local/share/man (MANDIR overrides); `just uninstall-man` removes
|
||||||
them
|
them
|
||||||
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
||||||
|
- [docs/GOOBJ.md](docs/GOOBJ.md): the GOOBJ object file format specification
|
||||||
|
- [docs/asm/](docs/asm/README.md): the Plan 9 assembly language reference
|
||||||
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
- [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md): development setup and recipes
|
||||||
- [CHANGELOG.md](CHANGELOG.md): release history
|
- [CHANGELOG.md](CHANGELOG.md): release history
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ releases do not receive them.
|
|||||||
|
|
||||||
| Version | Supported |
|
| Version | Supported |
|
||||||
|---|---|
|
|---|---|
|
||||||
| 0.34.0 | yes |
|
| 0.35.0 | yes |
|
||||||
| older releases | no |
|
| older releases | no |
|
||||||
|
|
||||||
## Reporting a vulnerability
|
## Reporting a vulnerability
|
||||||
|
|||||||
+105
-7
@@ -5,9 +5,13 @@
|
|||||||
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
|
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
|
||||||
// complete set of mnemonics it accepts for each architecture in
|
// complete set of mnemonics it accepts for each architecture in
|
||||||
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
|
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
|
||||||
// names so gasm-devkit supports every instruction the real assembler does,
|
// names so gasm-sdk supports every instruction the real assembler does,
|
||||||
// with no hand-maintained (and therefore inevitably incomplete) lists.
|
// with no hand-maintained (and therefore inevitably incomplete) lists.
|
||||||
//
|
//
|
||||||
|
// The same data feeds the generated instruction appendices of the assembly
|
||||||
|
// language reference, docs/asm/INSTRUCTIONS-<ARCH>.md, so that the reference
|
||||||
|
// cannot drift from the tables it documents.
|
||||||
|
//
|
||||||
// Usage (via the justfile):
|
// Usage (via the justfile):
|
||||||
//
|
//
|
||||||
// just gen
|
// just gen
|
||||||
@@ -26,9 +30,12 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
|
// archDirs maps a gasm-sdk architecture name to its obj sub-directory.
|
||||||
var archDirs = []struct {
|
var archDirs = []struct {
|
||||||
arch string
|
arch string
|
||||||
sub string
|
sub string
|
||||||
@@ -39,11 +46,30 @@ var archDirs = []struct {
|
|||||||
{"loong64", "loong64"},
|
{"loong64", "loong64"},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// docPages maps an architecture to its generated appendix in the language
|
||||||
|
// reference. The amd64 page carries a per-mnemonic encodability column,
|
||||||
|
// decided by asm.Encodable, which mirrors the encoder's own dispatch; the
|
||||||
|
// other targets have no single cheap predicate, so their pages carry the
|
||||||
|
// inventory and point at the live measurement instead.
|
||||||
|
var docPages = []struct {
|
||||||
|
arch arch.Arch
|
||||||
|
title string
|
||||||
|
file string
|
||||||
|
anames string
|
||||||
|
encodable bool
|
||||||
|
}{
|
||||||
|
{arch.AMD64, "AMD64", "INSTRUCTIONS-AMD64.md", "cmd/internal/obj/x86/anames.go", true},
|
||||||
|
{arch.ARM64, "ARM64", "INSTRUCTIONS-ARM64.md", "cmd/internal/obj/arm64/anames.go", false},
|
||||||
|
{arch.RISCV, "RISC-V 64", "INSTRUCTIONS-RISCV64.md", "cmd/internal/obj/riscv/anames.go", false},
|
||||||
|
{arch.LOONG64, "LoongArch 64", "INSTRUCTIONS-LOONG64.md", "cmd/internal/obj/loong64/anames.go", false},
|
||||||
|
}
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
goroot := strings.TrimSpace(runGoEnvGOROOT())
|
goroot := strings.TrimSpace(runGoEnvGOROOT())
|
||||||
if goroot == "" {
|
if goroot == "" {
|
||||||
fatal("could not determine GOROOT")
|
fatal("could not determine GOROOT")
|
||||||
}
|
}
|
||||||
|
version := strings.TrimSpace(runGoEnv("GOVERSION"))
|
||||||
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
|
// The common opcodes shared by every architecture (RET, JMP, NOP, CALL,
|
||||||
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
|
// TEXT, FUNCDATA, …) live in cmd/internal/obj/util.go.
|
||||||
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
|
commonPath := filepath.Join(goroot, "src", "cmd", "internal", "obj", "util.go")
|
||||||
@@ -57,16 +83,24 @@ func main() {
|
|||||||
}
|
}
|
||||||
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
|
fmt.Printf("%-8s %4d instructions -> arch/common_gen.go\n", "common", len(common))
|
||||||
|
|
||||||
|
names := map[string][]string{}
|
||||||
for _, a := range archDirs {
|
for _, a := range archDirs {
|
||||||
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
|
path := filepath.Join(goroot, "src", "cmd", "internal", "obj", a.sub, "anames.go")
|
||||||
names, err := extractInstrs(path)
|
names[a.arch], err = extractInstrs(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fatal("extract %s: %v", a.arch, err)
|
fatal("extract %s: %v", a.arch, err)
|
||||||
}
|
}
|
||||||
if err := writeGen(a.arch, a.sub, names); err != nil {
|
if err := writeGen(a.arch, a.sub, names[a.arch]); err != nil {
|
||||||
fatal("write %s: %v", a.arch, err)
|
fatal("write %s: %v", a.arch, err)
|
||||||
}
|
}
|
||||||
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names), a.arch)
|
fmt.Printf("%-8s %4d instructions -> arch/%s_gen.go\n", a.arch, len(names[a.arch]), a.arch)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, p := range docPages {
|
||||||
|
if err := writeDocPage(p.arch, p.title, p.file, p.anames, version, p.encodable); err != nil {
|
||||||
|
fatal("write %s: %v", p.file, err)
|
||||||
|
}
|
||||||
|
fmt.Printf("%-8s -> docs/asm/%s\n", p.arch, p.file)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -85,7 +119,7 @@ func filterCommon(names []string) []string {
|
|||||||
// writeCommon emits arch/common_gen.go.
|
// writeCommon emits arch/common_gen.go.
|
||||||
func writeCommon(names []string) error {
|
func writeCommon(names []string) error {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
|
||||||
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
|
||||||
b.WriteString("//\n")
|
b.WriteString("//\n")
|
||||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||||
@@ -156,7 +190,7 @@ func stringLit(elt ast.Expr) string {
|
|||||||
// writeGen emits arch/<arch>_gen.go.
|
// writeGen emits arch/<arch>_gen.go.
|
||||||
func writeGen(arch, sub string, names []string) error {
|
func writeGen(arch, sub string, names []string) error {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
|
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
|
||||||
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
|
||||||
b.WriteString("//\n")
|
b.WriteString("//\n")
|
||||||
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
|
||||||
@@ -172,6 +206,61 @@ func writeGen(arch, sub string, names []string) error {
|
|||||||
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
|
return os.WriteFile(filepath.Join("arch", arch+"_gen.go"), []byte(b.String()), 0o644)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// writeDocPage emits docs/asm/<file>, the generated instruction appendix of
|
||||||
|
// the language reference for one architecture: every mnemonic the toolchain
|
||||||
|
// accepts, with the curated summary where the architecture table carries one
|
||||||
|
// and, on amd64, a per-mnemonic encodability column.
|
||||||
|
func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bool) error {
|
||||||
|
table := arch.ForArch(a)
|
||||||
|
instrs := table.Instructions()
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
b.WriteString("# " + title + ": instruction inventory\n\n")
|
||||||
|
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
|
||||||
|
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
|
||||||
|
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
|
||||||
|
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
|
||||||
|
b.WriteString("and a name present here may still be one gasm's encoder cannot emit yet.\n\n")
|
||||||
|
|
||||||
|
encodableCount := 0
|
||||||
|
if encodable {
|
||||||
|
b.WriteString("The `gasm encodes` column reports whether gasm's encoder can emit the\n")
|
||||||
|
b.WriteString("mnemonic today; the gap is the encoder backlog, measured live by\n")
|
||||||
|
b.WriteString("`gasm audit-instructions`.\n\n")
|
||||||
|
b.WriteString("| Mnemonic | gasm encodes | Notes |\n")
|
||||||
|
b.WriteString("|---|---|---|\n")
|
||||||
|
for _, in := range instrs {
|
||||||
|
ok := asm.Encodable(in.Name)
|
||||||
|
if ok {
|
||||||
|
encodableCount++
|
||||||
|
}
|
||||||
|
b.WriteString("| `" + in.Name + "` | " + yesNo(ok) + " | " + in.Summary + " |\n")
|
||||||
|
}
|
||||||
|
b.WriteString("\n")
|
||||||
|
fmt.Fprintf(&b, "Recognised: %d mnemonics. gasm encodes: %d.\n", len(instrs), encodableCount)
|
||||||
|
} else {
|
||||||
|
b.WriteString("The inventory carries no per-mnemonic encoder column: on this target\n")
|
||||||
|
b.WriteString("encodability is decided per operand shape, and the live measured\n")
|
||||||
|
b.WriteString("coverage is reported by `gasm audit-instructions`.\n\n")
|
||||||
|
b.WriteString("| Mnemonic | Notes |\n")
|
||||||
|
b.WriteString("|---|---|\n")
|
||||||
|
for _, in := range instrs {
|
||||||
|
b.WriteString("| `" + in.Name + "` | " + in.Summary + " |\n")
|
||||||
|
}
|
||||||
|
b.WriteString("\n")
|
||||||
|
fmt.Fprintf(&b, "Recognised: %d mnemonics.\n", len(instrs))
|
||||||
|
}
|
||||||
|
return os.WriteFile(filepath.Join("docs", "asm", file), []byte(b.String()), 0o644)
|
||||||
|
}
|
||||||
|
|
||||||
|
// yesNo renders a boolean as the word the appendix tables use.
|
||||||
|
func yesNo(v bool) string {
|
||||||
|
if v {
|
||||||
|
return "yes"
|
||||||
|
}
|
||||||
|
return "no"
|
||||||
|
}
|
||||||
|
|
||||||
func runGoEnvGOROOT() string {
|
func runGoEnvGOROOT() string {
|
||||||
out, err := exec.Command("go", "env", "GOROOT").Output()
|
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -180,6 +269,15 @@ func runGoEnvGOROOT() string {
|
|||||||
return string(out)
|
return string(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// runGoEnv runs `go env` for a single variable.
|
||||||
|
func runGoEnv(name string) string {
|
||||||
|
out, err := exec.Command("go", "env", name).Output()
|
||||||
|
if err != nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return string(out)
|
||||||
|
}
|
||||||
|
|
||||||
func fatal(format string, args ...any) {
|
func fatal(format string, args ...any) {
|
||||||
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
|
fmt.Fprintf(os.Stderr, "gen: "+format+"\n", args...)
|
||||||
os.Exit(1)
|
os.Exit(1)
|
||||||
|
|||||||
@@ -70,6 +70,10 @@ func amd64Registers() []Register {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
add(fmt.Sprintf("K%d", i), Mask, "AVX-512 mask register")
|
||||||
}
|
}
|
||||||
|
// x87 stack registers (FMOVD and the other x87 moves).
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
add(fmt.Sprintf("F%d", i), Float, "x87 stack register")
|
||||||
|
}
|
||||||
return regs
|
return regs
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
|||||||
@@ -33,6 +33,7 @@ func arm64Registers() []Register {
|
|||||||
for i := 0; i <= 30; i++ {
|
for i := 0; i <= 30; i++ {
|
||||||
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
add(fmt.Sprintf("R%d", i), GPR, "64-bit general-purpose register")
|
||||||
}
|
}
|
||||||
|
add("R18_PLATFORM", GPR, "R18 under its toolchain-reserved Windows name (an alias of R18)")
|
||||||
add("ZR", Special, "zero register (reads as 0)")
|
add("ZR", Special, "zero register (reads as 0)")
|
||||||
add("SP", Special, "stack pointer")
|
add("SP", Special, "stack pointer")
|
||||||
add("LR", Special, "link register (alias of R30)")
|
add("LR", Special, "link register (alias of R30)")
|
||||||
@@ -150,6 +151,33 @@ func arm64Curated() []Instr {
|
|||||||
t = append(t, i(op, "Atomic memory operation"))
|
t = append(t, i(op, "Atomic memory operation"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Register-pair loads and stores.
|
||||||
|
for _, op := range []string{"LDP", "STP", "LDPW", "STPW", "FLDPD", "FSTPD"} {
|
||||||
|
t = append(t, ic(op, "Register-pair load or store", 2, 2))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cache maintenance and prefetch.
|
||||||
|
t = append(t, i("DC", "Data cache maintenance"))
|
||||||
|
t = append(t, i("PRFM", "Memory prefetch"))
|
||||||
|
for _, op := range []string{"LDADDAL", "LDCLRAL", "LDORAL", "SWPAL"} {
|
||||||
|
t = append(t, i(op, "Atomic memory operation with acquire and release semantics"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cryptographic extensions.
|
||||||
|
for _, op := range []string{"AESE", "AESD", "AESMC", "AESIMC"} {
|
||||||
|
t = append(t, i(op, "AES round"))
|
||||||
|
}
|
||||||
|
for _, op := range []string{
|
||||||
|
"SHA1C", "SHA1P", "SHA1M", "SHA1H", "SHA1SU0", "SHA1SU1",
|
||||||
|
"SHA256H", "SHA256H2", "SHA256SU0", "SHA256SU1",
|
||||||
|
"SHA512H", "SHA512H2", "SHA512SU0", "SHA512SU1",
|
||||||
|
} {
|
||||||
|
t = append(t, i(op, "SHA round"))
|
||||||
|
}
|
||||||
|
for _, op := range []string{"VEOR3", "VBCAX", "VXAR", "VRAX1"} {
|
||||||
|
t = append(t, i(op, "Three-way XOR / rotate crypto vector operation"))
|
||||||
|
}
|
||||||
|
|
||||||
// Floating-point scalar.
|
// Floating-point scalar.
|
||||||
for _, op := range []string{
|
for _, op := range []string{
|
||||||
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
"FADD", "FSUB", "FMUL", "FDIV", "FNEG", "FABS", "FSQRT", "FMIN", "FMAX",
|
||||||
|
|||||||
+108
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
@@ -364,6 +364,8 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"REVW",
|
"REVW",
|
||||||
"ROR",
|
"ROR",
|
||||||
"RORW",
|
"RORW",
|
||||||
|
"RPRFM",
|
||||||
|
"SB",
|
||||||
"SBC",
|
"SBC",
|
||||||
"SBCS",
|
"SBCS",
|
||||||
"SBCSW",
|
"SBCSW",
|
||||||
@@ -477,23 +479,68 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"UXTH",
|
"UXTH",
|
||||||
"UXTHW",
|
"UXTHW",
|
||||||
"UXTW",
|
"UXTW",
|
||||||
|
"VABS",
|
||||||
"VADD",
|
"VADD",
|
||||||
"VADDP",
|
"VADDP",
|
||||||
"VADDV",
|
"VADDV",
|
||||||
"VAND",
|
"VAND",
|
||||||
"VBCAX",
|
"VBCAX",
|
||||||
|
"VBIC",
|
||||||
"VBIF",
|
"VBIF",
|
||||||
"VBIT",
|
"VBIT",
|
||||||
"VBSL",
|
"VBSL",
|
||||||
|
"VCLS",
|
||||||
|
"VCLZ",
|
||||||
"VCMEQ",
|
"VCMEQ",
|
||||||
|
"VCMGE",
|
||||||
|
"VCMGT",
|
||||||
|
"VCMHI",
|
||||||
|
"VCMHS",
|
||||||
|
"VCMLE",
|
||||||
|
"VCMLT",
|
||||||
"VCMTST",
|
"VCMTST",
|
||||||
"VCNT",
|
"VCNT",
|
||||||
"VDUP",
|
"VDUP",
|
||||||
"VEOR",
|
"VEOR",
|
||||||
"VEOR3",
|
"VEOR3",
|
||||||
"VEXT",
|
"VEXT",
|
||||||
|
"VFABS",
|
||||||
|
"VFADD",
|
||||||
|
"VFADDP",
|
||||||
|
"VFCMEQ",
|
||||||
|
"VFCMGE",
|
||||||
|
"VFCMGT",
|
||||||
|
"VFCMLE",
|
||||||
|
"VFCMLT",
|
||||||
|
"VFCVTL",
|
||||||
|
"VFCVTL2",
|
||||||
|
"VFCVTN",
|
||||||
|
"VFCVTN2",
|
||||||
|
"VFCVTZS",
|
||||||
|
"VFCVTZU",
|
||||||
|
"VFDIV",
|
||||||
|
"VFMAX",
|
||||||
|
"VFMAXNM",
|
||||||
|
"VFMAXNMP",
|
||||||
|
"VFMAXNMV",
|
||||||
|
"VFMAXP",
|
||||||
|
"VFMAXV",
|
||||||
|
"VFMIN",
|
||||||
|
"VFMINNM",
|
||||||
|
"VFMINNMP",
|
||||||
|
"VFMINNMV",
|
||||||
|
"VFMINP",
|
||||||
|
"VFMINV",
|
||||||
"VFMLA",
|
"VFMLA",
|
||||||
"VFMLS",
|
"VFMLS",
|
||||||
|
"VFMUL",
|
||||||
|
"VFNEG",
|
||||||
|
"VFRINTM",
|
||||||
|
"VFRINTN",
|
||||||
|
"VFRINTP",
|
||||||
|
"VFRINTZ",
|
||||||
|
"VFSQRT",
|
||||||
|
"VFSUB",
|
||||||
"VLD1",
|
"VLD1",
|
||||||
"VLD1R",
|
"VLD1R",
|
||||||
"VLD2",
|
"VLD2",
|
||||||
@@ -502,11 +549,17 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VLD3R",
|
"VLD3R",
|
||||||
"VLD4",
|
"VLD4",
|
||||||
"VLD4R",
|
"VLD4R",
|
||||||
|
"VMLA",
|
||||||
|
"VMLS",
|
||||||
"VMOV",
|
"VMOV",
|
||||||
"VMOVD",
|
"VMOVD",
|
||||||
"VMOVI",
|
"VMOVI",
|
||||||
"VMOVQ",
|
"VMOVQ",
|
||||||
"VMOVS",
|
"VMOVS",
|
||||||
|
"VMUL",
|
||||||
|
"VNEG",
|
||||||
|
"VNOT",
|
||||||
|
"VORN",
|
||||||
"VORR",
|
"VORR",
|
||||||
"VPMULL",
|
"VPMULL",
|
||||||
"VPMULL2",
|
"VPMULL2",
|
||||||
@@ -515,14 +568,47 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VREV16",
|
"VREV16",
|
||||||
"VREV32",
|
"VREV32",
|
||||||
"VREV64",
|
"VREV64",
|
||||||
|
"VSCVTF",
|
||||||
|
"VSHADD",
|
||||||
"VSHL",
|
"VSHL",
|
||||||
|
"VSHRN",
|
||||||
|
"VSHRN2",
|
||||||
"VSLI",
|
"VSLI",
|
||||||
|
"VSMAX",
|
||||||
|
"VSMAXP",
|
||||||
|
"VSMAXV",
|
||||||
|
"VSMIN",
|
||||||
|
"VSMINP",
|
||||||
|
"VSMINV",
|
||||||
|
"VSMLAL",
|
||||||
|
"VSMLAL2",
|
||||||
|
"VSMLSL",
|
||||||
|
"VSMLSL2",
|
||||||
|
"VSMULL",
|
||||||
|
"VSMULL2",
|
||||||
|
"VSQABS",
|
||||||
|
"VSQADD",
|
||||||
|
"VSQNEG",
|
||||||
|
"VSQSHL",
|
||||||
|
"VSQSUB",
|
||||||
|
"VSQXTN",
|
||||||
|
"VSQXTN2",
|
||||||
|
"VSQXTUN",
|
||||||
|
"VSQXTUN2",
|
||||||
|
"VSRHADD",
|
||||||
"VSRI",
|
"VSRI",
|
||||||
|
"VSRSHR",
|
||||||
|
"VSSHL",
|
||||||
|
"VSSHLL",
|
||||||
|
"VSSHLL2",
|
||||||
|
"VSSHR",
|
||||||
"VST1",
|
"VST1",
|
||||||
"VST2",
|
"VST2",
|
||||||
"VST3",
|
"VST3",
|
||||||
"VST4",
|
"VST4",
|
||||||
"VSUB",
|
"VSUB",
|
||||||
|
"VSXTL",
|
||||||
|
"VSXTL2",
|
||||||
"VTBL",
|
"VTBL",
|
||||||
"VTBX",
|
"VTBX",
|
||||||
"VTRN1",
|
"VTRN1",
|
||||||
@@ -530,8 +616,27 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VUADDLV",
|
"VUADDLV",
|
||||||
"VUADDW",
|
"VUADDW",
|
||||||
"VUADDW2",
|
"VUADDW2",
|
||||||
|
"VUCVTF",
|
||||||
|
"VUHADD",
|
||||||
"VUMAX",
|
"VUMAX",
|
||||||
|
"VUMAXP",
|
||||||
|
"VUMAXV",
|
||||||
"VUMIN",
|
"VUMIN",
|
||||||
|
"VUMINP",
|
||||||
|
"VUMINV",
|
||||||
|
"VUMLAL",
|
||||||
|
"VUMLAL2",
|
||||||
|
"VUMLSL",
|
||||||
|
"VUMLSL2",
|
||||||
|
"VUMULL",
|
||||||
|
"VUMULL2",
|
||||||
|
"VUQADD",
|
||||||
|
"VUQSHL",
|
||||||
|
"VUQSUB",
|
||||||
|
"VUQXTN",
|
||||||
|
"VUQXTN2",
|
||||||
|
"VURHADD",
|
||||||
|
"VUSHL",
|
||||||
"VUSHLL",
|
"VUSHLL",
|
||||||
"VUSHLL2",
|
"VUSHLL2",
|
||||||
"VUSHR",
|
"VUSHR",
|
||||||
@@ -541,6 +646,8 @@ var arm64GeneratedInstrs = []string{
|
|||||||
"VUZP1",
|
"VUZP1",
|
||||||
"VUZP2",
|
"VUZP2",
|
||||||
"VXAR",
|
"VXAR",
|
||||||
|
"VXTN",
|
||||||
|
"VXTN2",
|
||||||
"VZIP1",
|
"VZIP1",
|
||||||
"VZIP2",
|
"VZIP2",
|
||||||
"WFE",
|
"WFE",
|
||||||
|
|||||||
+1
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
// Source: cmd/internal/obj/util.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
|||||||
+10
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
@@ -152,6 +152,8 @@ var loong64GeneratedInstrs = []string{
|
|||||||
"FNMADDF",
|
"FNMADDF",
|
||||||
"FNMSUBD",
|
"FNMSUBD",
|
||||||
"FNMSUBF",
|
"FNMSUBF",
|
||||||
|
"FRINTD",
|
||||||
|
"FRINTF",
|
||||||
"FSCALEBD",
|
"FSCALEBD",
|
||||||
"FSCALEBF",
|
"FSCALEBF",
|
||||||
"FSEL",
|
"FSEL",
|
||||||
@@ -177,7 +179,10 @@ var loong64GeneratedInstrs = []string{
|
|||||||
"FTINTWF",
|
"FTINTWF",
|
||||||
"JIRL",
|
"JIRL",
|
||||||
"LL",
|
"LL",
|
||||||
|
"LLACQV",
|
||||||
|
"LLACQW",
|
||||||
"LLV",
|
"LLV",
|
||||||
|
"LLW",
|
||||||
"LU12IW",
|
"LU12IW",
|
||||||
"LU32ID",
|
"LU32ID",
|
||||||
"LU52ID",
|
"LU52ID",
|
||||||
@@ -248,7 +253,11 @@ var loong64GeneratedInstrs = []string{
|
|||||||
"ROTR",
|
"ROTR",
|
||||||
"ROTRV",
|
"ROTRV",
|
||||||
"SC",
|
"SC",
|
||||||
|
"SCQ",
|
||||||
|
"SCRELV",
|
||||||
|
"SCRELW",
|
||||||
"SCV",
|
"SCV",
|
||||||
|
"SCW",
|
||||||
"SGT",
|
"SGT",
|
||||||
"SGTU",
|
"SGTU",
|
||||||
"SLL",
|
"SLL",
|
||||||
|
|||||||
+32
-1
@@ -1,4 +1,4 @@
|
|||||||
// Code generated by gasm-devkit _gen; DO NOT EDIT.
|
// Code generated by gasm-sdk _gen; DO NOT EDIT.
|
||||||
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
|
||||||
//
|
//
|
||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
@@ -81,6 +81,9 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"CLD",
|
"CLD",
|
||||||
"CLDSP",
|
"CLDSP",
|
||||||
"CLI",
|
"CLI",
|
||||||
|
"CLMUL",
|
||||||
|
"CLMULH",
|
||||||
|
"CLMULR",
|
||||||
"CLUI",
|
"CLUI",
|
||||||
"CLW",
|
"CLW",
|
||||||
"CLWSP",
|
"CLWSP",
|
||||||
@@ -95,13 +98,20 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"CSDSP",
|
"CSDSP",
|
||||||
"CSLLI",
|
"CSLLI",
|
||||||
"CSRAI",
|
"CSRAI",
|
||||||
|
"CSRC",
|
||||||
|
"CSRCI",
|
||||||
"CSRLI",
|
"CSRLI",
|
||||||
|
"CSRR",
|
||||||
"CSRRC",
|
"CSRRC",
|
||||||
"CSRRCI",
|
"CSRRCI",
|
||||||
"CSRRS",
|
"CSRRS",
|
||||||
"CSRRSI",
|
"CSRRSI",
|
||||||
"CSRRW",
|
"CSRRW",
|
||||||
"CSRRWI",
|
"CSRRWI",
|
||||||
|
"CSRS",
|
||||||
|
"CSRSI",
|
||||||
|
"CSRW",
|
||||||
|
"CSRWI",
|
||||||
"CSUB",
|
"CSUB",
|
||||||
"CSUBW",
|
"CSUBW",
|
||||||
"CSW",
|
"CSW",
|
||||||
@@ -259,6 +269,7 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"ORCB",
|
"ORCB",
|
||||||
"ORI",
|
"ORI",
|
||||||
"ORN",
|
"ORN",
|
||||||
|
"PAUSE",
|
||||||
"RDCYCLE",
|
"RDCYCLE",
|
||||||
"RDINSTRET",
|
"RDINSTRET",
|
||||||
"RDTIME",
|
"RDTIME",
|
||||||
@@ -322,6 +333,8 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VADDVI",
|
"VADDVI",
|
||||||
"VADDVV",
|
"VADDVV",
|
||||||
"VADDVX",
|
"VADDVX",
|
||||||
|
"VANDNVV",
|
||||||
|
"VANDNVX",
|
||||||
"VANDVI",
|
"VANDVI",
|
||||||
"VANDVV",
|
"VANDVV",
|
||||||
"VANDVX",
|
"VANDVX",
|
||||||
@@ -329,8 +342,17 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VASUBUVX",
|
"VASUBUVX",
|
||||||
"VASUBVV",
|
"VASUBVV",
|
||||||
"VASUBVX",
|
"VASUBVX",
|
||||||
|
"VBREV8V",
|
||||||
|
"VBREVV",
|
||||||
|
"VCLMULHVV",
|
||||||
|
"VCLMULHVX",
|
||||||
|
"VCLMULVV",
|
||||||
|
"VCLMULVX",
|
||||||
|
"VCLZV",
|
||||||
"VCOMPRESSVM",
|
"VCOMPRESSVM",
|
||||||
"VCPOPM",
|
"VCPOPM",
|
||||||
|
"VCPOPV",
|
||||||
|
"VCTZV",
|
||||||
"VDIVUVV",
|
"VDIVUVV",
|
||||||
"VDIVUVX",
|
"VDIVUVX",
|
||||||
"VDIVVV",
|
"VDIVVV",
|
||||||
@@ -743,10 +765,16 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VREMUVX",
|
"VREMUVX",
|
||||||
"VREMVV",
|
"VREMVV",
|
||||||
"VREMVX",
|
"VREMVX",
|
||||||
|
"VREV8V",
|
||||||
"VRGATHEREI16VV",
|
"VRGATHEREI16VV",
|
||||||
"VRGATHERVI",
|
"VRGATHERVI",
|
||||||
"VRGATHERVV",
|
"VRGATHERVV",
|
||||||
"VRGATHERVX",
|
"VRGATHERVX",
|
||||||
|
"VROLVV",
|
||||||
|
"VROLVX",
|
||||||
|
"VRORVI",
|
||||||
|
"VRORVV",
|
||||||
|
"VRORVX",
|
||||||
"VRSUBVI",
|
"VRSUBVI",
|
||||||
"VRSUBVX",
|
"VRSUBVX",
|
||||||
"VS1RV",
|
"VS1RV",
|
||||||
@@ -950,6 +978,9 @@ var riscvGeneratedInstrs = []string{
|
|||||||
"VWMULVX",
|
"VWMULVX",
|
||||||
"VWREDSUMUVS",
|
"VWREDSUMUVS",
|
||||||
"VWREDSUMVS",
|
"VWREDSUMVS",
|
||||||
|
"VWSLLVI",
|
||||||
|
"VWSLLVV",
|
||||||
|
"VWSLLVX",
|
||||||
"VWSUBUVV",
|
"VWSUBUVV",
|
||||||
"VWSUBUVX",
|
"VWSUBUVX",
|
||||||
"VWSUBUWV",
|
"VWSUBUWV",
|
||||||
|
|||||||
+102
-1
@@ -11,7 +11,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
|
||||||
@@ -245,3 +245,104 @@ func main() {
|
|||||||
t.Error("binary does not contain expected symbol")
|
t.Error("binary does not contain expected symbol")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestGOObjectAARCH64DataSymbolLink does for symbol-valued DATA fields what
|
||||||
|
// the rt0 files do ("DATA _rt0…lib+0(SB)/8, $_rt0…lib(SB)"): the gasm object
|
||||||
|
// carries an R_ADDR against the file's own TEXT symbol, the toolchain links
|
||||||
|
// it, and the binary is checked for the symbol (no arm64 host to run it).
|
||||||
|
func TestGOObjectAARCH64DataSymbolLink(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
asmSrc := `#include "textflag.h"
|
||||||
|
GLOBL entry(SB), NOPTR, $8
|
||||||
|
DATA entry+0(SB)/8, $·keepme(SB)
|
||||||
|
|
||||||
|
TEXT ·keepme(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·entryptr(SB), NOSPLIT, $0-8
|
||||||
|
MOVD entry+0(SB), R4
|
||||||
|
MOVD R4, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_arm64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mainSrc := `package main
|
||||||
|
|
||||||
|
func keepme()
|
||||||
|
func entryptr() uintptr
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
if entryptr() == 0 {
|
||||||
|
panic("the entry word is empty")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module a64dlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
build.Env = append(os.Environ(), "GOARCH=arm64")
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
var work, linkLine, asmObj string
|
||||||
|
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
|
||||||
|
asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if work == "" || asmObj == "" || linkLine == "" {
|
||||||
|
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(work)
|
||||||
|
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||||
|
|
||||||
|
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse("main_arm64.s", string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObjectAARCH64: %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||||
|
t.Fatalf("write gasm object: %v", err)
|
||||||
|
}
|
||||||
|
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||||
|
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
|
||||||
|
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(string(binData), "keepme") {
|
||||||
|
t.Error("binary does not contain the keepme symbol")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+2787
-98
File diff suppressed because it is too large
Load Diff
+758
-27
@@ -27,7 +27,14 @@ package asm
|
|||||||
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
||||||
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
||||||
|
|
||||||
import "maps"
|
import (
|
||||||
|
"maps"
|
||||||
|
"math/bits"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
|
)
|
||||||
|
|
||||||
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
||||||
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
|
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
|
||||||
@@ -72,6 +79,11 @@ func arm64RegNum(name string) int {
|
|||||||
return 17
|
return 17
|
||||||
case "R18":
|
case "R18":
|
||||||
return 18
|
return 18
|
||||||
|
case "R18_PLATFORM":
|
||||||
|
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
|
||||||
|
// cmd/asm/internal/arch so assembly cannot use it by accident, and
|
||||||
|
// sys_windows_arm64.s references it only through this name.
|
||||||
|
return 18
|
||||||
case "R19":
|
case "R19":
|
||||||
return 19
|
return 19
|
||||||
case "R20":
|
case "R20":
|
||||||
@@ -153,6 +165,67 @@ func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
|
|||||||
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- logical immediate ----
|
||||||
|
|
||||||
|
// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the
|
||||||
|
// given lane width (32 or 64): it returns the N, immr and imms fields of the
|
||||||
|
// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's
|
||||||
|
// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest
|
||||||
|
// repeating element, find the run of ones and its rotation. ok is false when
|
||||||
|
// v is not expressible (all zeros, all ones, or not a single cyclic run).
|
||||||
|
func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) {
|
||||||
|
u := uint64(v)
|
||||||
|
if width == 32 {
|
||||||
|
u &= 0xFFFFFFFF
|
||||||
|
}
|
||||||
|
size := uint64(width)
|
||||||
|
mask := ^uint64(0)
|
||||||
|
if size < 64 {
|
||||||
|
mask = uint64(1)<<size - 1
|
||||||
|
}
|
||||||
|
u &= mask
|
||||||
|
// All zeros and all ones are MOV territory, not bitmask immediates.
|
||||||
|
if u == 0 || u == mask {
|
||||||
|
return 0, 0, 0, false
|
||||||
|
}
|
||||||
|
// Shrink to the smallest repeating element.
|
||||||
|
for size > 2 {
|
||||||
|
half := size / 2
|
||||||
|
hm := uint64(1)<<half - 1
|
||||||
|
if u&hm == u>>half&hm {
|
||||||
|
size = half
|
||||||
|
u &= hm
|
||||||
|
} else {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ones := bits.OnesCount64(u)
|
||||||
|
// Find the right-rotation that lays the ones out contiguously at the
|
||||||
|
// bottom of the element; the hardware applies the inverse rotation.
|
||||||
|
em := uint64(1)<<size - 1
|
||||||
|
expected := uint64(1)<<ones - 1
|
||||||
|
rot := -1
|
||||||
|
for r := 0; r < int(size); r++ {
|
||||||
|
rotated := u>>r | u<<(int(size)-r)
|
||||||
|
if size < 64 {
|
||||||
|
rotated &= em
|
||||||
|
}
|
||||||
|
if rotated == expected {
|
||||||
|
rot = r
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if rot < 0 {
|
||||||
|
return 0, 0, 0, false
|
||||||
|
}
|
||||||
|
if size == 64 {
|
||||||
|
n = 1
|
||||||
|
}
|
||||||
|
immr = uint32((int(size) - rot) % int(size))
|
||||||
|
imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1)
|
||||||
|
return n, immr, imms, true
|
||||||
|
}
|
||||||
|
|
||||||
// ---- load/store (unsigned immediate, scaled) ----
|
// ---- load/store (unsigned immediate, scaled) ----
|
||||||
|
|
||||||
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
||||||
@@ -240,6 +313,8 @@ const (
|
|||||||
a64CondLT = 0xb
|
a64CondLT = 0xb
|
||||||
a64CondGT = 0xc
|
a64CondGT = 0xc
|
||||||
a64CondLE = 0xd
|
a64CondLE = 0xd
|
||||||
|
a64CondAL = 0xe
|
||||||
|
a64CondNV = 0xf
|
||||||
)
|
)
|
||||||
|
|
||||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||||
@@ -260,6 +335,8 @@ var arm64CondMap = map[string]uint32{
|
|||||||
"LT": a64CondLT,
|
"LT": a64CondLT,
|
||||||
"GT": a64CondGT,
|
"GT": a64CondGT,
|
||||||
"LE": a64CondLE,
|
"LE": a64CondLE,
|
||||||
|
"AL": a64CondAL,
|
||||||
|
"NV": a64CondNV,
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- instruction format tags ----
|
// ---- instruction format tags ----
|
||||||
@@ -267,28 +344,47 @@ var arm64CondMap = map[string]uint32{
|
|||||||
type a64Format uint8
|
type a64Format uint8
|
||||||
|
|
||||||
const (
|
const (
|
||||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||||
a64FBranch // unconditional branch (B/BL)
|
a64FBranch // unconditional branch (B/BL)
|
||||||
a64FBranchCond // conditional branch (B.cond)
|
a64FBranchCond // conditional branch (B.cond)
|
||||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||||
a64FADR // ADR/ADRP
|
a64FADR // ADR/ADRP
|
||||||
a64FEXTR // EXTR
|
a64FEXTR // EXTR
|
||||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
|
||||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||||
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
||||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||||
a64FCRC32 // CRC32
|
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
a64FCRC32 // CRC32
|
||||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
||||||
a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL
|
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||||
|
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
||||||
|
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
||||||
|
a64FCondCmp // conditional compare: CCMP, CCMN
|
||||||
|
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
||||||
|
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
||||||
|
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
||||||
|
a64FAcqRel // acquire/release: LDAR family, STLR family
|
||||||
|
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
||||||
|
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
||||||
|
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
||||||
|
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
||||||
|
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
||||||
|
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
||||||
|
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
||||||
|
a64FVTBL // SIMD table lookup: VTBL
|
||||||
|
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
||||||
|
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
||||||
|
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
||||||
|
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
||||||
)
|
)
|
||||||
|
|
||||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
@@ -401,6 +497,16 @@ func init() {
|
|||||||
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||||
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||||
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||||
|
// The widening multiplies: a 64-bit result riding the same layout, the
|
||||||
|
// three-operand forms reading the accumulate register as ZR.
|
||||||
|
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
|
||||||
|
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
|
||||||
|
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
|
||||||
|
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
|
||||||
|
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
|
||||||
|
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
|
||||||
|
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
|
||||||
|
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
|
||||||
|
|
||||||
// ---- move wide ----
|
// ---- move wide ----
|
||||||
// MOVZ/MOVN/MOVK
|
// MOVZ/MOVN/MOVK
|
||||||
@@ -450,6 +556,15 @@ func init() {
|
|||||||
// ---- bitfield ----
|
// ---- bitfield ----
|
||||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||||
|
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
|
||||||
|
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||||
|
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||||
|
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||||
|
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||||
|
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
|
||||||
|
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
|
||||||
|
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
|
||||||
|
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
|
||||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||||
@@ -622,10 +737,626 @@ func init() {
|
|||||||
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
||||||
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
||||||
|
|
||||||
// ---- SIMD basics ----
|
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
|
||||||
a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400}
|
// VSUB, VMUL and every other three-register vector op. ----
|
||||||
a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400}
|
|
||||||
a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00}
|
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
|
||||||
|
dp1 := map[string]uint32{
|
||||||
|
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
||||||
|
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
||||||
|
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
||||||
|
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
|
||||||
|
// the source width.
|
||||||
|
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
|
||||||
|
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
|
||||||
|
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
|
||||||
|
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
|
||||||
|
"REV16W": 0x5ac00400,
|
||||||
|
}
|
||||||
|
for m, op := range dp1 {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
|
||||||
|
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
|
||||||
|
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
|
||||||
|
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
|
||||||
|
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
|
||||||
|
|
||||||
|
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
|
||||||
|
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
|
||||||
|
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
|
||||||
|
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
|
||||||
|
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||||
|
|
||||||
|
// ---- system operations ----
|
||||||
|
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- compare/test and branch ----
|
||||||
|
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
|
||||||
|
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
|
||||||
|
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
|
||||||
|
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
|
||||||
|
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
|
||||||
|
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
|
||||||
|
|
||||||
|
// ---- load/store pair (signed offset) ----
|
||||||
|
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
|
||||||
|
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
|
||||||
|
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
|
||||||
|
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
|
||||||
|
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
|
||||||
|
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
|
||||||
|
|
||||||
|
// ---- acquire/release loads and stores ----
|
||||||
|
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
|
||||||
|
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
|
||||||
|
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
|
||||||
|
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
|
||||||
|
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
|
||||||
|
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
|
||||||
|
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
|
||||||
|
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
|
||||||
|
|
||||||
|
// ---- LSE atomics with acquire and release semantics ----
|
||||||
|
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
|
||||||
|
// LDOR/SWP families leave Rs free for the returned value.
|
||||||
|
lse := map[string]uint32{
|
||||||
|
"CASALD": 0xc8e0fc00,
|
||||||
|
"CASALW": 0x88e0fc00,
|
||||||
|
"LDADDALD": 0xf8e00000,
|
||||||
|
"LDADDALW": 0xb8e00000,
|
||||||
|
"LDCLRALB": 0x38e01000,
|
||||||
|
"LDCLRALW": 0xb8e01000,
|
||||||
|
"LDCLRALD": 0xf8e01000,
|
||||||
|
"LDORALB": 0x38e03000,
|
||||||
|
"LDORALW": 0xb8e03000,
|
||||||
|
"LDORALD": 0xf8e03000,
|
||||||
|
"SWPALB": 0x38e08000,
|
||||||
|
"SWPALW": 0xb8e08000,
|
||||||
|
"SWPALD": 0xf8e08000,
|
||||||
|
}
|
||||||
|
for m, op := range lse {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||||
|
}
|
||||||
|
// The remaining width and ordering spellings of the same shapes, and the
|
||||||
|
// CAS compare-and-swap family, word-verified against go tool asm.
|
||||||
|
lseMore := map[string]uint32{
|
||||||
|
"LDADDAB": 0x38a00000,
|
||||||
|
"LDADDAH": 0x78a00000,
|
||||||
|
"LDADDALB": 0x38e00000,
|
||||||
|
"LDADDALH": 0x78e00000,
|
||||||
|
"LDADDLB": 0x38600000,
|
||||||
|
"LDADDLD": 0xf8600000,
|
||||||
|
"LDADDLH": 0x78600000,
|
||||||
|
"LDADDLW": 0xb8600000,
|
||||||
|
"LDCLRAB": 0x38a01000,
|
||||||
|
"LDCLRAH": 0x78a01000,
|
||||||
|
"LDCLRALH": 0x78e01000,
|
||||||
|
"LDCLRB": 0x38201000,
|
||||||
|
"LDCLRD": 0xf8201000,
|
||||||
|
"LDCLRH": 0x78201000,
|
||||||
|
"LDCLRLB": 0x38601000,
|
||||||
|
"LDCLRLD": 0xf8601000,
|
||||||
|
"LDCLRLH": 0x78601000,
|
||||||
|
"LDCLRLW": 0xb8601000,
|
||||||
|
"LDCLRW": 0xb8201000,
|
||||||
|
"LDEORAB": 0x38a02000,
|
||||||
|
"LDEORAD": 0xf8a02000,
|
||||||
|
"LDEORAH": 0x78a02000,
|
||||||
|
"LDEORALB": 0x38e02000,
|
||||||
|
"LDEORALH": 0x78e02000,
|
||||||
|
"LDEORAW": 0xb8a02000,
|
||||||
|
"LDEORB": 0x38202000,
|
||||||
|
"LDEORD": 0xf8202000,
|
||||||
|
"LDEORH": 0x78202000,
|
||||||
|
"LDEORLB": 0x38602000,
|
||||||
|
"LDEORLD": 0xf8602000,
|
||||||
|
"LDEORLH": 0x78602000,
|
||||||
|
"LDEORLW": 0xb8602000,
|
||||||
|
"LDEORW": 0xb8202000,
|
||||||
|
"LDORAB": 0x38a03000,
|
||||||
|
"LDORAD": 0xf8a03000,
|
||||||
|
"LDORAH": 0x78a03000,
|
||||||
|
"LDORALH": 0x78e03000,
|
||||||
|
"LDORAW": 0xb8a03000,
|
||||||
|
"LDORB": 0x38203000,
|
||||||
|
"LDORD": 0xf8203000,
|
||||||
|
"LDORH": 0x78203000,
|
||||||
|
"LDORLB": 0x38603000,
|
||||||
|
"LDORLD": 0xf8603000,
|
||||||
|
"LDORLH": 0x78603000,
|
||||||
|
"LDORLW": 0xb8603000,
|
||||||
|
"LDORW": 0xb8203000,
|
||||||
|
"SWPAB": 0x38a08000,
|
||||||
|
"SWPAD": 0xf8a08000,
|
||||||
|
"SWPAH": 0x78a08000,
|
||||||
|
"SWPALH": 0x78e08000,
|
||||||
|
"SWPAW": 0xb8a08000,
|
||||||
|
"SWPB": 0x38208000,
|
||||||
|
"SWPH": 0x78208000,
|
||||||
|
"SWPLB": 0x38608000,
|
||||||
|
"SWPLD": 0xf8608000,
|
||||||
|
"SWPLH": 0x78608000,
|
||||||
|
"SWPLW": 0xb8608000,
|
||||||
|
"CASAD": 0xc8e07c00,
|
||||||
|
"CASALB": 0x08e0fc00,
|
||||||
|
"CASLW": 0x88a0fc00,
|
||||||
|
}
|
||||||
|
for m, op := range lseMore {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||||
|
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||||
|
// register preset to ZR (bits 14:10 = 11111).
|
||||||
|
dpsrExtra := map[string]uint32{
|
||||||
|
"ADC": 0x9a000000, "ADCW": 0x1a000000,
|
||||||
|
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
||||||
|
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
||||||
|
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
||||||
|
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
|
||||||
|
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
|
||||||
|
"NGC": 0xda000000, "NGCW": 0x5a000000,
|
||||||
|
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
|
||||||
|
"NEGSW": 0x6b000000,
|
||||||
|
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||||
|
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
||||||
|
}
|
||||||
|
for m, op := range dpsrExtra {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
|
||||||
|
crypto2 := map[string]uint32{
|
||||||
|
"AESD": 0x4e285800, "AESE": 0x4e284800,
|
||||||
|
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
|
||||||
|
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
|
||||||
|
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
|
||||||
|
}
|
||||||
|
for m, op := range crypto2 {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
|
||||||
|
}
|
||||||
|
crypto3 := map[string]uint32{
|
||||||
|
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
|
||||||
|
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
|
||||||
|
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
|
||||||
|
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
|
||||||
|
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
|
||||||
|
}
|
||||||
|
for m, op := range crypto3 {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
|
||||||
|
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
|
||||||
|
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
|
||||||
|
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
|
||||||
|
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
|
||||||
|
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
|
||||||
|
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
|
||||||
|
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
|
||||||
|
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
|
||||||
|
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
|
||||||
|
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||||
|
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||||
|
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||||
|
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
||||||
|
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
||||||
|
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
||||||
|
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
||||||
|
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
||||||
|
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
|
||||||
|
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
|
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
|
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
|
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
||||||
|
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
||||||
|
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
|
||||||
|
// for instructions that exist at a single arrangement and carry that
|
||||||
|
// arrangement's bits inside the base already, the fixed flag.
|
||||||
|
type a64SimdVSpec struct {
|
||||||
|
base uint32
|
||||||
|
arrs uint16
|
||||||
|
fixed bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64Arr names the vector arrangements the encoders deal with, indexed by
|
||||||
|
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
|
||||||
|
// D1 and the 128-bit halves B16, H8, S4, D2.
|
||||||
|
const (
|
||||||
|
a64Arr8B = iota
|
||||||
|
a64Arr16B
|
||||||
|
a64Arr4H
|
||||||
|
a64Arr8H
|
||||||
|
a64Arr2S
|
||||||
|
a64Arr4S
|
||||||
|
a64Arr2D
|
||||||
|
a64ArrD1
|
||||||
|
a64ArrQ1
|
||||||
|
a64ArrCount
|
||||||
|
)
|
||||||
|
|
||||||
|
// a64ArrNames maps an arrangement to its source spelling (element letter
|
||||||
|
// first, as the toolchain writes it).
|
||||||
|
var a64ArrNames = [a64ArrCount]string{
|
||||||
|
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
|
||||||
|
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
|
||||||
|
// unknown.
|
||||||
|
func a64ArrIndex(s string) int {
|
||||||
|
for i, n := range a64ArrNames {
|
||||||
|
if n == s {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
|
||||||
|
// as it appears in element operands such as V13.S[0].
|
||||||
|
func a64ElemLetter(s string) bool {
|
||||||
|
switch s {
|
||||||
|
case "B", "H", "S", "D", "Q":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
||||||
|
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
||||||
|
// the across-vector reductions.
|
||||||
|
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||||
|
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
||||||
|
|
||||||
|
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
||||||
|
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
||||||
|
// and pairwise compares among them. Word-verified against go tool asm.
|
||||||
|
var a64SimdQOnly = map[string]bool{
|
||||||
|
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
|
||||||
|
"VFABS": true, "VFNEG": true, "VFSQRT": true,
|
||||||
|
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
|
||||||
|
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
|
||||||
|
"VFMAXV": true, "VFMAXNMV": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
||||||
|
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
||||||
|
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
||||||
|
var a64ArrBits = [a64ArrCount]uint32{
|
||||||
|
a64Arr8B: 0,
|
||||||
|
a64Arr16B: 1 << 30,
|
||||||
|
a64Arr4H: 1 << 22,
|
||||||
|
a64Arr8H: 1<<30 | 1<<22,
|
||||||
|
a64Arr2S: 1 << 23,
|
||||||
|
a64Arr4S: 1<<30 | 1<<23,
|
||||||
|
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
|
||||||
|
a64ArrD1: 1<<23 | 1<<22,
|
||||||
|
a64ArrQ1: 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdVTable holds the arrangement-aware three-register SIMD
|
||||||
|
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
||||||
|
// word and arrangement bit was read off go tool asm.
|
||||||
|
var a64SimdVTable = map[string]a64SimdVSpec{
|
||||||
|
"VADD": {0x0e208400, 0x7f, false},
|
||||||
|
"VSUB": {0x2e208400, 0x7f, false},
|
||||||
|
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
||||||
|
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
||||||
|
"VEOR": {0x2e201c00, 0x03, false},
|
||||||
|
"VORR": {0x0ea01c00, 0x03, false},
|
||||||
|
"VADDP": {0x0e20bc00, 0x7f, false},
|
||||||
|
"VZIP1": {0x0e003800, 0x7f, false},
|
||||||
|
"VZIP2": {0x0e007800, 0x7f, false},
|
||||||
|
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||||
|
"VCMGE": {0x0e203c00, 0x7f, false},
|
||||||
|
"VCMGT": {0x0e203400, 0x7f, false},
|
||||||
|
"VCMHI": {0x2e203400, 0x7f, false},
|
||||||
|
"VCMHS": {0x2e203c00, 0x7f, false},
|
||||||
|
// FP compares take H, S and D arrangements only (the toolchain rejects
|
||||||
|
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
||||||
|
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
// FP arithmetic shares the same arrangement restriction.
|
||||||
|
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
||||||
|
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
||||||
|
// tool asm.
|
||||||
|
"VBIC": {0x0e601c00, 0x7f, false},
|
||||||
|
"VBIF": {0x2ee01c00, 0x7f, false},
|
||||||
|
"VBIT": {0x6ea01c00, 0x7f, false},
|
||||||
|
"VBSL": {0x6e601c00, 0x7f, false},
|
||||||
|
"VCMTST": {0x0e208c00, 0x7f, false},
|
||||||
|
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VMLA": {0x4ea09400, 0x7f, false},
|
||||||
|
"VMLS": {0x6ea09400, 0x7f, false},
|
||||||
|
"VORN": {0x4ee01c00, 0x7f, false},
|
||||||
|
"VSHADD": {0x4ea00400, 0x7f, false},
|
||||||
|
"VSRHADD": {0x4ea01400, 0x7f, false},
|
||||||
|
"VUHADD": {0x6ea00400, 0x7f, false},
|
||||||
|
"VURHADD": {0x6ea01400, 0x7f, false},
|
||||||
|
"VSMAX": {0x4ea06400, 0x7f, false},
|
||||||
|
"VSMIN": {0x4ea06c00, 0x7f, false},
|
||||||
|
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
||||||
|
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
||||||
|
"VUMAX": {0x2e206400, 0x7f, false},
|
||||||
|
"VUMIN": {0x2e206c00, 0x7f, false},
|
||||||
|
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
||||||
|
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
||||||
|
"VSQADD": {0x4ea00c00, 0x7f, false},
|
||||||
|
"VUQADD": {0x6ea00c00, 0x7f, false},
|
||||||
|
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
||||||
|
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
||||||
|
"VSSHL": {0x4ee04400, 0x7f, false},
|
||||||
|
"VUSHL": {0x6ee04400, 0x7f, false},
|
||||||
|
"VUZP1": {0x0e001800, 0x7f, false},
|
||||||
|
"VUZP2": {0x4ec05800, 0x7f, false},
|
||||||
|
"VTRN1": {0x4ec02800, 0x7f, false},
|
||||||
|
"VTRN2": {0x4ec06800, 0x7f, false},
|
||||||
|
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||||
|
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||||
|
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
||||||
|
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
|
||||||
|
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
|
||||||
|
// combination for them, so they stay out and the encoder rejects the shape.
|
||||||
|
var a64SimdVZero = map[string]uint32{
|
||||||
|
"VCMEQ": 0x0e209800,
|
||||||
|
"VCMGT": 0x0e208800,
|
||||||
|
"VCMGE": 0x2e208800,
|
||||||
|
"VCMLT": 0x0e20a800,
|
||||||
|
"VCMLE": 0x2e209800,
|
||||||
|
// FP compares against (0.0): the register forms above carry the U and op
|
||||||
|
// bits; the zero forms reshape them.
|
||||||
|
"VFCMEQ": 0x0ea0d800,
|
||||||
|
"VFCMGE": 0x2ea0c800,
|
||||||
|
"VFCMGT": 0x0ea0c800,
|
||||||
|
"VFCMLE": 0x2ea0d800,
|
||||||
|
"VFCMLT": 0x0ea0e800,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
||||||
|
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
|
||||||
|
// the register pair spelling ORR Vd, Vn, Vm.
|
||||||
|
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||||
|
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||||
|
"VREV64": {0x0e200800, 0x3f, false},
|
||||||
|
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||||
|
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||||
|
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||||
|
// Two-register data-processing across one arrangement.
|
||||||
|
"VABS": {0x0e20b800, 0x7f, false},
|
||||||
|
"VNEG": {0x2e20b800, 0x7f, false},
|
||||||
|
"VCLS": {0x0e204800, 0x7f, false},
|
||||||
|
"VCLZ": {0x2e204800, 0x7f, false},
|
||||||
|
"VCNT": {0x0e205800, 0x7f, false},
|
||||||
|
"VNOT": {0x2e205800, 0x7f, false},
|
||||||
|
"VSQABS": {0x0e207800, 0x7f, false},
|
||||||
|
"VSQNEG": {0x2e207800, 0x7f, false},
|
||||||
|
"VRBIT": {0x6e605800, 0x7f, false},
|
||||||
|
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
||||||
|
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
||||||
|
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
||||||
|
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
||||||
|
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
||||||
|
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
||||||
|
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
||||||
|
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
||||||
|
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
||||||
|
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
||||||
|
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
||||||
|
// Across-vector reductions: the operand arrangement rides as usual and
|
||||||
|
// the destination stays a bare V register.
|
||||||
|
"VADDV": {0x0e31b800, 0x3f, false},
|
||||||
|
"VSMAXV": {0x0e30a800, 0x3f, false},
|
||||||
|
"VSMINV": {0x0e31a800, 0x3f, false},
|
||||||
|
"VUMAXV": {0x2e30a800, 0x3f, false},
|
||||||
|
"VUMINV": {0x2e31a800, 0x3f, false},
|
||||||
|
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
||||||
|
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
||||||
|
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
||||||
|
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||||
|
// carry when they spell one at all; a bare V/F spelling is accepted as is.
|
||||||
|
var a64CryptoArr = map[string]int{
|
||||||
|
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
|
||||||
|
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
|
||||||
|
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
|
||||||
|
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
|
||||||
|
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64DCOps maps the data-cache maintenance operation names to their fixed
|
||||||
|
// word (the register rides bits 4:0).
|
||||||
|
var a64DCOps = map[string]uint32{
|
||||||
|
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
|
||||||
|
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64MRSOps maps the system register names GOROOT reads to their fixed word
|
||||||
|
// (the destination register rides bits 4:0).
|
||||||
|
var a64MRSOps = map[string]uint32{
|
||||||
|
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
|
||||||
|
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
|
||||||
|
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
|
||||||
|
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
|
||||||
|
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
|
||||||
|
"NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64MSRRegOps maps the system register names GOROOT writes through the
|
||||||
|
// MSR (register) form, spelled in Go assembly as MOVD Rn, <sysreg> or
|
||||||
|
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||||
|
var a64MSRRegOps = map[string]uint32{
|
||||||
|
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||||
|
"ELR_EL1": 0xd5184020,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||||
|
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
|
||||||
|
var a64MSROps = map[string]uint32{
|
||||||
|
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64PRFOps maps the prefetch operation names to their prfop immediate
|
||||||
|
// (word = 0xf9800000 | Rn<<5 | prfop).
|
||||||
|
var a64PRFOps = map[string]int{
|
||||||
|
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
|
||||||
|
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
|
||||||
|
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
|
||||||
|
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
|
||||||
|
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
|
||||||
|
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64VLD1Base holds the fixed words of the multi-register structure
|
||||||
|
// accesses, indexed by register count 1..4, before the Q and size bits.
|
||||||
|
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
|
||||||
|
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
|
||||||
|
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
|
||||||
|
|
||||||
|
// a64Vec is a parsed vector operand: the register number, the arrangement
|
||||||
|
// ("" when the operand spells none) and, for element forms, the lane index.
|
||||||
|
type a64Vec struct {
|
||||||
|
reg int
|
||||||
|
arr string
|
||||||
|
idx int
|
||||||
|
hasIdx bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
|
||||||
|
// the same architectural registers the scalar floating-point spellings use),
|
||||||
|
// optionally with an arrangement suffix such as V0.B16 and, for element
|
||||||
|
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
|
||||||
|
// else, including X/W and R spellings, which the toolchain's vector
|
||||||
|
// operands reject as well.
|
||||||
|
func a64VecReg(name string) (v a64Vec, ok bool) {
|
||||||
|
s := strings.TrimSpace(name)
|
||||||
|
if i := strings.IndexByte(s, '.'); i >= 0 {
|
||||||
|
v.arr = strings.TrimSpace(s[i+1:])
|
||||||
|
s = s[:i]
|
||||||
|
}
|
||||||
|
if v.arr != "" {
|
||||||
|
// Element form: B[3], S[2] and friends.
|
||||||
|
if j := strings.IndexByte(v.arr, '['); j >= 0 {
|
||||||
|
k := strings.LastIndexByte(v.arr, ']')
|
||||||
|
if k < j {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
|
||||||
|
if err != nil || n < 0 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
v.idx, v.hasIdx = n, true
|
||||||
|
v.arr = strings.TrimSpace(v.arr[:j])
|
||||||
|
}
|
||||||
|
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 1; i < len(s); i++ {
|
||||||
|
if s[i] < '0' || s[i] > '9' {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
n = n*10 + int(s[i]-'0')
|
||||||
|
}
|
||||||
|
if n > 31 {
|
||||||
|
return v, false
|
||||||
|
}
|
||||||
|
v.reg = n
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
|
||||||
|
// index shifted by the element scale, with the scale's own bit set. B gets
|
||||||
|
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
|
||||||
|
func a64ElemField(arr string, idx int) (uint32, bool) {
|
||||||
|
var shift, low uint32
|
||||||
|
switch arr {
|
||||||
|
case "B8", "B16", "B":
|
||||||
|
shift, low = 1, 1
|
||||||
|
case "H4", "H8", "H":
|
||||||
|
shift, low = 2, 2
|
||||||
|
case "S2", "S4", "S":
|
||||||
|
shift, low = 3, 4
|
||||||
|
case "D1", "D2", "D":
|
||||||
|
shift, low = 4, 8
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
if idx < 0 || idx >= 1<<(5-shift) {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return uint32(idx)<<shift | low, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
|
||||||
|
// The parser keeps parenthesised groups whole but splits bracketed lists on
|
||||||
|
// the commas, so a list arrives as one operand run whose first Raw starts
|
||||||
|
// with "[" and whose last Raw ends with "]". It returns the parsed
|
||||||
|
// registers with the brackets and spaces removed.
|
||||||
|
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
|
||||||
|
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
end = start
|
||||||
|
for end < len(ops) {
|
||||||
|
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
end++
|
||||||
|
}
|
||||||
|
if end >= len(ops) {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
for i := start; i <= end; i++ {
|
||||||
|
s := strings.TrimSpace(ops[i].Raw)
|
||||||
|
s = strings.TrimPrefix(s, "[")
|
||||||
|
s = strings.TrimSuffix(s, "]")
|
||||||
|
if s == "" && len(ops) > start+1 {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
for part := range strings.SplitSeq(s, ",") {
|
||||||
|
v, ok := a64VecReg(part)
|
||||||
|
if !ok {
|
||||||
|
return nil, 0, false
|
||||||
|
}
|
||||||
|
vs = append(vs, v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return vs, end, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- load/store helper tables ----
|
// ---- load/store helper tables ----
|
||||||
|
|||||||
+774
-20
@@ -4,10 +4,11 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestArm64LDRSTREncoding(t *testing.T) {
|
func TestArm64LDRSTREncoding(t *testing.T) {
|
||||||
@@ -104,6 +105,7 @@ func TestArm64RegNum(t *testing.T) {
|
|||||||
}{
|
}{
|
||||||
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
{"R0", 0}, {"R4", 4}, {"R29", 29}, {"R30", 30}, {"R31", 31},
|
||||||
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
{"FP", 29}, {"LR", 30}, {"LINK", 30}, {"SP", 31}, {"ZR", 31},
|
||||||
|
{"R18_PLATFORM", 18},
|
||||||
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
{"F0", 0}, {"F4", 4}, {"F31", 31},
|
||||||
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
{"INVALID", -1}, {"X0", -1}, {"", -1},
|
||||||
}
|
}
|
||||||
@@ -472,12 +474,614 @@ TEXT ·f(SB), NOSPLIT, $0-0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestArm64SIMD tests SIMD encoding (via the instruction table).
|
// TestArm64SIMD tests SIMD encoding (via the arrangement-aware table).
|
||||||
func TestArm64SIMD(t *testing.T) {
|
func TestArm64SIMD(t *testing.T) {
|
||||||
// Verify SIMD instructions are in the table.
|
// Verify SIMD instructions are in the arrangement table.
|
||||||
for _, mnem := range []string{"VADD", "VSUB", "VMUL"} {
|
for _, mnem := range []string{"VADD", "VSUB", "VMUL", "VAND", "VEOR", "VORR", "VCMEQ", "VZIP1", "VZIP2"} {
|
||||||
if _, ok := a64InstrTable[mnem]; !ok {
|
if _, ok := a64SimdVTable[mnem]; !ok {
|
||||||
t.Errorf("%s not in instruction table", mnem)
|
t.Errorf("%s not in the SIMD arrangement table", mnem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64CarryAndBitOps pins the carry-setting arithmetic, the widening
|
||||||
|
// multiplies and the data-processing (1 source) group against go tool asm.
|
||||||
|
func TestArm64CarryAndBitOps(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tADC R0, R2, R12\n\tADCS $0, R1\n\tSBCS R5, R9, R5\n\tSBC R25, R10, R26\n"+
|
||||||
|
"\tMUL R4, R3, R0\n\tUMULH R24, R20, R24\n\tSMULH R1, R2, R3\n\tMSUB R19, R16, R26, R2\n"+
|
||||||
|
"\tRBIT R11, R4\n\tREV R1, R2\n\tCLZ R21, R9\n\tREVW R1, R2\n\tCLSW R1, R2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9a00004c, // ADC R12, R2, R0
|
||||||
|
0xba1f0021, // ADCS R1, R1, ZR
|
||||||
|
0xfa050125, // SBCS R5, R9, R5
|
||||||
|
0xda19015a, // SBC R26, R10, R25
|
||||||
|
0x9b047c60, // MUL R0, R3, R4
|
||||||
|
0x9bd87e98, // UMULH R24, R20, R24
|
||||||
|
0x9b417c43, // SMULH R3, R2, R1
|
||||||
|
0x9b13c342, // MSUB R2, R26, R19, R16
|
||||||
|
0xdac00164, // RBIT R4, R11
|
||||||
|
0xdac00c22, // REV R2, R1
|
||||||
|
0xdac012a9, // CLZ R9, R21
|
||||||
|
0x5ac00822, // REVW R2, R1
|
||||||
|
0x5ac01422, // CLSW R2, R1
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64BitfieldExtract pins UBFX/SBFX: immr wraps to the register
|
||||||
|
// width, an out-of-range imms is an error.
|
||||||
|
func TestArm64BitfieldExtract(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tUBFX $33, R17, $25, R5\n\tUBFXW $4, R1, $9, R2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd361e625, // UBFX immr=1 (33 wrapped), imms=25
|
||||||
|
0x53043022, // UBFXW immr=4, imms=9
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, body := range []string{"\tUBFX $33, R17, $70, R5\n", "\tUBFX $-1, R17, $3, R5\n"} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", body)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64CondCompare pins CCMP/CCMN.
|
||||||
|
func TestArm64CondCompare(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tCCMP LE, R7, $19, $3\n\tCCMP LT, R30, R6, $7\n\tCCMN EQ, R1, R2, $3\n\tCCMPW LE, R7, $19, $3\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xfa53d8e3, // CCMP imm form
|
||||||
|
0xfa46b3c7, // CCMP register form
|
||||||
|
0xba420023, // CCMN register form
|
||||||
|
0x7a53d8e3, // CCMPW
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64CompareBranch pins CBZ/CBNZ/TBZ/TBNZ against a label five and
|
||||||
|
// six words ahead, matching go tool asm's own offsets.
|
||||||
|
func TestArm64CompareBranch(t *testing.T) {
|
||||||
|
// Layout: CBZ(0) TBZ(4) TBNZ(8) CBNZ(12) NOP(16) NOP(17th word...) done.
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
|
||||||
|
"\tCBZ R1, done\n\tTBZ $4, R7, done\n\tTBNZ $33, R7, done\n\tCBNZW R2, done\n" +
|
||||||
|
"\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
// done sits at word 6 from each branch's own pc: CBZ rel 6, TBZ rel 5,
|
||||||
|
// TBNZ rel 4, CBNZW rel 3.
|
||||||
|
want := []uint32{
|
||||||
|
0xb40000c1, // CBZ R1, +6
|
||||||
|
0x362000a7, // TBZ $4, R7, +5
|
||||||
|
0xb7080087, // TBNZ $33, R7, +4
|
||||||
|
0x35000062, // CBNZW R2, +3
|
||||||
|
0xd503201f, 0xd503201f, 0xd503201f,
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64ADR pins ADR against a forward label.
|
||||||
|
func TestArm64ADR(t *testing.T) {
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
|
||||||
|
"\tADR done, R10\n\tNOP\n\tNOP\n\tdone:\tNOP\n\tRET\n"
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
// rel = 12 bytes: immlo 0, immhi 3.
|
||||||
|
want := []uint32{0x1000006a, 0xd503201f, 0xd503201f, 0xd503201f, 0xd65f03c0}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64PairLoadStore pins LDP/STP/LDPW/FLDPD/FSTPD.
|
||||||
|
func TestArm64PairLoadStore(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tSTP (R2, R3), 8(R5)\n\tLDP -8(R5), (R2, R3)\n\tLDPW 4(R0), (R1, R2)\n\tSTPW (R1, R2), 4(R0)\n"+
|
||||||
|
"\tFLDPD 8(R0), (F1, F2)\n\tFSTPD (F3, F4), -8(R5)\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xa9008ca2, // STP (R2, R3), 8(R5)
|
||||||
|
0xa97f8ca2, // LDP -8(R5), (R2, R3)
|
||||||
|
0x29408801, // LDPW 4(R0), (R1, R2)
|
||||||
|
0x29008801, // STPW (R1, R2), 4(R0)
|
||||||
|
0x6d408801, // FLDPD 8(R0), (F1, F2)
|
||||||
|
0x6d3f90a3, // FSTPD (F3, F4), -8(R5)
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64AcquireRelease pins LDAR/STLR and the acquire/release LSE
|
||||||
|
// families.
|
||||||
|
func TestArm64AcquireRelease(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tLDAR (R27), R22\n\tLDARB (R25), R2\n\tLDARW (R12), R29\n\tSTLR R3, (R24)\n\tSTLRB R11, (R22)\n"+
|
||||||
|
"\tCASALD R5, (R6), R7\n\tLDADDALD R5, (R6), R7\n\tLDCLRALB R5, (R6), R7\n\tLDORALD R5, (RSP), R7\n\tSWPALW R5, (R6), R7\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xc8dfff76, // LDAR R22, (R27)
|
||||||
|
0x08dfff22, // LDARB R2, (R25)
|
||||||
|
0x88dffd9d, // LDARW R29, (R12)
|
||||||
|
0xc89fff03, // STLR R3, (R24)
|
||||||
|
0x089ffecb, // STLRB R11, (R22)
|
||||||
|
0xc8e5fcc7, // CASALD R7, (R6), R5
|
||||||
|
0xf8e500c7, // LDADDALD R7, (R6), R5
|
||||||
|
0x38e510c7, // LDCLRALB R7, (R6), R5
|
||||||
|
0xf8e533e7, // LDORALD R7, (RSP), R5
|
||||||
|
0xb8e580c7, // SWPALW R7, (R6), R5
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64BTI pins the landing-pad family against the toolchain words:
|
||||||
|
// only the uppercase C/J/JC spellings assemble, and bare BTI is a
|
||||||
|
// diagnostic, never a panic.
|
||||||
|
func TestArm64BTI(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tBTI C\n\tBTI J\n\tBTI JC\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd503245f, // BTI C
|
||||||
|
0xd503249f, // BTI J
|
||||||
|
0xd50324df, // BTI JC
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("got %d words, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %#x, want %#x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, src := range []string{"\tBTI\n", "\tBTI c\n", "\tBTI B\n"} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+src+"\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("BTI spelling %q should be rejected, as go tool asm rejects it", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64System pins BRK, SVC, the barriers, cache maintenance and the
|
||||||
|
// system register accesses.
|
||||||
|
func TestArm64System(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tBRK $35943\n\tBRK\n\tSVC $7165\n\tDMB $1\n\tDSB $1\n\tISB $15\n"+
|
||||||
|
"\tDC ZVA, R4\n\tDC IVAC, R1\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $9, DAIFSet\n\tMSR $3, SPSel\n"+
|
||||||
|
"\tPRFM (R0), PLDL1KEEP\n\tPRFM (R3), PLDL3KEEP\n\tPRFM (R2), $25\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd4318ce0, // BRK $35943
|
||||||
|
0xd4200000, // BRK
|
||||||
|
0xd4037fa1, // SVC $7165
|
||||||
|
0xd50331bf, // DMB $1
|
||||||
|
0xd503319f, // DSB $1
|
||||||
|
0xd5033fdf, // ISB $15
|
||||||
|
0xd50b7424, // DC ZVA, R4
|
||||||
|
0xd5087621, // DC IVAC, R1
|
||||||
|
0xd53b00e3, // MRS DCZID_EL0, R3
|
||||||
|
0xd53be040, // MRS CNTVCT_EL0, R0
|
||||||
|
0xd50349df, // MSR $9, DAIFSet
|
||||||
|
0xd50043bf, // MSR $3, SPSel
|
||||||
|
0xf9800000, // PRFM (R0), PLDL1KEEP
|
||||||
|
0xf9800064, // PRFM (R3), PLDL3KEEP
|
||||||
|
0xf9800059, // PRFM (R2), $25
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64Crypto pins the AES and SHA families.
|
||||||
|
func TestArm64Crypto(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tAESE V31.B16, V29.B16\n\tAESD V22.B16, V19.B16\n\tAESIMC V12.B16, V27.B16\n\tAESMC V14.B16, V28.B16\n"+
|
||||||
|
"\tSHA1C V8.S4, V8, V2\n\tSHA1H V17, V25\n\tSHA1P V3.S4, V20, V27\n\tSHA1SU0 V17.S4, V13.S4, V16.S4\n\tSHA1SU1 V24.S4, V23.S4\n"+
|
||||||
|
"\tSHA256H V4.S4, V2, V11\n\tSHA256H2 V6.S4, V16, V11\n\tSHA256SU0 V0.S4, V16.S4\n\tSHA256SU1 V31.S4, V3.S4, V15.S4\n"+
|
||||||
|
"\tSHA512H V2.D2, V1, V0\n\tSHA512H2 V4.D2, V3, V2\n\tSHA512SU0 V9.D2, V8.D2\n\tSHA512SU1 V7.D2, V6.D2, V5.D2\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4e284bfd, // AESE
|
||||||
|
0x4e285ad3, // AESD
|
||||||
|
0x4e28799b, // AESIMC
|
||||||
|
0x4e2869dc, // AESMC
|
||||||
|
0x5e080102, // SHA1C
|
||||||
|
0x5e280a39, // SHA1H
|
||||||
|
0x5e03129b, // SHA1P
|
||||||
|
0x5e1131b0, // SHA1SU0
|
||||||
|
0x5e281b17, // SHA1SU1
|
||||||
|
0x5e04404b, // SHA256H
|
||||||
|
0x5e06520b, // SHA256H2
|
||||||
|
0x5e282810, // SHA256SU0
|
||||||
|
0x5e1f606f, // SHA256SU1
|
||||||
|
0xce628020, // SHA512H
|
||||||
|
0xce648462, // SHA512H2
|
||||||
|
0xcec08128, // SHA512SU0
|
||||||
|
0xce6788c5, // SHA512SU1
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDLogical pins the arrangement-aware three- and two-register
|
||||||
|
// SIMD paths.
|
||||||
|
func TestArm64SIMDLogical(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVADD V1.B16, V2.B16, V3.B16\n\tVAND V4.B16, V4.B16, V9.B16\n\tVEOR V0.B16, V1.B16, V0.B16\n"+
|
||||||
|
"\tVORR V5.B16, V4.B16, V3.B16\n\tVADDP V1.H8, V2.H8, V3.H8\n\tVZIP1 V16.H8, V3.H8, V19.H8\n\tVZIP2 V22.D2, V25.D2, V21.D2\n"+
|
||||||
|
"\tVCMEQ V24.S4, V13.S4, V12.S4\n\tVCMEQ $0, V2.H4, V3.H4\n\tVREV32 V2.H8, V1.H8\n\tVREV64 V2.S4, V3.S4\n\tVUADDLV V31.S4, V11\n"+
|
||||||
|
"\tVPMULL V2.D1, V1.D1, V3.Q1\n\tVPMULL2 V2.B16, V1.B16, V4.H8\n\tVRAX1 V26.D2, V29.D2, V30.D2\n\tVMOV V2.B16, V4.B16\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4e218443, // VADD 16B
|
||||||
|
0x4e241c89, // VAND
|
||||||
|
0x6e201c20, // VEOR
|
||||||
|
0x4ea51c83, // VORR
|
||||||
|
0x4e61bc43, // VADDP 8H
|
||||||
|
0x4e503873, // VZIP1 8H
|
||||||
|
0x4ed67b35, // VZIP2 2D
|
||||||
|
0x6eb88dac, // VCMEQ 4S
|
||||||
|
0x0e609843, // VCMEQ $0, 4H
|
||||||
|
0x6e600841, // VREV32 8H
|
||||||
|
0x4ea00843, // VREV64 4S
|
||||||
|
0x6eb03beb, // VUADDLV 4S
|
||||||
|
0x0ee2e023, // VPMULL D1
|
||||||
|
0x4e22e024, // VPMULL2 16B
|
||||||
|
0xce7a8fbe, // VRAX1 2D
|
||||||
|
0x4ea21c44, // VMOV 16B pair
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDWide pins the four-register crypto group, VXAR, VEXT and the
|
||||||
|
// shift-by-immediate encodings.
|
||||||
|
func TestArm64SIMDWide(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVEOR3 V2.B16, V7.B16, V12.B16, V25.B16\n\tVBCAX V1.B16, V2.B16, V26.B16, V31.B16\n"+
|
||||||
|
"\tVXAR $63, V27.D2, V21.D2, V26.D2\n\tVEXT $4, V2.B8, V1.B8, V3.B8\n\tVEXT $8, V2.B16, V1.B16, V3.B16\n"+
|
||||||
|
"\tVSHL $7, V22.D2, V25.D2\n\tVUSHR $6, V22.H8, V23.H8\n\tVSRI $24, V1.S4, V2.S4\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xce070999, // VEOR3
|
||||||
|
0xce22075f, // VBCAX
|
||||||
|
0xce9bfeba, // VXAR
|
||||||
|
0x2e022023, // VEXT B8
|
||||||
|
0x6e024023, // VEXT B16
|
||||||
|
0x4f4756d9, // VSHL D2 $7
|
||||||
|
0x6f1a06d7, // VUSHR H8 $6
|
||||||
|
0x6f284422, // VSRI S4 $24
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDElement pins VDUP and the VMOV element forms.
|
||||||
|
func TestArm64SIMDElement(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVDUP V31.B[15], V18\n\tVDUP V19.S[3], V18.S4\n\tVDUP V1.D[1], V2.D2\n"+
|
||||||
|
"\tVMOV V13.S[0], R20\n\tVMOV V11.B[11], V16.B[12]\n\tVMOV R20, V21.B[2]\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x5e1f07f2, // VDUP element to register
|
||||||
|
0x4e1c0672, // VDUP element across S4
|
||||||
|
0x4e180422, // VDUP element across D2
|
||||||
|
0x0e043db4, // VMOV element to register
|
||||||
|
0x6e195d70, // VMOV element to element
|
||||||
|
0x4e051e95, // VMOV register into element
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64GPIntoVector pins the whole-vector moves VMOV/VDUP Rs, Vd.<T>
|
||||||
|
// against `go tool asm -S` output (Go 1.27, arm64): word = Q | 7<<25 |
|
||||||
|
// imm5<<16 | 3<<10 | rs<<5 | rd, shared by both mnemonics, the form
|
||||||
|
// sys_windows_arm64.s and the bytealg loops use. The D1 destination is
|
||||||
|
// rejected, as the toolchain rejects it.
|
||||||
|
func TestArm64GPIntoVector(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVMOV R5, V5.B16\n\tVMOV R1, V2.B8\n\tVMOV R3, V4.H4\n"+
|
||||||
|
"\tVMOV R9, V10.S4\n\tVMOV R7, V31.H8\n\tVMOV R11, V12.D2\n"+
|
||||||
|
"\tVDUP R5, V5.B16\n\tVDUP R9, V10.H8\n\tVMOV V4.B16, V20.B16\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4e010ca5, // VMOV R5, V5.B16
|
||||||
|
0x0e010c22, // VMOV R1, V2.B8
|
||||||
|
0x0e020c64, // VMOV R3, V4.H4
|
||||||
|
0x4e040d2a, // VMOV R9, V10.S4
|
||||||
|
0x4e020cff, // VMOV R7, V31.H8
|
||||||
|
0x4e080d6c, // VMOV R11, V12.D2
|
||||||
|
0x4e010ca5, // VDUP R5, V5.B16 (same word as VMOV)
|
||||||
|
0x4e020d2a, // VDUP R9, V10.H8
|
||||||
|
0x4ea41c94, // VMOV V4.B16, V20.B16 (vector to vector stays ORR)
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tVMOV R7, V8.D1\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("VMOV R7, V8.D1 assembled, want an arrangement error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SimdTwoOperand pins the two-operand accumulate spellings
|
||||||
|
// VADD/VSUB Vm, Vn against `go tool asm -S` output (Go 1.27, arm64):
|
||||||
|
// word = 5<<28|7<<25|7<<21|1<<15|1<<10 for VADD (7<<28 for VSUB) with
|
||||||
|
// rf<<16 | rn<<5 | rn, bare V registers only (asm7.go case 89).
|
||||||
|
func TestArm64SimdTwoOperand(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVADD V7, V8\n\tVSUB V7, V8\n\tVADD V1, V2\n\tVADD V0.B16, V1.B16, V2.B16\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x5ee78508, // VADD V7, V8
|
||||||
|
0x7ee78508, // VSUB V7, V8
|
||||||
|
0x5ee18442, // VADD V1, V2
|
||||||
|
0x4e208422, // VADD arranged: the ordinary three-register path
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64TruncMove pins the truncating register moves against
|
||||||
|
// `go tool asm -S` output (Go 1.27, arm64): the signed forms lower to SXTB,
|
||||||
|
// SXTH and SXTW (SBFM), the unsigned byte and halfword forms to UXTB and
|
||||||
|
// UXTH (UBFM), MOVWU to a W ORR, and a narrow move out of the zero register
|
||||||
|
// drops to the W ORR too (asm7.go case 45).
|
||||||
|
func TestArm64TruncMove(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVB R3, R4\n\tMOVH R5, R6\n\tMOVW R9, R10\n"+
|
||||||
|
"\tMOVBU R3, R4\n\tMOVHU R3, R4\n\tMOVWU R3, R4\n\tMOVD R3, R4\n"+
|
||||||
|
"\tMOVD ZR, R4\n\tMOVB ZR, R4\n\tMOVWU ZR, R5\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x93401c64, // MOVB = SXTB
|
||||||
|
0x93403ca6, // MOVH = SXTH
|
||||||
|
0x93407d2a, // MOVW = SXTW
|
||||||
|
0xd3401c64, // MOVBU = UXTB
|
||||||
|
0xd3403c64, // MOVHU = UXTH
|
||||||
|
0x2a0303e4, // MOVWU = ORR W
|
||||||
|
0xaa0303e4, // MOVD = ORR X
|
||||||
|
0xaa1f03e4, // MOVD ZR, R4 keeps the X form
|
||||||
|
0x2a1f03e4, // MOVB ZR, R4 drops to the W form
|
||||||
|
0x2a1f03e5, // MOVWU ZR, R5
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64SIMDLoadStore pins the structure loads and stores.
|
||||||
|
func TestArm64SIMDLoadStore(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tVLD1 (R2), [V21.B16]\n\tVLD1 (R1), [V2.B16, V3.B16]\n\tVLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]\n"+
|
||||||
|
"\tVLD1.P 32(R1), [V2.B16, V3.B16]\n\tVST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n"+
|
||||||
|
"\tVLD1R (R1), [V9.B8]\n\tVLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x4c407055, // VLD1 one register
|
||||||
|
0x4c40a022, // VLD1 two registers
|
||||||
|
0x0c402fae, // VLD1 four registers D1
|
||||||
|
0x4cdfa022, // VLD1.P two registers
|
||||||
|
0x4c0029c2, // VST1 four registers S4
|
||||||
|
0x4c9f7022, // VST1.P one register
|
||||||
|
0x0d40c029, // VLD1R
|
||||||
|
0x0d60e000, // VLD4R
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MoviLiteral pins the VMOVS/VMOVD/VMOVQ constant loads: three
|
||||||
|
// words each (ADRP, ADD, wide load) plus the pooled literal in the data
|
||||||
|
// section.
|
||||||
|
func TestArm64MoviLiteral(t *testing.T) {
|
||||||
|
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n" +
|
||||||
|
"\tVMOVS $0x80402010, V11\n\tVMOVD $0x8040201008040201, V20\n" +
|
||||||
|
"\tVMOVQ $0x7040201008040201, $0x8040201008040201, V10\n\tRET\n"
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
if img.Funcs[0].Size != 12*3+4 {
|
||||||
|
t.Errorf("func size = %d, want %d", img.Funcs[0].Size, 12*3+4)
|
||||||
|
}
|
||||||
|
want := []uint32{
|
||||||
|
0x9000001b, 0x9100037b, 0xbd40036b, // VMOVS: ADRP, ADD, LDR S
|
||||||
|
0x9000001b, 0x9100037b, 0xfd400374, // VMOVD: ADRP, ADD, LDR D
|
||||||
|
0x9000001b, 0x9100037b, 0x3dc0036a, // VMOVQ: ADRP, ADD, LDR Q
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The literals sit in the data section.
|
||||||
|
var found32, found64, found128 bool
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
switch d.Name {
|
||||||
|
case "$i32.80402010":
|
||||||
|
found32 = d.Size == 4
|
||||||
|
case "$i64.8040201008040201":
|
||||||
|
found64 = d.Size == 8
|
||||||
|
case "$i128.80402010080402017040201008040201":
|
||||||
|
found128 = d.Size == 16
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found32 || !found64 || !found128 {
|
||||||
|
t.Errorf("literals missing: i32=%v i64=%v i128=%v", found32, found64, found128)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MOVK pins standalone MOVK with the hw field derived from the
|
||||||
|
// chunk position.
|
||||||
|
func TestArm64MOVK(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVK $1234, R5\n\tMOVK $305397760, R5\n\tMOVKW $1234, R5\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xf2809a45, // MOVK hw=0
|
||||||
|
0xf2a24685, // MOVK hw=1
|
||||||
|
0x72809a45, // MOVKW hw=0
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MOVKHighLane pins the shifted high-lane immediate the arm64 test
|
||||||
|
// kernels write: $(40000<<48) folds to a negative int64, and the toolchain
|
||||||
|
// reads the value as an unsigned 64-bit pattern when it picks the lane.
|
||||||
|
func TestArm64MOVKHighLane(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tMOVK $(40000<<48), R0\n\tMOVK $0x9c40000000000000, R1\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xf2f38800, // MOVK $(40000<<48), R0 (go tool asm: f2f38800)
|
||||||
|
0xf2f38801, // MOVK hw=3
|
||||||
|
0xd65f03c0,
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64MoveWideZeroImmediate pins the toolchain's rejection of a zero
|
||||||
|
// immediate in the move-wide family (optab case 33: "zero shifts cannot be
|
||||||
|
// handled"): every lane is zero, so no hw field can carry it.
|
||||||
|
func TestArm64MoveWideZeroImmediate(t *testing.T) {
|
||||||
|
for _, mnem := range []string{"MOVK", "MOVZ", "MOVN"} {
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t"+mnem+" $0, R0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("%s: parse: %v", mnem, errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Errorf("%s $0: expected error, got nil", mnem)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -851,20 +1455,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
|
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
|
||||||
// rejected instead of wrapping through int32.
|
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
|
||||||
func TestArm64AddSubImmRange(t *testing.T) {
|
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
|
||||||
for _, body := range []string{
|
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
|
||||||
"\tADD $0x100000000, R0, R1\n",
|
// Comparisons never split, and the W forms classify the 32-bit value. Every
|
||||||
"\tSUB $-0x100000000, R0, R1\n",
|
// word is go tool asm's own for the same source.
|
||||||
"\tCMP $0x100000000, R0\n",
|
func TestArm64AddSubImmWide(t *testing.T) {
|
||||||
} {
|
got := arm64Words(t, strings.Join([]string{
|
||||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
"\tADD $0xaaaaaa, R2, R3",
|
||||||
if len(errs) > 0 {
|
"\tSUB $0xaaaaaa, R2",
|
||||||
t.Fatalf("parse: %v", errs)
|
"\tADD $0x186a0, R2, R5",
|
||||||
|
"\tADD $0x1ffe00, R2, R3",
|
||||||
|
"\tADD $0x3fffffffc000, R5",
|
||||||
|
"\tADD $-100000, R2, R3",
|
||||||
|
"\tADD $-2048, R2, R3",
|
||||||
|
"\tCMP $0xaaaaaa, R2",
|
||||||
|
"\tCMP $0xffffffffffa0, R3",
|
||||||
|
"\tCMPW $27745, R2",
|
||||||
|
"\tCMPW $0x60060, R2",
|
||||||
|
"\tADDS $0xaaaaaa, R2, R3",
|
||||||
|
"\tADD $0x12345678, R2, R3",
|
||||||
|
"\tADDW $0x60060, R2",
|
||||||
|
"\tSUB $0xe7791f700, R3, R1",
|
||||||
|
"\tADDW $0x12345678, R2, R3",
|
||||||
|
"\tCMN $0x1000000, R2",
|
||||||
|
}, "\n")+"\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
|
||||||
|
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
|
||||||
|
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
|
||||||
|
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
|
||||||
|
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
|
||||||
|
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
|
||||||
|
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
|
||||||
|
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
|
||||||
|
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
|
||||||
|
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
|
||||||
|
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
|
||||||
|
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
|
||||||
|
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
|
||||||
|
0x11018042, 0x11418042, // ADDW $0x60060: W split
|
||||||
|
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
|
||||||
|
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
|
||||||
|
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
}
|
}
|
||||||
if _, err := AssembleFileARM64(f); err == nil {
|
}
|
||||||
t.Errorf("%s: expected an error, got none", body)
|
}
|
||||||
|
|
||||||
|
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
|
||||||
|
// three operands, the ROR shift on the logical group (and its rejection for
|
||||||
|
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
|
||||||
|
// alias with an element selector. Words are go tool asm's own.
|
||||||
|
func TestArm64CarryShiftAlias(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
|
||||||
|
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
|
||||||
|
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9a1f0294, // ADC ZR, R20, R20
|
||||||
|
0x9a1f0284, // ADC ZR, R20, R4
|
||||||
|
0xfa1f008c, // SBCS ZR, R4, R12
|
||||||
|
0xfa0f008c, // SBCS R15, R4, R12
|
||||||
|
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
|
||||||
|
0x8ac18443, // AND R1 ROR 33, R2, R3
|
||||||
|
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
|
||||||
|
0xda0203e7, // SBC ZR, R7, R2
|
||||||
|
0x9b0eff77, // MSUB ZR, R27, R14, R23
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ROR on an arithmetic form is unallocated: the toolchain reports an
|
||||||
|
// unsupported shift operator.
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Error("ADD R1@>33: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
|
||||||
|
// operand with an element selector and inside a split register list: the
|
||||||
|
// aliases resolve textually where the parser carries the selector apart from
|
||||||
|
// the name. Words are go tool asm's own.
|
||||||
|
func TestArm64VecAliasElement(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
|
||||||
|
#define POLY V15
|
||||||
|
#define ACC0 V8
|
||||||
|
#define ACC1 V9
|
||||||
|
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
VMOV R1, POLY.D[0]
|
||||||
|
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||||
|
VLD1 (R0), [ACC0.B16]
|
||||||
|
VLD1.P (R0), [ACC0.B16, ACC1.B16]
|
||||||
|
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
want := []uint32{
|
||||||
|
0x4e081c2f, // INS V15.D[0], R1
|
||||||
|
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
|
||||||
|
0x4c407008, // VLD1 (R0), [V8.B16]
|
||||||
|
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
|
||||||
|
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
|
||||||
|
// once the value leaves every imm12 form: a constant sequence into REGTMP
|
||||||
|
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
|
||||||
|
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
|
||||||
|
// tool asm's own.
|
||||||
|
func TestArm64AddSubImmBeyond32(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||||
|
0x8b1b0001, // ADD R27, R0, R1
|
||||||
|
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
|
||||||
|
0xcb1b0001, // SUB R27, R0, R1
|
||||||
|
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||||
|
0xeb1b001f, // CMP R27, R0
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -53,7 +53,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// parseArm64File is a helper assembling one arm64 source file.
|
// parseArm64File is a helper assembling one arm64 source file.
|
||||||
|
|||||||
+568
-39
@@ -5,9 +5,10 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Assemble encodes the body of a TEXT function into x86-64 machine code,
|
// Assemble encodes the body of a TEXT function into x86-64 machine code,
|
||||||
@@ -28,7 +29,7 @@ import (
|
|||||||
// emitted: the bytes match go tool asm only for NOSPLIT functions or
|
// emitted: the bytes match go tool asm only for NOSPLIT functions or
|
||||||
// zero-frame leaves, where the toolchain emits no guard either.
|
// zero-frame leaves, where the toolchain emits no guard either.
|
||||||
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
||||||
code, _, labels, _, _, err := assemble(t, nil)
|
code, _, labels, _, _, _, err := assemble(t, nil)
|
||||||
return code, labels, err
|
return code, labels, err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -37,10 +38,25 @@ func Assemble(t *ast.Text) ([]byte, map[string]int, error) {
|
|||||||
// rejects SB operands outright (single-function assembly cannot resolve
|
// rejects SB operands outright (single-function assembly cannot resolve
|
||||||
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
|
// them). When allowExternal is set, a reference to a symbol no GLOBL in the
|
||||||
// file defines is recorded as an external relocation instead of failing
|
// file defines is recorded as an external relocation instead of failing
|
||||||
// the object-file emitters resolve it at link time.
|
// the object-file emitters resolve it at link time. goos selects the TLS
|
||||||
|
// access form: the empty default behaves as linux.
|
||||||
type linkInfo struct {
|
type linkInfo struct {
|
||||||
symbols map[string]bool
|
symbols map[string]bool
|
||||||
allowExternal bool
|
allowExternal bool
|
||||||
|
goos string
|
||||||
|
}
|
||||||
|
|
||||||
|
// tlsOneInsn reports the one-instruction TLS form, obj6.go's
|
||||||
|
// CanUse1InsnTLS for the GOOS gasm supports: the bare TLS load nops out and
|
||||||
|
// the (TLS*1) index folds to a segment-absolute access. Windows and plan9
|
||||||
|
// keep the two-instruction form; shared linux does too, which gasm's raw
|
||||||
|
// path does not model and therefore does not select.
|
||||||
|
func (l *linkInfo) tlsOneInsn() bool {
|
||||||
|
switch l.goos {
|
||||||
|
case "", "linux", "freebsd":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
// sbPatch is a function-relative static-symbol relocation: the disp32 field
|
// sbPatch is a function-relative static-symbol relocation: the disp32 field
|
||||||
@@ -66,7 +82,10 @@ type spadjStep struct {
|
|||||||
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
// assemble encodes a TEXT body, returning the machine code, the static-symbol
|
||||||
// patch sites (for the file-level layout to resolve), the label table and the
|
// patch sites (for the file-level layout to resolve), the label table and the
|
||||||
// stack-adjustment boundaries.
|
// stack-adjustment boundaries.
|
||||||
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, error) {
|
func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, []spadjStep, []LineEntry, []floatPoolEntry, error) {
|
||||||
|
if err := checkAdjspBalance(t); err != nil {
|
||||||
|
return nil, nil, nil, nil, nil, nil, err
|
||||||
|
}
|
||||||
fi := computeFrame(t)
|
fi := computeFrame(t)
|
||||||
chain := jumpChain(t)
|
chain := jumpChain(t)
|
||||||
resolve := func(name string) string {
|
resolve := func(name string) string {
|
||||||
@@ -82,29 +101,118 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
// outgrows the short form.
|
// outgrows the short form.
|
||||||
long := make([]bool, len(t.Body))
|
long := make([]bool, len(t.Body))
|
||||||
sizes := make([]int, len(t.Body))
|
sizes := make([]int, len(t.Body))
|
||||||
|
numTargets := make([]int, len(t.Body))
|
||||||
|
for i := range numTargets {
|
||||||
|
numTargets[i] = -1
|
||||||
|
}
|
||||||
offsets := map[string]int{}
|
offsets := map[string]int{}
|
||||||
pcs := make([]int, len(t.Body))
|
pcs := make([]int, len(t.Body))
|
||||||
var guardJBlong, guardJBElong, moreJMPlong bool
|
var guardJBlong, guardJBElong, moreJMPlong bool
|
||||||
|
poolSeen := map[string]bool{}
|
||||||
|
var poolList []floatPoolEntry
|
||||||
for {
|
for {
|
||||||
guard := fi.guardLen(guardJBlong, guardJBElong)
|
guard := fi.guardLen(guardJBlong, guardJBElong)
|
||||||
pos := guard + len(fi.prologue)
|
pos := guard + len(fi.prologue)
|
||||||
|
for i := range numTargets {
|
||||||
|
numTargets[i] = -1
|
||||||
|
}
|
||||||
|
idxAtPc := map[int]int{}
|
||||||
for i, stmt := range t.Body {
|
for i, stmt := range t.Body {
|
||||||
switch s := stmt.(type) {
|
switch s := stmt.(type) {
|
||||||
case *ast.Label:
|
case *ast.Label:
|
||||||
offsets[s.Name.Text] = pos
|
offsets[s.Name.Text] = pos
|
||||||
case *ast.Instr:
|
case *ast.Instr:
|
||||||
|
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
|
||||||
|
// The alignment pseudo-statement: its size is the
|
||||||
|
// padding to the next boundary at this very position,
|
||||||
|
// filled with NOPs at emission.
|
||||||
|
pad, err := pcAlignPad(pcAlignValue(s), pos)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, nil, nil, nil, fmt.Errorf("PCALIGN: %w", err)
|
||||||
|
}
|
||||||
|
sizes[i] = pad
|
||||||
|
pcs[i] = pos
|
||||||
|
pos += pad
|
||||||
|
continue
|
||||||
|
}
|
||||||
sz, err := instrSize(s, fi, long[i], link)
|
sz, err := instrSize(s, fi, long[i], link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||||
}
|
}
|
||||||
sizes[i] = sz
|
sizes[i] = sz
|
||||||
pcs[i] = pos
|
pcs[i] = pos
|
||||||
|
idxAtPc[pos] = i
|
||||||
pos += sz
|
pos += sz
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
bodyLen := pos - (guard + len(fi.prologue))
|
bodyLen := pos - (guard + len(fi.prologue))
|
||||||
// Expand any short jump whose displacement no longer fits rel8.
|
// Expand any short jump whose displacement no longer fits rel8.
|
||||||
changed := false
|
changed := false
|
||||||
|
// Numeric ±N(PC) jumps resolve against this iteration's layout; the
|
||||||
|
// emission pass reads the same table after the loop converges. A
|
||||||
|
// target that is itself an unconditional local JMP is chased to the
|
||||||
|
// ultimate target: the toolchain's brloop pass collapses branch-to-
|
||||||
|
// branch chains before it encodes, so matching its bytes requires
|
||||||
|
// the same redirection.
|
||||||
|
for i := range numTargets {
|
||||||
|
numTargets[i] = -1
|
||||||
|
}
|
||||||
|
for i, stmt := range t.Body {
|
||||||
|
s, ok := stmt.(*ast.Instr)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(s.Operands) == 1 {
|
||||||
|
if n, isNum := pcJumpOffset(s.Operands[0]); isNum {
|
||||||
|
if target, okT := pcJumpTarget(t, i, n, pcs); okT {
|
||||||
|
numTargets[i] = target
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for i := range numTargets {
|
||||||
|
if numTargets[i] < 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
tgt := numTargets[i]
|
||||||
|
for hop := 0; hop < len(t.Body); hop++ {
|
||||||
|
idx, ok := idxAtPc[tgt]
|
||||||
|
if !ok {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
in, ok := t.Body[idx].(*ast.Instr)
|
||||||
|
if !ok || strings.ToUpper(in.Mnemonic.Text) != "JMP" || len(in.Operands) != 1 {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if name, isLabel := labelName(in.Operands[0]); isLabel {
|
||||||
|
tgt = offsets[resolve(name)]
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if n, isNum := pcJumpOffset(in.Operands[0]); isNum {
|
||||||
|
next, okT := pcJumpTarget(t, idx, n, pcs)
|
||||||
|
if !okT {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
tgt = next
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
break // JMP through a register or memory: the chain ends
|
||||||
|
}
|
||||||
|
numTargets[i] = tgt
|
||||||
|
}
|
||||||
|
for i, stmt := range t.Body {
|
||||||
|
s, ok := stmt.(*ast.Instr)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if numTargets[i] >= 0 && !long[i] {
|
||||||
|
rel := int64(numTargets[i] - (pcs[i] + jumpSize(strings.ToUpper(s.Mnemonic.Text), false)))
|
||||||
|
if !fits8(rel) {
|
||||||
|
long[i] = true
|
||||||
|
changed = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
for i, stmt := range t.Body {
|
for i, stmt := range t.Body {
|
||||||
s, ok := stmt.(*ast.Instr)
|
s, ok := stmt.(*ast.Instr)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -203,6 +311,14 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
|
spadjStep{guardLen + len(fi.prologue), 8 + fi.size},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
// frameBase is the SP delta the prologue leaves: 8 for the saved base
|
||||||
|
// pointer plus the frame, 0 frameless. bodyDelta tracks the ADJSP
|
||||||
|
// statements' straight-line sum, so a mid-body step's value is the
|
||||||
|
// frame base plus what the body has opened so far.
|
||||||
|
frameBase, bodyDelta := 0, 0
|
||||||
|
if fi.useFP {
|
||||||
|
frameBase = 8 + fi.size
|
||||||
|
}
|
||||||
pos := guardLen + len(fi.prologue)
|
pos := guardLen + len(fi.prologue)
|
||||||
for i, stmt := range t.Body {
|
for i, stmt := range t.Body {
|
||||||
s, ok := stmt.(*ast.Instr)
|
s, ok := stmt.(*ast.Instr)
|
||||||
@@ -218,18 +334,34 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
spadjStep{pos + epi, 0},
|
spadjStep{pos + epi, 0},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
code, ps, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link)
|
code, ps, pool, err := encodeInstr(s, pos, offsets, fi, long[i], resolve, link, numTargets[i])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", s.Mnemonic.Text, err)
|
||||||
|
}
|
||||||
|
for _, entry := range pool {
|
||||||
|
if !poolSeen[entry.name] {
|
||||||
|
poolSeen[entry.name] = true
|
||||||
|
poolList = append(poolList, entry)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if len(code) != sizes[i] {
|
if len(code) != sizes[i] {
|
||||||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: size mismatch (%d vs %d)", s.Mnemonic.Text, len(code), sizes[i])
|
||||||
}
|
}
|
||||||
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
|
if strings.ToUpper(s.Mnemonic.Text) == "CALL" {
|
||||||
for k := range ps {
|
for k := range ps {
|
||||||
ps[k].kind = RelCall
|
ps[k].kind = RelCall
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if strings.ToUpper(s.Mnemonic.Text) == "ADJSP" && len(s.Operands) == 1 && s.Operands[0].Imm.HasVal {
|
||||||
|
// The statement shifted SP mid-body: record the new running
|
||||||
|
// delta as the value in effect from just past the instruction.
|
||||||
|
v := s.Operands[0].Imm.Val
|
||||||
|
if s.Operands[0].Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
bodyDelta += int(v)
|
||||||
|
steps = append(steps, spadjStep{pos + len(code), frameBase + bodyDelta})
|
||||||
|
}
|
||||||
patches = append(patches, ps...)
|
patches = append(patches, ps...)
|
||||||
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
lines = append(lines, LineEntry{Offset: pos, Line: s.Pos().Line})
|
||||||
out = append(out, code...)
|
out = append(out, code...)
|
||||||
@@ -251,7 +383,7 @@ func assemble(t *ast.Text, link *linkInfo) ([]byte, []sbPatch, map[string]int, [
|
|||||||
pos += len(suffix)
|
pos += len(suffix)
|
||||||
}
|
}
|
||||||
_ = pos
|
_ = pos
|
||||||
return out, patches, offsets, steps, lines, nil
|
return out, patches, offsets, steps, lines, poolList, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
// jumpChain precomputes jump-to-jump folding: a label whose first instruction
|
||||||
@@ -398,6 +530,99 @@ func computeFrame(t *ast.Text) frameInfo {
|
|||||||
return fi
|
return fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// pcJumpOffset recognises the numeric relative jump operand ±N(PC) and
|
||||||
|
// returns N: the toolchain counts instructions, not bytes, so +2(PC) targets
|
||||||
|
// the second instruction boundary after the branch.
|
||||||
|
func pcJumpOffset(op *ast.Operand) (int, bool) {
|
||||||
|
if op.Kind != ast.OpAddr || op.Addr.Base != "PC" {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return int(op.Addr.Offset), true
|
||||||
|
}
|
||||||
|
|
||||||
|
// pcJumpTarget resolves a numeric jump at statement index j: N counts the
|
||||||
|
// instruction statements after the jump itself (N = 0 is the jump's own
|
||||||
|
// address, the classic park loop), and the target is the start of the Nth
|
||||||
|
// one. A negative N counts the same way backwards, before the jump: the
|
||||||
|
// exit loops write JMP -3(PC) to land three instructions earlier. Labels
|
||||||
|
// count not, in either direction. It reports false when the count runs
|
||||||
|
// past the end of the function, or before its first instruction.
|
||||||
|
func pcJumpTarget(t *ast.Text, j, n int, pcs []int) (int, bool) {
|
||||||
|
if n == 0 {
|
||||||
|
return pcs[j], true
|
||||||
|
}
|
||||||
|
if n < 0 {
|
||||||
|
seen := 0
|
||||||
|
for k := j - 1; k >= 0; k-- {
|
||||||
|
if _, ok := t.Body[k].(*ast.Instr); !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen--
|
||||||
|
if seen == n {
|
||||||
|
return pcs[k], true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
seen := 0
|
||||||
|
for k := j + 1; k < len(t.Body); k++ {
|
||||||
|
if _, ok := t.Body[k].(*ast.Instr); !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen++
|
||||||
|
if seen == n {
|
||||||
|
return pcs[k], true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
|
||||||
|
// x86 NOP encodings, single-instruction no-ops of lengths 1 to 9 (the
|
||||||
|
// toolchain's asm6.go nop table); longer padding repeats the largest that
|
||||||
|
// fits, greedy from the end.
|
||||||
|
var x86Nops = [][]byte{
|
||||||
|
{0x90},
|
||||||
|
{0x66, 0x90},
|
||||||
|
{0x0F, 0x1F, 0x00},
|
||||||
|
{0x0F, 0x1F, 0x40, 0x00},
|
||||||
|
{0x0F, 0x1F, 0x44, 0x00, 0x00},
|
||||||
|
{0x66, 0x0F, 0x1F, 0x44, 0x00, 0x00},
|
||||||
|
{0x0F, 0x1F, 0x80, 0x00, 0x00, 0x00, 0x00},
|
||||||
|
{0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
|
||||||
|
{0x66, 0x0F, 0x1F, 0x84, 0x00, 0x00, 0x00, 0x00, 0x00},
|
||||||
|
}
|
||||||
|
|
||||||
|
// fillNOPs fills p with the greedy largest single-instruction NOPs, exactly
|
||||||
|
// the toolchain's fillnop.
|
||||||
|
func fillNOPs(p []byte) {
|
||||||
|
for len(p) > 0 {
|
||||||
|
m := min(len(p), len(x86Nops))
|
||||||
|
copy(p[:m], x86Nops[m-1])
|
||||||
|
p = p[m:]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// pcAlignPad computes the padding PCALIGN $align inserts at pos: the
|
||||||
|
// alignment must be a power of two in [8, 2048] and the padding runs to the
|
||||||
|
// next boundary (zero when the position is already aligned).
|
||||||
|
func pcAlignPad(align, pos int) (int, error) {
|
||||||
|
if align <= 0 || align&(align-1) != 0 || align < 8 || align > 2048 {
|
||||||
|
return 0, fmt.Errorf("alignment value of an instruction must be a power of two and in the range [8, 2048], got %d", align)
|
||||||
|
}
|
||||||
|
if lob := pos & (align - 1); lob != 0 {
|
||||||
|
return align - lob, nil
|
||||||
|
}
|
||||||
|
return 0, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// pcAlignValue reads a PCALIGN statement's alignment operand.
|
||||||
|
func pcAlignValue(s *ast.Instr) int {
|
||||||
|
if len(s.Operands) == 1 && s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.HasVal {
|
||||||
|
return int(s.Operands[0].Imm.Val)
|
||||||
|
}
|
||||||
|
return 0 // rejected by pcAlignPad's range check
|
||||||
|
}
|
||||||
|
|
||||||
// hasCall reports whether the function body contains a CALL instruction.
|
// hasCall reports whether the function body contains a CALL instruction.
|
||||||
func hasCall(t *ast.Text) bool {
|
func hasCall(t *ast.Text) bool {
|
||||||
for _, stmt := range t.Body {
|
for _, stmt := range t.Body {
|
||||||
@@ -412,6 +637,40 @@ func hasCall(t *ast.Text) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// checkAdjspBalance mirrors the toolchain's push/pop walk: every ADJSP
|
||||||
|
// shifts SP away from the entry state and every RET must see the shifts
|
||||||
|
// closed. The assembler's own prologue and epilogue contribute matching
|
||||||
|
// deltas on both sides, so the statements' straight-line sum must be zero
|
||||||
|
// at each RET; branches do not reset the walk, which runs over the program
|
||||||
|
// list in source order. go tool asm reports an offender as "unbalanced
|
||||||
|
// PUSH/POP" (verified against ADJSP $16 before a RET, accepted as a
|
||||||
|
// $16/$-16 pair, per-RET rather than per-function).
|
||||||
|
func checkAdjspBalance(t *ast.Text) error {
|
||||||
|
delta := 0
|
||||||
|
for _, stmt := range t.Body {
|
||||||
|
in, ok := stmt.(*ast.Instr)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
||||||
|
case "ADJSP":
|
||||||
|
if len(in.Operands) != 1 || !in.Operands[0].Imm.HasVal {
|
||||||
|
continue // reported during emission
|
||||||
|
}
|
||||||
|
v := in.Operands[0].Imm.Val
|
||||||
|
if in.Operands[0].Imm.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
delta += int(v)
|
||||||
|
case "RET":
|
||||||
|
if delta != 0 {
|
||||||
|
return fmt.Errorf("unbalanced PUSH/POP")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// guardLen returns the byte length of the stack-split guard prefix. The
|
// guardLen returns the byte length of the stack-split guard prefix. The
|
||||||
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
|
// final conditional branch (JBE, and JB in the big class) is 2 bytes in the
|
||||||
// short form and 6 in the long form.
|
// short form and 6 in the long form.
|
||||||
@@ -538,7 +797,7 @@ func instrSize(s *ast.Instr, fi frameInfo, long bool, link *linkInfo) (int, erro
|
|||||||
}
|
}
|
||||||
return jumpSize(mnem, long), nil
|
return jumpSize(mnem, long), nil
|
||||||
}
|
}
|
||||||
code, _, err := encodeInstr(s, 0, nil, fi, false, nil, link)
|
code, _, _, err := encodeInstr(s, 0, nil, fi, false, nil, link, -1)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
@@ -573,9 +832,21 @@ func jumpSize(mnem string, long bool) int {
|
|||||||
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
// (relative to pc, the instruction's own offset). A RET in a frame-pointer
|
||||||
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
// function is prefixed with the epilogue. resolve, when non-nil, redirects a
|
||||||
// jump label through the jump-to-jump chain before the offset lookup.
|
// jump label through the jump-to-jump chain before the offset lookup.
|
||||||
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo) ([]byte, []sbPatch, error) {
|
func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, long bool, resolve func(string) string, link *linkInfo, numTarget int) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||||
mnem := strings.ToUpper(s.Mnemonic.Text)
|
mnem := strings.ToUpper(s.Mnemonic.Text)
|
||||||
|
|
||||||
|
if mnem == "PCALIGN" {
|
||||||
|
// The layout pass already accounted the padding; emit the same
|
||||||
|
// amount of NOP bytes for the statement's own position.
|
||||||
|
pad, err := pcAlignPad(pcAlignValue(s), pc)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
out := make([]byte, pad)
|
||||||
|
fillNOPs(out)
|
||||||
|
return out, nil, nil, nil
|
||||||
|
}
|
||||||
|
|
||||||
var prefix []byte
|
var prefix []byte
|
||||||
if mnem == "RET" && fi.useFP {
|
if mnem == "RET" && fi.useFP {
|
||||||
prefix = fi.epilogue
|
prefix = fi.epilogue
|
||||||
@@ -583,6 +854,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
|||||||
|
|
||||||
var code []byte
|
var code []byte
|
||||||
var ps []sbPatch
|
var ps []sbPatch
|
||||||
|
var pool []floatPoolEntry
|
||||||
var err error
|
var err error
|
||||||
if isJumpMnemonic(mnem) {
|
if isJumpMnemonic(mnem) {
|
||||||
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
if (mnem == "CALL" || mnem == "JMP") && isSBCall(s) {
|
||||||
@@ -591,7 +863,7 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
|||||||
// or the linker.
|
// or the linker.
|
||||||
code, ps, err = encodeSBCall(s, link)
|
code, ps, err = encodeSBCall(s, link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
for i := range ps {
|
for i := range ps {
|
||||||
ps[i].kind = RelCall
|
ps[i].kind = RelCall
|
||||||
@@ -601,23 +873,23 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
|||||||
ps[i].off += body
|
ps[i].off += body
|
||||||
ps[i].after = body + len(code)
|
ps[i].after = body + len(code)
|
||||||
}
|
}
|
||||||
return append(prefix, code...), ps, nil
|
return append(prefix, code...), ps, nil, nil
|
||||||
}
|
}
|
||||||
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
if (mnem == "CALL" || mnem == "JMP") && indirectJumpTarget(s) {
|
||||||
// JMP/CALL through a register or memory: no relocation and no
|
// JMP/CALL through a register or memory: no relocation and no
|
||||||
// label to resolve, the operand fully determines the bytes.
|
// label to resolve, the operand fully determines the bytes.
|
||||||
code, err = encodeIndirectJump(s, mnem)
|
code, err = encodeIndirectJump(s, mnem)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
return append(prefix, code...), nil, nil
|
return append(prefix, code...), nil, nil, nil
|
||||||
}
|
}
|
||||||
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve)
|
code, err = encodeJump(s, mnem, pc+len(prefix), offsets, long, resolve, numTarget)
|
||||||
} else {
|
} else {
|
||||||
code, ps, err = encodeNormal(s, fi, link)
|
code, ps, pool, err = encodeNormal(s, fi, link)
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
// Anchor the patch fields at function-relative positions: off indexes the
|
// Anchor the patch fields at function-relative positions: off indexes the
|
||||||
// disp32 field, after is the address just past the instruction.
|
// disp32 field, after is the address just past the instruction.
|
||||||
@@ -626,49 +898,183 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
|
|||||||
ps[i].off += body
|
ps[i].off += body
|
||||||
ps[i].after = body + len(code)
|
ps[i].after = body + len(code)
|
||||||
}
|
}
|
||||||
return append(prefix, code...), ps, nil
|
return append(prefix, code...), ps, pool, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
|
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||||
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
mnemUpper := strings.ToUpper(s.Mnemonic.Text)
|
||||||
|
if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" {
|
||||||
|
code, err := encodeBookkeeping(mnemUpper, s)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
return code, nil, nil, nil
|
||||||
|
}
|
||||||
|
// MOVQ $sym±off(SB), r64: the toolchain assembles a symbol immediate as
|
||||||
|
// LEAQ disp32(RIP), r64 with an R_PCREL relocation at the disp32 field,
|
||||||
|
// never as a 64-bit absolute immediate (verified against go tool asm).
|
||||||
|
// MOVD is the MOVQ alias; the narrower widths reject the form outright.
|
||||||
|
if (mnemUpper == "MOVQ" || mnemUpper == "MOVD") && len(s.Operands) == 2 &&
|
||||||
|
s.Operands[0].Kind == ast.OpImmediate && s.Operands[0].Imm.Sym != nil &&
|
||||||
|
s.Operands[0].Imm.Sym.Pseudo == "SB" {
|
||||||
|
mem := &ast.Operand{Kind: ast.OpAddr, Addr: ast.Address{Sym: s.Operands[0].Imm.Sym}}
|
||||||
|
src, err := operandFromAST(mnemUpper, mem, 8, fi, link)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
dst, err := operandFromAST(mnemUpper, s.Operands[1], 8, fi, link)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
e := &enc{}
|
||||||
|
if err := e.encodeLea([]Operand{src, dst}, 8); err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
ps := make([]sbPatch, len(e.patches))
|
||||||
|
for i, p := range e.patches {
|
||||||
|
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
||||||
|
}
|
||||||
|
return e.out, ps, nil, nil
|
||||||
|
}
|
||||||
|
// MOVQ/MOVL TLS, r: the bare TLS load. The toolchain's progedit nops
|
||||||
|
// it out on the one-instruction TLS systems (linux and freebsd, not
|
||||||
|
// shared) and encodes the segment-prefixed load elsewhere; get_tls(r),
|
||||||
|
// the macro GOROOT's go_tls.h defines, expands to exactly this
|
||||||
|
// statement, and the toolchain's pairing pass removes it whenever the
|
||||||
|
// following instruction's (TLS*1) index folds.
|
||||||
|
if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) {
|
||||||
|
return encodeTLSBaseLoad(s, fi, link)
|
||||||
|
}
|
||||||
|
_, size := splitSize(mnemUpper)
|
||||||
if size == 0 {
|
if size == 0 {
|
||||||
size = 8
|
size = 8
|
||||||
}
|
}
|
||||||
ops := make([]Operand, len(s.Operands))
|
ops := make([]Operand, len(s.Operands))
|
||||||
for i, op := range s.Operands {
|
for i, op := range s.Operands {
|
||||||
o, err := operandFromAST(op, size, fi, link)
|
o, err := operandFromAST(mnemUpper, op, size, fi, link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
ops[i] = o
|
ops[i] = o
|
||||||
}
|
}
|
||||||
e := &enc{}
|
e := &enc{}
|
||||||
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
if err := e.encode(s.Mnemonic.Text, ops); err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, nil, err
|
||||||
}
|
}
|
||||||
ps := make([]sbPatch, len(e.patches))
|
ps := make([]sbPatch, len(e.patches))
|
||||||
for i, p := range e.patches {
|
for i, p := range e.patches {
|
||||||
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend}
|
||||||
|
if p.tls {
|
||||||
|
ps[i].kind = RelTLSLE
|
||||||
|
}
|
||||||
}
|
}
|
||||||
return e.out, ps, nil
|
return e.out, ps, e.floatPoolList(), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// isBareTLS reports whether the operand is the bare TLS pseudo-register
|
||||||
|
// load source, the expansion of go_tls.h's get_tls(r) macro.
|
||||||
|
func isBareTLS(op *ast.Operand) bool {
|
||||||
|
return op.Kind == ast.OpAddr && op.Addr.Sym != nil &&
|
||||||
|
op.Addr.Sym.Pseudo == "" && op.Addr.Sym.Name == "TLS" &&
|
||||||
|
op.Addr.Base == "" && op.Addr.Index == ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeTLSBaseLoad assembles MOVQ/MOVL TLS, r. On the one-instruction TLS
|
||||||
|
// systems (linux and freebsd outside -shared, obj6.go's CanUse1InsnTLS) the
|
||||||
|
// statement nops out: the following (TLS*1) access folds to a direct
|
||||||
|
// segment-absolute load. The two-instruction systems keep the segment load,
|
||||||
|
// nine bytes with the R_TLSLE patch site at the disp32.
|
||||||
|
func encodeTLSBaseLoad(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
|
||||||
|
_, size := splitSize(strings.ToUpper(s.Mnemonic.Text))
|
||||||
|
if size == 0 {
|
||||||
|
size = 8
|
||||||
|
}
|
||||||
|
dst, err := operandFromAST("MOVQ", s.Operands[1], 8, fi, link)
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
reg, ok := dst.(Reg)
|
||||||
|
if !ok || reg.isVec() {
|
||||||
|
return nil, nil, nil, fmt.Errorf("TLS: destination must be a general register")
|
||||||
|
}
|
||||||
|
if link == nil || link.tlsOneInsn() {
|
||||||
|
return nil, nil, nil, nil // noped out
|
||||||
|
}
|
||||||
|
seg := byte(0x64) // FS
|
||||||
|
if link.goos == "windows" {
|
||||||
|
seg = 0x65 // GS
|
||||||
|
}
|
||||||
|
e := &enc{}
|
||||||
|
i := &instr{
|
||||||
|
prefix: seg,
|
||||||
|
rexW: size == 8,
|
||||||
|
rexR: reg.idx >= 8,
|
||||||
|
opcode: []byte{0x8B},
|
||||||
|
modrm: 0x04 | (reg.idx&7)<<3,
|
||||||
|
sib: 0x25,
|
||||||
|
disp: le32(0),
|
||||||
|
tls: true,
|
||||||
|
}
|
||||||
|
if err := e.emit(i); err != nil {
|
||||||
|
return nil, nil, nil, err
|
||||||
|
}
|
||||||
|
ps := make([]sbPatch, len(e.patches))
|
||||||
|
for i, p := range e.patches {
|
||||||
|
ps[i] = sbPatch{off: p.off, name: p.name, addend: p.addend, kind: RelTLSLE}
|
||||||
|
}
|
||||||
|
return e.out, ps, nil, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeBookkeeping accepts-and-ignores FUNCDATA and PCDATA at the statement
|
||||||
|
// level, before operand conversion: the toolchain's shapes are FUNCDATA
|
||||||
|
// $n, sym(SB) and PCDATA $n, $m, and neither contributes a byte to the
|
||||||
|
// function body. The symbol reference must not run through the SB-operand
|
||||||
|
// path, which demands file-level resolution the statement never needs.
|
||||||
|
func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
|
||||||
|
if len(s.Operands) != 2 {
|
||||||
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", upper, len(s.Operands))
|
||||||
|
}
|
||||||
|
a, b := s.Operands[0], s.Operands[1]
|
||||||
|
if a.Kind != ast.OpImmediate || !a.Imm.HasVal {
|
||||||
|
return nil, fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||||
|
}
|
||||||
|
switch upper {
|
||||||
|
case "FUNCDATA":
|
||||||
|
if b.Kind != ast.OpAddr || b.Addr.Sym == nil || b.Addr.Sym.Pseudo != "SB" {
|
||||||
|
return nil, fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||||
|
}
|
||||||
|
case "PCDATA":
|
||||||
|
if b.Kind != ast.OpImmediate || !b.Imm.HasVal {
|
||||||
|
return nil, fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
|
||||||
// target label, in the short (rel8) or long (rel32) form.
|
// target label or from a numeric ±N(PC) instruction count, in the short
|
||||||
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string) ([]byte, error) {
|
// (rel8) or long (rel32) form. numTarget is the resolved byte offset of a
|
||||||
|
// numeric operand, negative when the operand is not one.
|
||||||
|
func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long bool, resolve func(string) string, numTarget int) ([]byte, error) {
|
||||||
if len(s.Operands) != 1 {
|
if len(s.Operands) != 1 {
|
||||||
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
|
return nil, fmt.Errorf("jump expects 1 operand, got %d", len(s.Operands))
|
||||||
}
|
}
|
||||||
name, ok := labelName(s.Operands[0])
|
name, isLabel := labelName(s.Operands[0])
|
||||||
if !ok {
|
if !isLabel && numTarget < 0 {
|
||||||
return nil, fmt.Errorf("jump target must be a local label")
|
return nil, fmt.Errorf("jump target must be a local label")
|
||||||
}
|
}
|
||||||
if resolve != nil && mnem != "CALL" {
|
var target int
|
||||||
name = resolve(name)
|
if isLabel {
|
||||||
}
|
if resolve != nil && mnem != "CALL" {
|
||||||
target, ok := offsets[name]
|
name = resolve(name)
|
||||||
if !ok {
|
}
|
||||||
return nil, fmt.Errorf("undefined label %q", name)
|
t, ok := offsets[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("undefined label %q", name)
|
||||||
|
}
|
||||||
|
target = t
|
||||||
|
} else {
|
||||||
|
target = numTarget
|
||||||
}
|
}
|
||||||
rel := int64(target - (pc + jumpSize(mnem, long)))
|
rel := int64(target - (pc + jumpSize(mnem, long)))
|
||||||
|
|
||||||
@@ -701,7 +1107,7 @@ func isSBCall(s *ast.Instr) bool {
|
|||||||
|
|
||||||
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
|
// encodeSBCall encodes CALL sym(SB) as E8 rel32 with a patch site.
|
||||||
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
|
func encodeSBCall(s *ast.Instr, link *linkInfo) ([]byte, []sbPatch, error) {
|
||||||
o, err := operandFromAST(s.Operands[0], 8, frameInfo{}, link)
|
o, err := operandFromAST(strings.ToUpper(s.Mnemonic.Text), s.Operands[0], 8, frameInfo{}, link)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, nil, err
|
return nil, nil, err
|
||||||
}
|
}
|
||||||
@@ -742,6 +1148,11 @@ func indirectJumpTarget(s *ast.Instr) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
a := s.Operands[0].Addr
|
a := s.Operands[0].Addr
|
||||||
|
// ±N(PC) is the numeric relative form, the PC counts instructions from
|
||||||
|
// the branch: relative, not indirect.
|
||||||
|
if a.Base == "PC" || a.Index == "PC" {
|
||||||
|
return false
|
||||||
|
}
|
||||||
if a.Base != "" || a.Index != "" {
|
if a.Base != "" || a.Index != "" {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
@@ -758,7 +1169,7 @@ func indirectJumpTarget(s *ast.Instr) bool {
|
|||||||
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
||||||
ops := make([]Operand, len(s.Operands))
|
ops := make([]Operand, len(s.Operands))
|
||||||
for i, op := range s.Operands {
|
for i, op := range s.Operands {
|
||||||
o, err := operandFromAST(op, 8, frameInfo{}, nil)
|
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -775,8 +1186,11 @@ func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
|
|||||||
var spReg = Reg{idx: 4, size: 8}
|
var spReg = Reg{idx: 4, size: 8}
|
||||||
|
|
||||||
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
// operandFromAST converts a parsed operand into an encoder Operand, applying
|
||||||
// the frame translation to FP/SP pseudo-register operands.
|
// the frame translation to FP/SP pseudo-register operands. mnemUpper is the
|
||||||
func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
// instruction's upper-case mnemonic, which the floating-point immediate gate
|
||||||
|
// needs: only the SSE mnemonics whose encoding takes an XMM/memory source
|
||||||
|
// accept one.
|
||||||
|
func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Operand, error) {
|
||||||
switch op.Kind {
|
switch op.Kind {
|
||||||
case ast.OpImmediate:
|
case ast.OpImmediate:
|
||||||
if op.Imm.HasVal {
|
if op.Imm.HasVal {
|
||||||
@@ -786,11 +1200,45 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
|||||||
}
|
}
|
||||||
return Imm(v), nil
|
return Imm(v), nil
|
||||||
}
|
}
|
||||||
|
// A floating-point immediate: $1.5, $-1.0 or the parenthesised
|
||||||
|
// $(-1.0) spelling (the constant-expression folder only folds
|
||||||
|
// integers, so that shape arrives with an empty Immediate and only
|
||||||
|
// the raw spelling carries the value). The toolchain rewrites it
|
||||||
|
// into a pooled-constant read on the SSE scalar paths and rejects
|
||||||
|
// it everywhere else.
|
||||||
|
if text, neg, ok := floatImmText(op); ok {
|
||||||
|
if !sseFloatImm[mnemUpper] {
|
||||||
|
return nil, fmt.Errorf("%s does not take a floating-point immediate", mnemUpper)
|
||||||
|
}
|
||||||
|
return FloatImm{Text: text, Neg: neg}, nil
|
||||||
|
}
|
||||||
return nil, fmt.Errorf("non-integer immediate not supported")
|
return nil, fmt.Errorf("non-integer immediate not supported")
|
||||||
|
|
||||||
case ast.OpAddr:
|
case ast.OpAddr:
|
||||||
a := op.Addr
|
a := op.Addr
|
||||||
|
|
||||||
|
// A bracketed register range, [Z0-Z3]: the four-register source of
|
||||||
|
// the 4FMAPS/4VNNIW families. The range must span four consecutive
|
||||||
|
// same-width vector registers, exactly what the toolchain's parser
|
||||||
|
// takes; the EVEX quad-register emit path reads the low end.
|
||||||
|
if a.Range != nil {
|
||||||
|
lo, ok := ParseReg(a.Range.Lo)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("unknown register %q in range", a.Range.Lo)
|
||||||
|
}
|
||||||
|
hi, ok := ParseReg(a.Range.Hi)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("unknown register %q in range", a.Range.Hi)
|
||||||
|
}
|
||||||
|
if !lo.isVec() || lo.size != hi.size {
|
||||||
|
return nil, fmt.Errorf("register range %q must span four same-width vector registers", op.Raw)
|
||||||
|
}
|
||||||
|
if hi.idx != lo.idx+3 {
|
||||||
|
return nil, fmt.Errorf("register range %q must span four consecutive registers", op.Raw)
|
||||||
|
}
|
||||||
|
return RegList{Lo: lo, Hi: hi}, nil
|
||||||
|
}
|
||||||
|
|
||||||
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
|
// FP-relative: x+N(FP) → (N + fpAdjust)(SP). The offset N lives in the
|
||||||
// symbol, not the address displacement.
|
// symbol, not the address displacement.
|
||||||
if a.Sym != nil && a.Sym.Pseudo == "FP" {
|
if a.Sym != nil && a.Sym.Pseudo == "FP" {
|
||||||
@@ -822,12 +1270,44 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
|||||||
|
|
||||||
// Memory with a real base register: (base), off(base), (base)(index*scale).
|
// Memory with a real base register: (base), off(base), (base)(index*scale).
|
||||||
if a.Base != "" {
|
if a.Base != "" {
|
||||||
|
// Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS
|
||||||
|
// spellings. The segment override prefixes a disp32 absolute
|
||||||
|
// reference with no relocation.
|
||||||
|
if a.Base == "GS" || a.Base == "FS" {
|
||||||
|
seg := byte(0x64)
|
||||||
|
if a.Base == "GS" {
|
||||||
|
seg = 0x65
|
||||||
|
}
|
||||||
|
return SegAbs{Disp: a.Offset, Size: size, Seg: seg}, nil
|
||||||
|
}
|
||||||
base, ok := ParseReg(a.Base)
|
base, ok := ParseReg(a.Base)
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("unknown base register %q", a.Base)
|
return nil, fmt.Errorf("unknown base register %q", a.Base)
|
||||||
}
|
}
|
||||||
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
|
m := Mem{Base: base, Disp: a.Offset, HasBase: true, Size: size}
|
||||||
if a.Index != "" {
|
if a.Index != "" {
|
||||||
|
if a.Index == "TLS" {
|
||||||
|
// off(base)(TLS*1): the thread-local annotation. The
|
||||||
|
// one-instruction TLS form folds it to off(TLS), the
|
||||||
|
// segment-prefixed absolute whose disp32 carries an
|
||||||
|
// R_TLS_LE patch site; the base register disappears
|
||||||
|
// from the encoding, exactly as the toolchain's
|
||||||
|
// progedit rewrites the address.
|
||||||
|
seg := byte(0x64) // FS on linux, freebsd, plan9
|
||||||
|
if link != nil && link.goos == "windows" {
|
||||||
|
seg = 0x65 // GS
|
||||||
|
}
|
||||||
|
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
|
||||||
|
}
|
||||||
|
if a.Index == "GS" || a.Index == "FS" {
|
||||||
|
// 0(CX)(GS): the segment annotation rides the base
|
||||||
|
// access as the override prefix.
|
||||||
|
m.Seg = 0x64
|
||||||
|
if a.Index == "GS" {
|
||||||
|
m.Seg = 0x65
|
||||||
|
}
|
||||||
|
return m, nil
|
||||||
|
}
|
||||||
idx, ok := ParseReg(a.Index)
|
idx, ok := ParseReg(a.Index)
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("unknown index register %q", a.Index)
|
return nil, fmt.Errorf("unknown index register %q", a.Index)
|
||||||
@@ -838,6 +1318,21 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
|||||||
}
|
}
|
||||||
return m, nil
|
return m, nil
|
||||||
}
|
}
|
||||||
|
// Index-only memory: the VSIB form the gather/scatter families
|
||||||
|
// read, 8(X4*1). A scaled vector index addresses memory with no
|
||||||
|
// base register; the mod=00 SIB with base field 101 carries it.
|
||||||
|
if a.Index != "" {
|
||||||
|
idx, ok := ParseReg(a.Index)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("unknown index register %q", a.Index)
|
||||||
|
}
|
||||||
|
return Mem{Index: idx, Scale: a.Scale, Disp: a.Offset, HasIndex: true, Size: size}, nil
|
||||||
|
}
|
||||||
|
// A bare displacement with no base: the absolute address form,
|
||||||
|
// MOVL $0xf1, 0xf1. No segment and no relocation.
|
||||||
|
if a.Sym == nil && a.Base == "" && a.Index == "" && a.HasOff {
|
||||||
|
return SegAbs{Disp: a.Offset, Size: size}, nil
|
||||||
|
}
|
||||||
// Bare register.
|
// Bare register.
|
||||||
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
|
if a.Sym != nil && a.Sym.Pseudo == "" && a.Sym.Name != "" {
|
||||||
if r, ok := ParseReg(a.Sym.Name); ok {
|
if r, ok := ParseReg(a.Sym.Name); ok {
|
||||||
@@ -848,3 +1343,37 @@ func operandFromAST(op *ast.Operand, size int, fi frameInfo, link *linkInfo) (Op
|
|||||||
}
|
}
|
||||||
return nil, fmt.Errorf("unsupported operand")
|
return nil, fmt.Errorf("unsupported operand")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// floatImmText recovers a floating-point immediate's magnitude and sign from
|
||||||
|
// the parsed operand. The ordinary spellings arrive in Imm.Float; the
|
||||||
|
// parenthesised $(-1.0) leaves the Immediate empty, because the integer
|
||||||
|
// folder cannot read it, and only the verbatim operand text still carries
|
||||||
|
// the value. Anything that is not a number a float parser accepts reports
|
||||||
|
// not-ok, so every other shape keeps its existing diagnostic.
|
||||||
|
func floatImmText(op *ast.Operand) (text string, neg bool, ok bool) {
|
||||||
|
if op.Imm.Float != "" {
|
||||||
|
return op.Imm.Float, op.Imm.Neg, true
|
||||||
|
}
|
||||||
|
if op.Imm.HasVal || op.Imm.Str != "" || op.Imm.Sym != nil {
|
||||||
|
return "", false, false
|
||||||
|
}
|
||||||
|
// joinRaw spaced the token texts; the compact spelling is what matters.
|
||||||
|
compact := strings.ReplaceAll(op.Raw, " ", "")
|
||||||
|
inner, ok := strings.CutPrefix(compact, "$(")
|
||||||
|
if !ok || !strings.HasSuffix(inner, ")") {
|
||||||
|
return "", false, false
|
||||||
|
}
|
||||||
|
inner = strings.TrimSuffix(inner, ")")
|
||||||
|
inner = strings.TrimPrefix(inner, "+")
|
||||||
|
if s, ok := strings.CutPrefix(inner, "-"); ok {
|
||||||
|
neg = true
|
||||||
|
inner = s
|
||||||
|
}
|
||||||
|
if inner == "" || !strings.ContainsAny(inner, "0123456789") {
|
||||||
|
return "", false, false
|
||||||
|
}
|
||||||
|
if _, err := strconv.ParseFloat(inner, 64); err != nil {
|
||||||
|
return "", false, false
|
||||||
|
}
|
||||||
|
return inner, neg, true
|
||||||
|
}
|
||||||
|
|||||||
+208
-2
@@ -10,8 +10,8 @@ import (
|
|||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// firstText parses src and returns its first TEXT function.
|
// firstText parses src and returns its first TEXT function.
|
||||||
@@ -370,6 +370,58 @@ end:
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
|
||||||
|
// counts instruction statements, skipping labels, in both directions (the
|
||||||
|
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
|
||||||
|
func TestAssembleNumericPCJumps(t *testing.T) {
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·exit(SB), NOSPLIT, $0
|
||||||
|
MOVB $1, AL
|
||||||
|
lab:
|
||||||
|
MOVB $2, AL
|
||||||
|
MOVB $3, AL
|
||||||
|
JMP -3(PC)
|
||||||
|
MOVB $4, AL
|
||||||
|
park:
|
||||||
|
JMP 0(PC)
|
||||||
|
MOVB $5, AL
|
||||||
|
JMP 2(PC)
|
||||||
|
MOVB $6, AL
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
// From the Go-assembled function:
|
||||||
|
// MOVB $1, AL b001
|
||||||
|
// MOVB $2, AL b002
|
||||||
|
// MOVB $3, AL b003
|
||||||
|
// JMP -3(PC) ebf8 (three instructions back, past lab:)
|
||||||
|
// MOVB $4, AL b004
|
||||||
|
// JMP 0(PC) ebfe (the park loop)
|
||||||
|
// MOVB $5, AL b005
|
||||||
|
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
|
||||||
|
// MOVB $6, AL b006
|
||||||
|
// RET c3
|
||||||
|
want := []byte{
|
||||||
|
0xb0, 0x01,
|
||||||
|
0xb0, 0x02,
|
||||||
|
0xb0, 0x03,
|
||||||
|
0xeb, 0xf8,
|
||||||
|
0xb0, 0x04,
|
||||||
|
0xeb, 0xfe,
|
||||||
|
0xb0, 0x05,
|
||||||
|
0xeb, 0x02,
|
||||||
|
0xb0, 0x06,
|
||||||
|
0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestAssemblePrefetch(t *testing.T) {
|
func TestAssemblePrefetch(t *testing.T) {
|
||||||
fn := firstText(t, `
|
fn := firstText(t, `
|
||||||
#include "textflag.h"
|
#include "textflag.h"
|
||||||
@@ -439,3 +491,157 @@ func TestSubSPEncodings(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAssemblePseudoStatements runs LOCK/REP, BYTE/WORD and END through the
|
||||||
|
// full statement pipeline, pinned against go tool asm (Go 1.27, amd64). It
|
||||||
|
// asserts the three behaviours the toolchain shows: each prefix statement is
|
||||||
|
// a standalone byte with a PC of its own (so a label placed on the LOCK
|
||||||
|
// points at the F0), the data pseudo-ops write their literal bytes inline,
|
||||||
|
// and END terminates nothing (the statements after it still belong to the
|
||||||
|
// function and carry no trace of it).
|
||||||
|
func TestAssemblePseudoStatements(t *testing.T) {
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·pseudo(SB), NOSPLIT, $0-0
|
||||||
|
pfx:
|
||||||
|
LOCK
|
||||||
|
CMPXCHGQ AX, (BX)
|
||||||
|
REP
|
||||||
|
MOVSQ
|
||||||
|
BYTE $0x0f
|
||||||
|
BYTE $0x1f
|
||||||
|
WORD $0x1234
|
||||||
|
END
|
||||||
|
BYTE $0x02
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, labels, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
// go tool asm: f0 480fb103 f3 48a5 0f 1f 3412 02 c3
|
||||||
|
want := []byte{
|
||||||
|
0xf0,
|
||||||
|
0x48, 0x0f, 0xb1, 0x03,
|
||||||
|
0xf3, 0x48, 0xa5,
|
||||||
|
0x0f, 0x1f, 0x34, 0x12,
|
||||||
|
0x02, 0xc3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("pseudo statements:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
// The label sits on the LOCK byte, exactly where the toolchain's PC
|
||||||
|
// listing puts it.
|
||||||
|
if off := labels["pfx"]; off != 0 {
|
||||||
|
t.Errorf("label pfx = %d, want 0 (the LOCK's own byte)", off)
|
||||||
|
}
|
||||||
|
// The trailing BYTE lands where the layout says: after the 8 bytes of
|
||||||
|
// LOCK, CMPXCHGQ, REP and MOVSQ plus the 4 data bytes, END contributing
|
||||||
|
// none.
|
||||||
|
if code[12] != 0x02 {
|
||||||
|
t.Errorf("byte at 12 = %02x, want 02 (the BYTE after END)", code[12])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAssembleAdjspBalance pins the toolchain's push/pop balance rule over
|
||||||
|
// ADJSP: the straight-line sum of the adjustments must be zero at each
|
||||||
|
// RET, branches in between counting for nothing (verified against go tool
|
||||||
|
// asm: ADJSP $16 before a RET is reported as "unbalanced PUSH/POP", a
|
||||||
|
// $16/$-16 pair with a JMP in between assembles).
|
||||||
|
func TestAssembleAdjspBalance(t *testing.T) {
|
||||||
|
// Balanced pair with a branch in between, bytes pinned from go tool asm.
|
||||||
|
fn := firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·adjsp(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $16
|
||||||
|
JMP body
|
||||||
|
body:
|
||||||
|
ADJSP $-16
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err := Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble: %v", err)
|
||||||
|
}
|
||||||
|
want := []byte{0x48, 0x83, 0xEC, 0x10, 0xEB, 0x00, 0x48, 0x83, 0xC4, 0x10, 0xC3}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("adjsp pair:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unbalanced at the RET: the toolchain diagnoses, so must we.
|
||||||
|
_, _, err = Assemble(firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·unbalanced(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $16
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||||
|
t.Errorf("unbalanced ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The check runs per RET: a closed pair before the first RET does not
|
||||||
|
// excuse an open adjustment before the second.
|
||||||
|
_, _, err = Assemble(firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·tworet(SB), NOSPLIT, $0-0
|
||||||
|
ADJSP $8
|
||||||
|
ADJSP $-8
|
||||||
|
RET
|
||||||
|
mid:
|
||||||
|
ADJSP $8
|
||||||
|
RET
|
||||||
|
`))
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "unbalanced PUSH/POP") {
|
||||||
|
t.Errorf("second RET with open ADJSP: err = %v, want unbalanced PUSH/POP", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A framed function: the assembler's own prologue and epilogue
|
||||||
|
// contribute matching deltas, so the pair in the body still balances,
|
||||||
|
// and the bytes match go tool asm end to end.
|
||||||
|
fn = firstText(t, `
|
||||||
|
#include "textflag.h"
|
||||||
|
TEXT ·framed(SB), $16-8
|
||||||
|
ADJSP $8
|
||||||
|
ADJSP $-8
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, err = Assemble(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Assemble framed: %v", err)
|
||||||
|
}
|
||||||
|
want = []byte{
|
||||||
|
0x55, 0x48, 0x89, 0xE5, 0x48, 0x83, 0xEC, 0x10, // prologue
|
||||||
|
0x48, 0x83, 0xEC, 0x08, // ADJSP $8
|
||||||
|
0x48, 0x83, 0xC4, 0x08, // ADJSP $-8
|
||||||
|
0x48, 0x83, 0xC4, 0x10, 0x5D, // epilogue
|
||||||
|
0xC3,
|
||||||
|
}
|
||||||
|
if hexBytes(code) != hexBytes(want) {
|
||||||
|
t.Errorf("framed adjsp:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAssembleRegRange pins the bracketed register range at the statement
|
||||||
|
// level: exactly four consecutive same-width vector registers assemble, the
|
||||||
|
// toolchain's rejected shapes all report an error.
|
||||||
|
func TestAssembleRegRange(t *testing.T) {
|
||||||
|
asm := func(t *testing.T, op string) ([]byte, error) {
|
||||||
|
t.Helper()
|
||||||
|
f, errs := parser.Parse("f_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tV4FMADDPS 17(SP), "+op+", K2, Z0\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse %s: %v", op, errs)
|
||||||
|
}
|
||||||
|
code, _, err := Assemble(f.Decls[0].(*ast.Text))
|
||||||
|
return code, err
|
||||||
|
}
|
||||||
|
for _, op := range []string{"[Z0-Z3]", "[Z4-Z7]", "[Z28-Z31]"} {
|
||||||
|
if _, err := asm(t, op); err != nil {
|
||||||
|
t.Errorf("%s: %v", op, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, op := range []string{"[Z0-Z4]", "[Z0-Z2]", "[Z0-Z0]", "[Z4-Z0]", "[Z1-Z0]", "[AX-Z3]", "[Z0-AX]"} {
|
||||||
|
if _, err := asm(t, op); err == nil {
|
||||||
|
t.Errorf("%s: assembled, want an error", op)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+65
-10
@@ -43,6 +43,9 @@ const (
|
|||||||
stInfoShift = 4
|
stInfoShift = 4
|
||||||
|
|
||||||
rX8664PC32 = 2
|
rX8664PC32 = 2
|
||||||
|
// R_X86_64_32 (debug/elf): the absolute 32-bit address of a symbol, the
|
||||||
|
// R_ADDR shape a 4-byte DATA field carries.
|
||||||
|
rX8664Abs32 = 10
|
||||||
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
|
// R_X86_64_TPOFF32 (debug/elf): the local-exec TLS offset the stack
|
||||||
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
|
// guard loads from FS. 20 is R_X86_64_TLSLD, a different relocation.
|
||||||
rX8664TPOFF32 = 23
|
rX8664TPOFF32 = 23
|
||||||
@@ -158,6 +161,50 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rX8664Abs64
|
||||||
|
case 4:
|
||||||
|
typ = rX8664Abs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// Serialise the string tables.
|
// Serialise the string tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -167,19 +214,13 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Section presence: .rela.text only when there are relocations.
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6 // NULL, .text, .data, .symtab, .strtab, .shstrtab
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Lay the file out: header, section data, section headers.
|
// Lay the file out: header, section data, section headers.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...) // ELF header, filled last
|
out = append(out, make([]byte, 64)...) // ELF header, filled last
|
||||||
@@ -214,7 +255,7 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -226,6 +267,17 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
@@ -284,6 +336,9 @@ func (img *Image) ELFObject() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
|
|
||||||
// DWARF section headers; their indices follow the write order.
|
// DWARF section headers; their indices follow the write order.
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
|
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
|
||||||
|
|||||||
+110
-2
@@ -12,8 +12,8 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// The object-file tests share one source: two exported functions, one
|
// The object-file tests share one source: two exported functions, one
|
||||||
@@ -750,3 +750,111 @@ func readFormSkip(t *testing.T, r *ulebIter, form uint64) {
|
|||||||
t.Fatalf("unsupported form %#x", form)
|
t.Fatalf("unsupported form %#x", form)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFObjectDataRelocation checks that a symbol-valued DATA field ("DATA
|
||||||
|
// s+0(SB)/8, $other(SB)") reaches the ELF object as a .rela.data entry: an
|
||||||
|
// absolute 64-bit relocation at the field's offset within .data, against
|
||||||
|
// the named symbol, external targets included.
|
||||||
|
func TestELFObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_amd64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $24
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFObject: %v", err)
|
||||||
|
}
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_X86_64_64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_X86_64_64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_X86_64_64), addend: 0, target: "extvar"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+68
-11
@@ -18,12 +18,16 @@ const (
|
|||||||
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
|
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
|
||||||
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
|
||||||
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
|
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
|
||||||
|
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
|
||||||
|
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
|
||||||
|
// lives with the DWARF fixup constants as rAARCH64Abs64.
|
||||||
|
rArm64Abs32 = 258
|
||||||
)
|
)
|
||||||
|
|
||||||
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
// ELFAARCH64Object returns the image as an ELF64 relocatable object file for
|
||||||
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
// AArch64 (EM_AARCH64, 64-bit, little-endian). The structure mirrors the
|
||||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||||
// optional .rela.text.
|
// optional .rela.text and an optional .rela.data.
|
||||||
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -133,6 +137,50 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rAARCH64Abs64
|
||||||
|
case 4:
|
||||||
|
typ = rArm64Abs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// String tables.
|
// String tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -142,18 +190,13 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Layout.
|
// Layout.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...)
|
out = append(out, make([]byte, 64)...)
|
||||||
@@ -188,7 +231,7 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -200,6 +243,17 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
@@ -257,6 +311,9 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
// DWARF section headers; their indices follow the write order.
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
|||||||
+116
-1
@@ -9,7 +9,7 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
|
||||||
@@ -197,3 +197,118 @@ TEXT ·add(SB), NOSPLIT, $0-24
|
|||||||
t.Error("unexpected .rela.text section when there are no relocations")
|
t.Error("unexpected .rela.text section when there are no relocations")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFAARCH64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)") reaches the AArch64 ELF object as a
|
||||||
|
// .rela.data entry: an R_AARCH64_ABS64 (ABS32 for a width-4 field) at the
|
||||||
|
// field's offset within .data, against the named symbol, external targets
|
||||||
|
// included.
|
||||||
|
func TestELFAARCH64ObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_arm64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
DATA holder+24(SB)/4, $Keep(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFAARCH64Object()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFAARCH64Object: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_AARCH64_ABS64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_AARCH64_ABS64), addend: 0, target: "extvar"},
|
||||||
|
{off: base + 24, typ: uint32(elf.R_AARCH64_ABS32), addend: 0, target: "Keep"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+68
-11
@@ -23,12 +23,16 @@ const (
|
|||||||
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i)
|
||||||
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st)
|
||||||
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
|
rLarchB26 = 66 // R_LARCH_B26 (b/bl, matches the Go linker's mapping)
|
||||||
|
// R_LARCH_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||||
|
// the R_ADDR shape a 4-byte DATA field carries. R_LARCH_64 (2) lives
|
||||||
|
// with the DWARF fixup constants as rLarchAbs64.
|
||||||
|
rLarchAbs32 = 1
|
||||||
)
|
)
|
||||||
|
|
||||||
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
// ELFLOONG64Object returns the image as an ELF64 relocatable object file for
|
||||||
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the
|
||||||
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an
|
// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab, an
|
||||||
// optional .rela.text.
|
// optional .rela.text and an optional .rela.data.
|
||||||
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -117,6 +121,50 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rLarchAbs64
|
||||||
|
case 4:
|
||||||
|
typ = rLarchAbs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// String tables.
|
// String tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -126,18 +174,13 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Layout.
|
// Layout.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...)
|
out = append(out, make([]byte, 64)...)
|
||||||
@@ -172,7 +215,7 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -184,6 +227,17 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
@@ -239,6 +293,9 @@ func (img *Image) ELFLOONG64Object() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
// DWARF section headers; their indices follow the write order.
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
|||||||
+116
-1
@@ -9,7 +9,7 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
|
||||||
@@ -245,3 +245,118 @@ func TestELFLOONG64BranchRelocation(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestELFLOONG64ObjectDataRelocation checks that a symbol-valued DATA field
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)") reaches the LoongArch ELF object as a
|
||||||
|
// .rela.data entry: an R_LARCH_64 (R_LARCH_32 for a width-4 field) at the
|
||||||
|
// field's offset within .data, against the named symbol, external targets
|
||||||
|
// included.
|
||||||
|
func TestELFLOONG64ObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_loong64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
DATA holder+24(SB)/4, $Keep(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileLOONG64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileLOONG64: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFLOONG64Object()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFLOONG64Object: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_LARCH_64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_LARCH_64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_LARCH_64), addend: 0, target: "extvar"},
|
||||||
|
{off: base + 24, typ: uint32(elf.R_LARCH_32), addend: 0, target: "Keep"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+68
-10
@@ -24,11 +24,16 @@ const (
|
|||||||
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
|
||||||
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
|
||||||
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
|
||||||
|
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
|
||||||
|
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
|
||||||
|
// with the DWARF fixup constants as rRISCVAbs64.
|
||||||
|
rRISVCAbs32 = 1
|
||||||
)
|
)
|
||||||
|
|
||||||
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
// ELFRISCVObject returns the image as an ELF64 relocatable object file for
|
||||||
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
// RISC-V (EM_RISCV, 64-bit, little-endian). The structure mirrors the amd64
|
||||||
// ELF emission: .text, .data, .symtab, .strtab and optional .rela.text.
|
// ELF emission: .text, .data, .symtab, .strtab, an optional .rela.text and
|
||||||
|
// an optional .rela.data.
|
||||||
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
func (img *Image) ELFRISCVObject() ([]byte, error) {
|
||||||
le := binary.LittleEndian
|
le := binary.LittleEndian
|
||||||
|
|
||||||
@@ -129,6 +134,50 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The data symbols' symbol-valued DATA fields ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)") become .rela.data entries: an absolute relocation of the
|
||||||
|
// DATA line's width at the field's data-section offset, S + A with no
|
||||||
|
// PC term. Widths 4 and 8 have ELF relocation shapes; narrower fields
|
||||||
|
// cannot hold an address, so they are refused rather than truncated.
|
||||||
|
var dataRelas []elfRela
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
idx, ok := symIdx[r.Name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("data relocation references unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
var typ uint32
|
||||||
|
switch r.Siz {
|
||||||
|
case 8:
|
||||||
|
typ = rRISCVAbs64
|
||||||
|
case 4:
|
||||||
|
typ = rRISVCAbs32
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: a symbol value of width %d has no ELF relocation", d.Name, r.Siz)
|
||||||
|
}
|
||||||
|
dataRelas = append(dataRelas, elfRela{
|
||||||
|
off: uint64(d.Offset + r.Off),
|
||||||
|
sym: idx,
|
||||||
|
typ: typ,
|
||||||
|
addend: r.Addend,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Section presence: .rela.text only when there are code relocations,
|
||||||
|
// .rela.data only when a DATA line holds a symbol value.
|
||||||
|
hasRela := len(relas) > 0
|
||||||
|
hasDataRela := len(dataRelas) > 0
|
||||||
|
nSections := 6
|
||||||
|
if hasRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
nSections++
|
||||||
|
}
|
||||||
|
secSymtab, secStrtab := 3, 4
|
||||||
|
secShstr := nSections - 1
|
||||||
|
|
||||||
// String tables.
|
// String tables.
|
||||||
stNames := newElfStrtab()
|
stNames := newElfStrtab()
|
||||||
for _, s := range syms {
|
for _, s := range syms {
|
||||||
@@ -138,18 +187,13 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
stSections.add(".rela.data")
|
||||||
|
}
|
||||||
for _, n := range dwarfSectionNames {
|
for _, n := range dwarfSectionNames {
|
||||||
stSections.add(n)
|
stSections.add(n)
|
||||||
}
|
}
|
||||||
|
|
||||||
hasRela := len(relas) > 0
|
|
||||||
nSections := 6
|
|
||||||
if hasRela {
|
|
||||||
nSections = 7
|
|
||||||
}
|
|
||||||
secSymtab, secStrtab := 3, 4
|
|
||||||
secShstr := nSections - 1
|
|
||||||
|
|
||||||
// Layout.
|
// Layout.
|
||||||
var out []byte
|
var out []byte
|
||||||
out = append(out, make([]byte, 64)...)
|
out = append(out, make([]byte, 64)...)
|
||||||
@@ -184,7 +228,7 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
strtabOff := len(out)
|
strtabOff := len(out)
|
||||||
out = append(out, stNames.bytes()...)
|
out = append(out, stNames.bytes()...)
|
||||||
|
|
||||||
var relaOff int
|
var relaOff, relaDataOff int
|
||||||
if hasRela {
|
if hasRela {
|
||||||
align(8)
|
align(8)
|
||||||
relaOff = len(out)
|
relaOff = len(out)
|
||||||
@@ -196,6 +240,17 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
out = append(out, b[:]...)
|
out = append(out, b[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
align(8)
|
||||||
|
relaDataOff = len(out)
|
||||||
|
for _, r := range dataRelas {
|
||||||
|
var b [24]byte
|
||||||
|
le.PutUint64(b[0:], r.off)
|
||||||
|
le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ))
|
||||||
|
le.PutUint64(b[16:], uint64(r.addend))
|
||||||
|
out = append(out, b[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
shstrOff := len(out)
|
shstrOff := len(out)
|
||||||
out = append(out, stSections.bytes()...)
|
out = append(out, stSections.bytes()...)
|
||||||
@@ -251,6 +306,9 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
|
|||||||
if hasRela {
|
if hasRela {
|
||||||
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24)
|
||||||
}
|
}
|
||||||
|
if hasDataRela {
|
||||||
|
putSh(".rela.data", shtRela, 0, relaDataOff, 24*len(dataRelas), secSymtab, secData, 8, 24)
|
||||||
|
}
|
||||||
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0)
|
||||||
// DWARF section headers; their indices follow the write order.
|
// DWARF section headers; their indices follow the write order.
|
||||||
if dw != nil {
|
if dw != nil {
|
||||||
|
|||||||
@@ -0,0 +1,128 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"debug/elf"
|
||||||
|
"encoding/binary"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)") reaches the RISC-V ELF object as a
|
||||||
|
// .rela.data entry: an R_RISCV_64 (R_RISCV_32 for a width-4 field) at the
|
||||||
|
// field's offset within .data, against the named symbol, external targets
|
||||||
|
// included.
|
||||||
|
func TestELFRISCVObjectDataRelocation(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("t_riscv64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+8(SB)/8, $holder(SB)
|
||||||
|
DATA holder+16(SB)/8, $extvar(SB)
|
||||||
|
DATA holder+24(SB)/4, $Keep(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.ELFRISCVObject()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ELFRISCVObject: %v", err)
|
||||||
|
}
|
||||||
|
checkELFSectionAccounting(t, obj)
|
||||||
|
ef, err := elf.NewFile(bytes.NewReader(obj))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse emitted object: %v", err)
|
||||||
|
}
|
||||||
|
defer ef.Close()
|
||||||
|
relaData := ef.Section(".rela.data")
|
||||||
|
if relaData == nil {
|
||||||
|
t.Fatal("missing .rela.data section")
|
||||||
|
}
|
||||||
|
if relaData.Type != elf.SHT_RELA {
|
||||||
|
t.Errorf(".rela.data type = %v, want SHT_RELA", relaData.Type)
|
||||||
|
}
|
||||||
|
if relaData.Link == 0 || ef.Sections[relaData.Link].Name != ".symtab" {
|
||||||
|
t.Errorf(".rela.data sh_link = %d, want the .symtab index", relaData.Link)
|
||||||
|
}
|
||||||
|
if ef.Sections[relaData.Info].Name != ".data" {
|
||||||
|
t.Errorf(".rela.data sh_info = %d, want the .data index", relaData.Info)
|
||||||
|
}
|
||||||
|
relas, err := relaData.Data()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var got []struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}
|
||||||
|
for i := 0; i+24 <= len(relas); i += 24 {
|
||||||
|
got = append(got, struct {
|
||||||
|
off uint64
|
||||||
|
sym uint32
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
}{
|
||||||
|
off: binary.LittleEndian.Uint64(relas[i:]),
|
||||||
|
// r_info packs the type in the low dword and the symbol index
|
||||||
|
// in the high dword.
|
||||||
|
typ: binary.LittleEndian.Uint32(relas[i+8:]),
|
||||||
|
sym: binary.LittleEndian.Uint32(relas[i+12:]),
|
||||||
|
addend: int64(binary.LittleEndian.Uint64(relas[i+16:])),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
// debug/elf hides the table's null entry, so raw index s names syms[s-1].
|
||||||
|
syms, err := ef.Symbols()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
name := func(idx uint32) string {
|
||||||
|
if idx >= 1 && int(idx) <= len(syms) {
|
||||||
|
return syms[idx-1].Name
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// The offsets are data-section-relative: the field's DATA offset plus
|
||||||
|
// the symbol's position in .data (the layout aligns each symbol to 16).
|
||||||
|
base := uint64(0)
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
if d.Name == "holder" {
|
||||||
|
base = uint64(d.Offset)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
off uint64
|
||||||
|
typ uint32
|
||||||
|
addend int64
|
||||||
|
target string
|
||||||
|
}{
|
||||||
|
{off: base + 0, typ: uint32(elf.R_RISCV_64), addend: 5, target: "Keep"},
|
||||||
|
{off: base + 8, typ: uint32(elf.R_RISCV_64), addend: 0, target: "holder"},
|
||||||
|
{off: base + 16, typ: uint32(elf.R_RISCV_64), addend: 0, target: "extvar"},
|
||||||
|
{off: base + 24, typ: uint32(elf.R_RISCV_32), addend: 0, target: "Keep"},
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf(".rela.data entries = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := got[i]
|
||||||
|
if g.off != w.off || g.typ != w.typ || g.addend != w.addend {
|
||||||
|
t.Errorf("entry %d = {off %d typ %d addend %d}, want {off %d typ %d addend %d}",
|
||||||
|
i, g.off, g.typ, g.addend, w.off, w.typ, w.addend)
|
||||||
|
}
|
||||||
|
if n := name(g.sym); n != w.target {
|
||||||
|
t.Errorf("entry %d names %q, want %q", i, n, w.target)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+35
-8
@@ -18,7 +18,14 @@ func Encodable(mnemonic string) bool {
|
|||||||
|
|
||||||
// Fixed-name instructions (no size suffix).
|
// Fixed-name instructions (no size suffix).
|
||||||
switch upper {
|
switch upper {
|
||||||
case "RET", "NOP", "CALL", "JMP":
|
case "RET", "NOP", "CALL", "JMP",
|
||||||
|
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
|
||||||
|
// The literal-data pseudo-ops, the accepted-and-ignored END and
|
||||||
|
// bookkeeping statements, and the SP adjust.
|
||||||
|
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := noOperandTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := condCode(upper); ok {
|
if _, ok := condCode(upper); ok {
|
||||||
@@ -31,7 +38,7 @@ func Encodable(mnemonic string) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
base == "KMOVW" || base == "KMOVQ" {
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -53,13 +60,28 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Legacy SSE shuffles and packed binaries dispatch on the full name.
|
// Legacy SSE shuffles and packed binaries dispatch on the full name; so
|
||||||
|
// do the imm8-controlled instructions, the lane extracts and inserts and
|
||||||
|
// the packed integer shifts (their trailing width letters belong to the
|
||||||
|
// mnemonic).
|
||||||
if _, ok := sseShufTable[upper]; ok {
|
if _, ok := sseShufTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
if _, ok := sseBinTable[upper]; ok {
|
if _, ok := sseBinTable[upper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
if _, ok := sseImm3Table[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseExtractTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseInsertTable[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
// The size-suffix split: retry the tables and the scalar switch on the
|
// The size-suffix split: retry the tables and the scalar switch on the
|
||||||
// base.
|
// base.
|
||||||
@@ -74,12 +96,15 @@ func Encodable(mnemonic string) bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
switch base2 {
|
switch base2 {
|
||||||
case "MOV",
|
case "MOV", "MOVD",
|
||||||
"ADD", "SUB", "AND", "OR", "XOR", "CMP",
|
"ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB",
|
||||||
"TEST",
|
"TEST",
|
||||||
"LEA",
|
"LEA",
|
||||||
"INC", "DEC", "NEG", "NOT",
|
"INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV",
|
||||||
"SHL", "SHR", "SAR",
|
"SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR",
|
||||||
|
"BT", "BTS", "BTR", "BTC",
|
||||||
|
"XCHG", "CMPXCHG", "XADD", "CRC32", "ADCX", "ADOX",
|
||||||
|
"MOVS", "STOS",
|
||||||
"IMUL", "IMUL3",
|
"IMUL", "IMUL3",
|
||||||
"PUSH", "POP",
|
"PUSH", "POP",
|
||||||
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
"BSF", "BSR", "LZCNT", "TZCNT", "POPCNT",
|
||||||
@@ -88,7 +113,9 @@ func Encodable(mnemonic string) bool {
|
|||||||
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
"MOVBLZX", "MOVBQZX", "MOVWLZX", "MOVWQZX", "MOVWLSX", "MOVLQSX",
|
||||||
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
"MOVBWZX", "MOVBWSX", "MOVBLSX", "MOVBQSX", "MOVWQSX", "MOVLQZX",
|
||||||
"CVTSL2SD", "CVTSQ2SD",
|
"CVTSL2SD", "CVTSQ2SD",
|
||||||
"MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
"CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S",
|
||||||
|
"FMOVD",
|
||||||
|
"MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
// Full-name dispatches the size split would eat (a trailing width
|
// Full-name dispatches the size split would eat (a trailing width
|
||||||
|
|||||||
+387
-8
@@ -5,6 +5,8 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"math"
|
||||||
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -21,6 +23,39 @@ func Encode(mnemonic string, ops ...Operand) ([]byte, error) {
|
|||||||
type enc struct {
|
type enc struct {
|
||||||
out []byte
|
out []byte
|
||||||
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
patches []encPatch // disp32 fields awaiting static-symbol resolution
|
||||||
|
|
||||||
|
// FloatPool collects the pooled constants the floating-point
|
||||||
|
// immediates reference, in first-use order.
|
||||||
|
floatPool []floatPoolEntry
|
||||||
|
floatPoolSeen map[string]bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatPoolEntry is one pooled floating-point constant: the symbol name
|
||||||
|
// the emitted RIP-relative load refers to and its IEEE-754 bytes.
|
||||||
|
type floatPoolEntry struct {
|
||||||
|
name string
|
||||||
|
data []byte
|
||||||
|
}
|
||||||
|
|
||||||
|
// addFloatPool records a pooled constant, deduplicated by symbol name.
|
||||||
|
func (e *enc) addFloatPool(name string, bits uint64, width int) {
|
||||||
|
if e.floatPoolSeen == nil {
|
||||||
|
e.floatPoolSeen = map[string]bool{}
|
||||||
|
}
|
||||||
|
if e.floatPoolSeen[name] {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
e.floatPoolSeen[name] = true
|
||||||
|
data := make([]byte, width)
|
||||||
|
for i := range width {
|
||||||
|
data[i] = byte(bits >> (8 * i))
|
||||||
|
}
|
||||||
|
e.floatPool = append(e.floatPool, floatPoolEntry{name: name, data: data})
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatPoolList returns the pooled constants in first-use order.
|
||||||
|
func (e *enc) floatPoolList() []floatPoolEntry {
|
||||||
|
return e.floatPool
|
||||||
}
|
}
|
||||||
|
|
||||||
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
// encPatch marks a 4-byte displacement field in enc.out that must receive the
|
||||||
@@ -29,6 +64,7 @@ type encPatch struct {
|
|||||||
off int
|
off int
|
||||||
name string
|
name string
|
||||||
addend int64
|
addend int64
|
||||||
|
tls bool // a TLS slot offset: the patch is R_TLSLE with no symbol
|
||||||
}
|
}
|
||||||
|
|
||||||
func (e *enc) encode(mnem string, ops []Operand) error {
|
func (e *enc) encode(mnem string, ops []Operand) error {
|
||||||
@@ -58,6 +94,56 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if cc, ok := condCode(upper); ok {
|
if cc, ok := condCode(upper); ok {
|
||||||
return e.encodeJcc(cc, ops)
|
return e.encodeJcc(cc, ops)
|
||||||
}
|
}
|
||||||
|
// No-operand system and string-control instructions (CPUID, RDTSC,
|
||||||
|
// SYSCALL, the fences, UNDEF, …).
|
||||||
|
if op, ok := noOperandTable[upper]; ok {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
|
||||||
|
}
|
||||||
|
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
|
||||||
|
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
|
||||||
|
switch upper {
|
||||||
|
case "POPFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("POPFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9D}, modrm: -1, sib: -1})
|
||||||
|
case "PUSHFQ":
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("PUSHFQ takes no operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
|
||||||
|
case "INT":
|
||||||
|
return e.encodeInt(ops)
|
||||||
|
case "LDMXCSR":
|
||||||
|
return e.encodeMxcsr(2, ops)
|
||||||
|
case "STMXCSR":
|
||||||
|
return e.encodeMxcsr(3, ops)
|
||||||
|
// CMPSD is the scalar double compare, whose predicate immediate comes
|
||||||
|
// LAST in Plan 9 order (src, dst, $imm).
|
||||||
|
case "CMPSD":
|
||||||
|
return e.encodeCmpsd(ops)
|
||||||
|
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
|
||||||
|
case "SHA256RNDS2":
|
||||||
|
return e.encodeSha256rnds2(ops)
|
||||||
|
// BYTE, WORD, LONG and QUAD write the immediate into the text stream
|
||||||
|
// itself: 1, 2, 4 or 8 literal bytes, little-endian. END is accepted
|
||||||
|
// and ignored. ADJSP adjusts SP by the immediate, sign-chosen between
|
||||||
|
// the SUBQ and ADDQ forms.
|
||||||
|
case "BYTE", "WORD", "LONG", "QUAD":
|
||||||
|
return e.encodeData(upper, ops)
|
||||||
|
case "END":
|
||||||
|
return e.encodeEnd(ops)
|
||||||
|
case "ADJSP":
|
||||||
|
return e.encodeAdjsp(ops)
|
||||||
|
// The runtime's bookkeeping statements carry no text bytes: go tool asm
|
||||||
|
// records FUNCDATA and PCDATA in the program list only, so the encoded
|
||||||
|
// body shows nothing, on every architecture.
|
||||||
|
case "FUNCDATA", "PCDATA":
|
||||||
|
return e.encodeFuncdata(upper, ops)
|
||||||
|
}
|
||||||
|
|
||||||
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
// VEX (AVX/AVX2) and EVEX (AVX-512) instructions: the trailing
|
||||||
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
// B/W/L/Q/D is part of the mnemonic, not a size suffix, so dispatch
|
||||||
@@ -67,7 +153,9 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) || base == "KMOVW" || base == "KMOVQ" {
|
if isVex(base) || isEvex(base) || isKOp(base) || isGather(base) || isScatter(base) ||
|
||||||
|
isEvexPrefGather(base) ||
|
||||||
|
base == "KMOVW" || base == "KMOVQ" || base == "KMOVB" || base == "KMOVD" {
|
||||||
return e.encodeVec(base, ops, sfx)
|
return e.encodeVec(base, ops, sfx)
|
||||||
}
|
}
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
@@ -94,13 +182,35 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
// Legacy SSE packed binaries dispatch on the full name: the packed
|
// Legacy SSE packed binaries dispatch on the full name: the packed
|
||||||
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
// integer mnemonics carry real width suffixes (PADDB/PCMPGTW/...),
|
||||||
// which the size split must not eat.
|
// which the size split must not eat. A floating-point immediate
|
||||||
|
// rewrites into a pooled-constant read on the scalar members.
|
||||||
if m, ok := sseBinTable[upper]; ok {
|
if m, ok := sseBinTable[upper]; ok {
|
||||||
|
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||||
|
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||||
|
}
|
||||||
return e.encodeSSEBin(m, ops)
|
return e.encodeSSEBin(m, ops)
|
||||||
}
|
}
|
||||||
if m, ok := sseBinTable[base]; ok {
|
if m, ok := sseBinTable[base]; ok {
|
||||||
|
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||||
|
return e.encodeSSEFloatBin(upper, m, f, ops)
|
||||||
|
}
|
||||||
return e.encodeSSEBin(m, ops)
|
return e.encodeSSEBin(m, ops)
|
||||||
}
|
}
|
||||||
|
// The imm8-controlled legacy instructions, the lane extracts and inserts
|
||||||
|
// and the packed integer shifts all dispatch on the full name: a trailing
|
||||||
|
// width letter here belongs to the mnemonic, not to the size split.
|
||||||
|
if m, ok := sseImm3Table[upper]; ok {
|
||||||
|
return e.encodeSSEImm3(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseExtractTable[upper]; ok {
|
||||||
|
return e.encodeSSEExtract(m, ops)
|
||||||
|
}
|
||||||
|
if m, ok := sseInsertTable[upper]; ok {
|
||||||
|
return e.encodeSSEInsert(m, ops)
|
||||||
|
}
|
||||||
|
if _, ok := sseShiftImm[upper]; ok {
|
||||||
|
return e.encodeSSEShift(upper, ops)
|
||||||
|
}
|
||||||
// PMOVMSKB ends in a width letter the size split would eat, so it
|
// PMOVMSKB ends in a width letter the size split would eat, so it
|
||||||
// dispatches on the full name like the packed binaries above.
|
// dispatches on the full name like the packed binaries above.
|
||||||
if upper == "PMOVMSKB" {
|
if upper == "PMOVMSKB" {
|
||||||
@@ -109,16 +219,36 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
switch base {
|
switch base {
|
||||||
case "MOV":
|
case "MOV":
|
||||||
return e.encodeMov(ops, size)
|
return e.encodeMov(ops, size)
|
||||||
case "ADD", "SUB", "AND", "OR", "XOR", "CMP":
|
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
|
||||||
|
// REX.W and all.
|
||||||
|
case "MOVD":
|
||||||
|
return e.encodeMov(ops, 8)
|
||||||
|
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
|
||||||
return e.encodeALU(aluOp[base], ops, size)
|
return e.encodeALU(aluOp[base], ops, size)
|
||||||
case "TEST":
|
case "TEST":
|
||||||
return e.encodeTest(ops, size)
|
return e.encodeTest(ops, size)
|
||||||
case "LEA":
|
case "LEA":
|
||||||
return e.encodeLea(ops, size)
|
return e.encodeLea(ops, size)
|
||||||
case "INC", "DEC", "NEG", "NOT":
|
case "INC", "DEC", "NEG", "NOT", "MUL", "DIV", "IDIV":
|
||||||
return e.encodeUnary(unaryOp[base], ops, size)
|
return e.encodeUnary(unaryOp[base], ops, size)
|
||||||
case "SHL", "SHR", "SAR":
|
case "SHL", "SHR", "SAR", "SAL", "ROL", "ROR", "RCL", "RCR":
|
||||||
return e.encodeShift(shiftOp[base], ops, size)
|
return e.encodeShift(base, ops, size)
|
||||||
|
case "BT", "BTS", "BTR", "BTC":
|
||||||
|
return e.encodeBitTest(base, ops, size)
|
||||||
|
case "XCHG":
|
||||||
|
return e.encodeExchange(ops, size)
|
||||||
|
case "CMPXCHG":
|
||||||
|
return e.encodeRegRegOp(0xB0, 0xB1, base, ops, size)
|
||||||
|
case "XADD":
|
||||||
|
return e.encodeRegRegOp(0xC0, 0xC1, base, ops, size)
|
||||||
|
case "CRC32":
|
||||||
|
return e.encodeCrc32(ops, size)
|
||||||
|
case "ADCX":
|
||||||
|
return e.encodeCarryExt(0x66, ops, size)
|
||||||
|
case "ADOX":
|
||||||
|
return e.encodeCarryExt(0xF3, ops, size)
|
||||||
|
case "MOVS", "STOS":
|
||||||
|
return e.encodeStringOp(base, ops, size)
|
||||||
case "IMUL", "IMUL3":
|
case "IMUL", "IMUL3":
|
||||||
return e.encodeImul(ops, size)
|
return e.encodeImul(ops, size)
|
||||||
case "PUSH":
|
case "PUSH":
|
||||||
@@ -136,7 +266,16 @@ func (e *enc) encode(mnem string, ops []Operand) error {
|
|||||||
return e.encodeMovExtend(base, ops)
|
return e.encodeMovExtend(base, ops)
|
||||||
case "CVTSL2SD", "CVTSQ2SD":
|
case "CVTSL2SD", "CVTSQ2SD":
|
||||||
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
return e.encodeCvtsi2sd(base == "CVTSQ2SD", ops)
|
||||||
case "MOVOU", "MOVO", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD", "MOVSD", "MOVSS":
|
case "CVTSD2S", "CVTTSD2S", "CVTSS2S", "CVTTSS2S":
|
||||||
|
return e.encodeCvtInt(base, ops, size)
|
||||||
|
case "FMOVD":
|
||||||
|
return e.encodeFmov(ops)
|
||||||
|
case "MOVSD", "MOVSS":
|
||||||
|
if f, isFloat := floatImmOperand(ops); isFloat {
|
||||||
|
return e.encodeSSEFloatMove(upper, f, ops)
|
||||||
|
}
|
||||||
|
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||||
|
case "MOVOU", "MOVO", "MOVOA", "MOVUPS", "MOVAPS", "MOVUPD", "MOVAPD":
|
||||||
return e.encodeSSEMove(sseMoveTable[base], ops)
|
return e.encodeSSEMove(sseMoveTable[base], ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unsupported instruction %q", mnem)
|
return fmt.Errorf("unsupported instruction %q", mnem)
|
||||||
@@ -166,6 +305,214 @@ var prefetchVariant = map[string]int{
|
|||||||
"PREFETCHT2": 3,
|
"PREFETCHT2": 3,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dataWidth is the literal byte count of each data-emission pseudo-op.
|
||||||
|
var dataWidth = map[string]int{
|
||||||
|
"BYTE": 1,
|
||||||
|
"WORD": 2,
|
||||||
|
"LONG": 4,
|
||||||
|
"QUAD": 8,
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeData emits the literal-data pseudo-ops: BYTE, WORD, LONG and QUAD
|
||||||
|
// write the immediate into the text stream as 1, 2, 4 or 8 bytes,
|
||||||
|
// little-endian, with no opcode lookup. The value is truncated to the
|
||||||
|
// width rather than range-checked, exactly as go tool asm behaves (BYTE
|
||||||
|
// $0x1FF emits FF, WORD $0x12345 emits 45 23, both without an error), and
|
||||||
|
// exactly one immediate is accepted: the toolchain rejects a list such as
|
||||||
|
// BYTE $1, $2, $3.
|
||||||
|
func (e *enc) encodeData(mnem string, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 1 immediate operand, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s requires an integer immediate", mnem)
|
||||||
|
}
|
||||||
|
width := dataWidth[mnem]
|
||||||
|
out := make([]byte, width)
|
||||||
|
u := uint64(imm)
|
||||||
|
for i := range width {
|
||||||
|
out[i] = byte(u >> (8 * i))
|
||||||
|
}
|
||||||
|
e.out = append(e.out, out...)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeFuncdata accepts-and-ignores the runtime bookkeeping statements:
|
||||||
|
// FUNCDATA $n, sym(SB) and PCDATA $n, $m. go tool asm emits no text bytes
|
||||||
|
// for either (the entries live in the object's ancillary tables, not the
|
||||||
|
// function body), and the operand shapes it takes are exactly these: an
|
||||||
|
// integer count first, then a symbol reference for FUNCDATA and an integer
|
||||||
|
// value for PCDATA. The other architectures accept-and-ignore the same
|
||||||
|
// statements; amd64 now matches.
|
||||||
|
func (e *enc) encodeFuncdata(upper string, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
|
||||||
|
}
|
||||||
|
if _, ok := ops[0].(Imm); !ok {
|
||||||
|
return fmt.Errorf("%s: first operand must be an integer immediate", upper)
|
||||||
|
}
|
||||||
|
switch upper {
|
||||||
|
case "FUNCDATA":
|
||||||
|
if _, ok := ops[1].(sbMem); !ok {
|
||||||
|
return fmt.Errorf("FUNCDATA: second operand must be a symbol reference")
|
||||||
|
}
|
||||||
|
case "PCDATA":
|
||||||
|
if _, ok := ops[1].(Imm); !ok {
|
||||||
|
return fmt.Errorf("PCDATA: second operand must be an integer immediate")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeEnd accepts-and-ignores END. go tool asm drops the statement
|
||||||
|
// entirely: the AEND Prog is skipped when the program list is flushed, so
|
||||||
|
// the statements after an END still belong to the same function and the
|
||||||
|
// encoded body carries no trace of it, whatever operands follow the name
|
||||||
|
// (the toolchain takes END $0 and END AX alike). Zero bytes, no effect.
|
||||||
|
func (e *enc) encodeEnd(ops []Operand) error {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeAdjsp emits ADJSP $imm: a positive value is SUBQ $imm, SP, a
|
||||||
|
// negative one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude
|
||||||
|
// picks (the same selection subSP and addSP make for the frame). go tool
|
||||||
|
// asm refuses ADJSP $0 outright, so a zero value is an error here too; the
|
||||||
|
// statement's effect on the SP balance is checked by the function-level
|
||||||
|
// assembly (checkAdjspBalance), as the toolchain's push/pop walk does.
|
||||||
|
func (e *enc) encodeAdjsp(ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("ADJSP expects 1 immediate operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("ADJSP requires an integer immediate")
|
||||||
|
}
|
||||||
|
switch v := int(imm); {
|
||||||
|
case v > 0:
|
||||||
|
e.out = append(e.out, subSP(v)...)
|
||||||
|
case v < 0:
|
||||||
|
e.out = append(e.out, addSP(-v)...)
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("ADJSP $0 has no encoding")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- floating-point immediates ----------------------------------------------
|
||||||
|
|
||||||
|
// sseFloatImm lists the mnemonics whose first operand may be a floating-point
|
||||||
|
// immediate, the set go tool asm rewrites into a pooled-constant read: the
|
||||||
|
// scalar moves, the four scalar arithmetic pairs and the scalar compares.
|
||||||
|
// The packed members and the uniform forms (MAXSD, MINSD, SQRTSD, CMPSD)
|
||||||
|
// reject the immediate in the toolchain and are absent here on purpose.
|
||||||
|
var sseFloatImm = map[string]bool{
|
||||||
|
"MOVSD": true, "MOVSS": true,
|
||||||
|
"ADDSD": true, "ADDSS": true,
|
||||||
|
"SUBSD": true, "SUBSS": true,
|
||||||
|
"MULSD": true, "MULSS": true,
|
||||||
|
"DIVSD": true, "DIVSS": true,
|
||||||
|
"COMISD": true, "COMISS": true,
|
||||||
|
"UCOMISD": true, "UCOMISS": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatImmOperand reports whether the operand list opens with a
|
||||||
|
// floating-point immediate in the two-operand spelling (imm, dst).
|
||||||
|
func floatImmOperand(ops []Operand) (FloatImm, bool) {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return FloatImm{}, false
|
||||||
|
}
|
||||||
|
f, ok := ops[0].(FloatImm)
|
||||||
|
return f, ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// floatPoolValue evaluates a floating-point immediate at the width its
|
||||||
|
// mnemonic encodes and names the pool constant the toolchain synthesises:
|
||||||
|
// $f64.<16 hex> for the doubles, $f32.<8 hex> for the singles (the float32
|
||||||
|
// rounding of the parsed value). The name carries the IEEE-754 bits; the
|
||||||
|
// section holds them little-endian.
|
||||||
|
func floatPoolValue(mnem string, f FloatImm) (bits uint64, name string, err error) {
|
||||||
|
v, err := strconv.ParseFloat(f.Text, 64)
|
||||||
|
if err != nil {
|
||||||
|
return 0, "", fmt.Errorf("invalid floating-point immediate %q", f.Text)
|
||||||
|
}
|
||||||
|
if f.Neg {
|
||||||
|
v = -v
|
||||||
|
}
|
||||||
|
if strings.HasSuffix(mnem, "D") {
|
||||||
|
bits = math.Float64bits(v)
|
||||||
|
return bits, fmt.Sprintf("$f64.%016x", bits), nil
|
||||||
|
}
|
||||||
|
bits = uint64(math.Float32bits(float32(v)))
|
||||||
|
return bits, fmt.Sprintf("$f32.%08x", bits), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEFloatMove encodes MOVSD/MOVSS with a floating-point immediate
|
||||||
|
// source. A positive zero needs no memory read: the toolchain emits
|
||||||
|
// XORPS dst, dst. Anything else loads the pooled constant RIP-relative
|
||||||
|
// ($f64.<hex>(SB) / $f32.<hex>(SB)), the displacement a patch site the
|
||||||
|
// file-level layout or the linker resolves.
|
||||||
|
func (e *enc) encodeSSEFloatMove(mnem string, f FloatImm, ops []Operand) error {
|
||||||
|
if !sseFloatImm[mnem] {
|
||||||
|
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||||
|
}
|
||||||
|
dst, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
bits, name, err := floatPoolValue(mnem, f)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.addFloatPool(name, bits, mwidth(mnem))
|
||||||
|
if bits == 0 {
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0x57}, modrm: -1, sib: -1} // XORPS
|
||||||
|
if err := setRM(i, dst, dst, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
m := sseMoveTable[mnem]
|
||||||
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.load}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEFloatBin encodes the scalar arithmetic and compare mnemonics with
|
||||||
|
// a floating-point immediate source: the constant is read from the pool into
|
||||||
|
// the instruction's r/m side (reg = destination), the rewrite go tool asm
|
||||||
|
// performs at the source level.
|
||||||
|
func (e *enc) encodeSSEFloatBin(mnem string, m sseBin, f FloatImm, ops []Operand) error {
|
||||||
|
if !sseFloatImm[mnem] {
|
||||||
|
return fmt.Errorf("%s does not take a floating-point immediate", mnem)
|
||||||
|
}
|
||||||
|
dst, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dst.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
bits, name, err := floatPoolValue(mnem, f)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.addFloatPool(name, bits, mwidth(mnem))
|
||||||
|
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dst, sbMem{size: mwidth(mnem), name: name}, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// mwidth returns the operand width a scalar SSE mnemonic encodes: the double
|
||||||
|
// spellings end in D, the single spellings in S.
|
||||||
|
func mwidth(mnem string) int {
|
||||||
|
if strings.HasSuffix(mnem, "D") {
|
||||||
|
return 8
|
||||||
|
}
|
||||||
|
return 4
|
||||||
|
}
|
||||||
|
|
||||||
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
// splitSize separates a trailing B/W/L/Q size suffix from the mnemonic.
|
||||||
func splitSize(upper string) (base string, size int) {
|
func splitSize(upper string) (base string, size int) {
|
||||||
if upper == "" {
|
if upper == "" {
|
||||||
@@ -195,7 +542,7 @@ func (e *enc) encodeVec(upper string, ops []Operand, sfx evexSuffix) error {
|
|||||||
if ss, ok := scatterTable[upper]; ok {
|
if ss, ok := scatterTable[upper]; ok {
|
||||||
return e.encodeScatter(upper, ss, ops, sfx)
|
return e.encodeScatter(upper, ss, ops, sfx)
|
||||||
}
|
}
|
||||||
if upper == "KMOVW" || upper == "KMOVQ" {
|
if upper == "KMOVW" || upper == "KMOVQ" || upper == "KMOVB" || upper == "KMOVD" {
|
||||||
if sfx.any() {
|
if sfx.any() {
|
||||||
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
return fmt.Errorf("%s takes no EVEX suffixes", upper)
|
||||||
}
|
}
|
||||||
@@ -232,6 +579,7 @@ type instr struct {
|
|||||||
disp []byte
|
disp []byte
|
||||||
imm []byte
|
imm []byte
|
||||||
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
sb *sbRef // static-symbol displacement in disp, awaiting resolution
|
||||||
|
tls bool // the displacement is a TLS slot offset, patched R_TLSLE
|
||||||
}
|
}
|
||||||
|
|
||||||
// sbRef records that an instruction's displacement refers to a static symbol
|
// sbRef records that an instruction's displacement refers to a static symbol
|
||||||
@@ -274,6 +622,9 @@ func (e *enc) emit(i *instr) error {
|
|||||||
if i.sb != nil {
|
if i.sb != nil {
|
||||||
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
e.patches = append(e.patches, encPatch{off: len(e.out), name: i.sb.name, addend: i.sb.addend})
|
||||||
}
|
}
|
||||||
|
if i.tls {
|
||||||
|
e.patches = append(e.patches, encPatch{off: len(e.out), tls: true})
|
||||||
|
}
|
||||||
e.out = append(e.out, i.disp...)
|
e.out = append(e.out, i.disp...)
|
||||||
e.out = append(e.out, i.imm...)
|
e.out = append(e.out, i.imm...)
|
||||||
return nil
|
return nil
|
||||||
@@ -328,12 +679,30 @@ func setRMReg(i *instr, regField int, rexR, regForced bool, rm Operand, opSize i
|
|||||||
i.disp = le32(0)
|
i.disp = le32(0)
|
||||||
i.sb = &sbRef{name: r.name, addend: r.addend}
|
i.sb = &sbRef{name: r.name, addend: r.addend}
|
||||||
return nil
|
return nil
|
||||||
|
case TLSMem:
|
||||||
|
// off(TLS): the segment-prefixed absolute access, mod=00 with the
|
||||||
|
// SIB escape's disp32 absolute form. The displacement is the TLS
|
||||||
|
// slot offset, patched by the linker's TLS relocation.
|
||||||
|
i.prefix = r.Seg
|
||||||
|
i.modrm = 0x04 | regField<<3
|
||||||
|
i.sib = 0x25
|
||||||
|
i.disp = le32(r.Disp)
|
||||||
|
i.tls = true
|
||||||
|
return nil
|
||||||
|
case SegAbs:
|
||||||
|
// 0x30(GS): the segment override with the SIB escape's disp32
|
||||||
|
// absolute form, no relocation.
|
||||||
|
setSegAbs(i, regField, r)
|
||||||
|
return nil
|
||||||
default:
|
default:
|
||||||
return fmt.Errorf("invalid r/m operand %T", rm)
|
return fmt.Errorf("invalid r/m operand %T", rm)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func setMem(i *instr, regField int, m Mem) error {
|
func setMem(i *instr, regField int, m Mem) error {
|
||||||
|
if m.Seg != 0 {
|
||||||
|
i.prefix = m.Seg
|
||||||
|
}
|
||||||
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
|
modrm, sib, disp, xBit, bBit, err := memComponents(regField, m)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -346,6 +715,16 @@ func setMem(i *instr, regField int, m Mem) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// setSegAbs assembles a segment-absolute operand, 0x30(GS): the segment
|
||||||
|
// override with the mod=00 SIB escape's disp32 absolute form and no
|
||||||
|
// relocation.
|
||||||
|
func setSegAbs(i *instr, regField int, m SegAbs) {
|
||||||
|
i.prefix = m.Seg
|
||||||
|
i.modrm = 0x04 | regField<<3
|
||||||
|
i.sib = 0x25
|
||||||
|
i.disp = le32(m.Disp)
|
||||||
|
}
|
||||||
|
|
||||||
// memComponents computes the ModR/M byte (with the given reg field), the SIB
|
// memComponents computes the ModR/M byte (with the given reg field), the SIB
|
||||||
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
|
// byte (-1 if none), the displacement bytes, and the high index/base bits, for
|
||||||
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
// a memory operand. It is shared by the REX (scalar) and VEX (vector) paths.
|
||||||
|
|||||||
@@ -9,6 +9,9 @@ import (
|
|||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// decode encodes an instruction and decodes it back, returning the decoded
|
// decode encodes an instruction and decodes it back, returning the decoded
|
||||||
@@ -200,10 +203,60 @@ func TestUnary(t *testing.T) {
|
|||||||
func TestShift(t *testing.T) {
|
func TestShift(t *testing.T) {
|
||||||
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
|
checkSyntax(t, "shl rdx, 0x2", "SHLQ", Imm(2), DX)
|
||||||
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
|
checkSyntax(t, "shl rdx, cl", "SHLQ", CL, DX)
|
||||||
|
checkSyntax(t, "shl rdx, cl", "SHLQ", CX, DX)
|
||||||
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
|
checkSyntax(t, "shl rdx, 0x1", "SHLQ", Imm(1), DX)
|
||||||
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
|
checkSyntax(t, "sar rcx, 0x1f", "SARQ", Imm(31), CX)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestDoubleShift pins the three-operand SHL/SHR form, which encodes as
|
||||||
|
// SHLD/SHRD: go tool asm accepts it for SHL/SHR at W/L/Q widths and rejects
|
||||||
|
// it for SAR, SAL, the rotates and the B width. The byte pins mirror the
|
||||||
|
// oracle's objdump output (48 0f a4 fe 0d for the first case, and so on).
|
||||||
|
func TestDoubleShift(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string // hex encoding
|
||||||
|
}{
|
||||||
|
{"SHLQ imm", "SHLQ", []Operand{Imm(0x0d), DI, SI}, "480fa4fe0d"},
|
||||||
|
{"SHLQ CX high regs", "SHLQ", []Operand{CX, Reg{idx: 8, size: 8}, Reg{idx: 9, size: 8}}, "4d0fa5c1"},
|
||||||
|
{"SHRQ imm", "SHRQ", []Operand{Imm(1), AX, CX}, "480facc101"},
|
||||||
|
{"SHLW imm", "SHLW", []Operand{Imm(1), AX, CX}, "660fa4c101"},
|
||||||
|
{"SHRD CL", "SHRQ", []Operand{CL, AX, CX}, "480fadc1"},
|
||||||
|
{"SHLD imm high regs", "SHLQ", []Operand{Imm(2), Reg{idx: 10, size: 8}, Reg{idx: 11, size: 8}}, "4d0fa4d302"},
|
||||||
|
{"SHRD imm max", "SHRQ", []Operand{Imm(63), Reg{idx: 9, size: 8}, Reg{idx: 15, size: 8}}, "4d0faccf3f"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Rejected forms: the oracle rejects every one of these.
|
||||||
|
rejected := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"SARQ three operands", "SARQ", []Operand{Imm(1), AX, CX}},
|
||||||
|
{"SALQ three operands", "SALQ", []Operand{Imm(1), AX, CX}},
|
||||||
|
{"ROLQ three operands", "ROLQ", []Operand{Imm(1), AX, CX}},
|
||||||
|
{"SHLB three operands", "SHLB", []Operand{Imm(1), AL, CL}},
|
||||||
|
{"SHRQ memory source", "SHRQ", []Operand{Imm(1), Ptr(AX, 0, 8), CX}},
|
||||||
|
{"SHRQ ECX count", "SHRQ", []Operand{Reg{idx: 1, size: 4}, AX, CX}},
|
||||||
|
}
|
||||||
|
for _, c := range rejected {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: Encode succeeded, want rejection", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestImul(t *testing.T) {
|
func TestImul(t *testing.T) {
|
||||||
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
|
checkSyntax(t, "imul rdx, rcx", "IMULQ", CX, DX)
|
||||||
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
|
checkSyntax(t, "imul edx, edx, 0x3", "IMULL", Imm(3), DX, DX)
|
||||||
@@ -277,6 +330,12 @@ func TestSSEMoveGroundTruth(t *testing.T) {
|
|||||||
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
|
{"MOVSD (SI),X1", "MOVSD", []Operand{Ptr(SI, 0, 8), vreg(t, "X1")}, "f20f100e", "MOVSD_XMM"},
|
||||||
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
|
{"MOVSD X1,X2", "MOVSD", []Operand{vreg(t, "X1"), vreg(t, "X2")}, "f20f10d1", "MOVSD_XMM"},
|
||||||
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
|
{"MOVSS X3,(DI)", "MOVSS", []Operand{vreg(t, "X3"), Ptr(DI, 0, 4)}, "f30f111f", "MOVSS"},
|
||||||
|
// Static-symbol (SB) references: the GOROOT crypto kernels load and
|
||||||
|
// store octa constants by name (MOVOU bswapMask<>+0(SB), X0).
|
||||||
|
{"MOVOU sym,X0", "MOVOU", []Operand{sbMem{size: 16, name: "bswapMask"}, vreg(t, "X0")}, "f30f6f0500000000", "MOVDQU"},
|
||||||
|
{"MOVOU X0,sym+8", "MOVOU", []Operand{vreg(t, "X0"), sbMem{size: 16, name: "bswapMask", addend: 8}}, "f30f7f0500000000", "MOVDQU"},
|
||||||
|
{"MOVO sym,X1", "MOVO", []Operand{sbMem{size: 16, name: "gcmPoly"}, vreg(t, "X1")}, "660f6f0d00000000", "MOVDQA"},
|
||||||
|
{"MOVO X2,sym", "MOVO", []Operand{vreg(t, "X2"), sbMem{size: 16, name: "gcmPoly"}}, "660f7f1500000000", "MOVDQA"},
|
||||||
}
|
}
|
||||||
for _, c := range cases {
|
for _, c := range cases {
|
||||||
code, err := Encode(c.mnem, c.ops...)
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
@@ -546,6 +605,248 @@ func TestEncodableCmovSize(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestCarryShiftMulGroundTruth pins the carry-flag ALU family (ADC/SBB with
|
||||||
|
// their accumulator immediate forms), the rotate family, MUL/DIV/IDIV and the
|
||||||
|
// bit-test family byte for byte against go tool asm (see
|
||||||
|
// testdata/verify/scalar_amd64.s).
|
||||||
|
func TestCarryShiftMulGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ADCQ AX,BX", "ADCQ", []Operand{AX, BX}, "4811c3"},
|
||||||
|
{"ADCL AX,BX", "ADCL", []Operand{AX, BX}, "11c3"},
|
||||||
|
{"ADCB AL,BL", "ADCB", []Operand{AL, BL}, "10c3"},
|
||||||
|
{"ADCW AX,BX", "ADCW", []Operand{AX, BX}, "6611c3"},
|
||||||
|
{"SBBQ AX,BX", "SBBQ", []Operand{AX, BX}, "4819c3"},
|
||||||
|
{"ADCQ $5,BX", "ADCQ", []Operand{Imm(5), BX}, "4883d305"},
|
||||||
|
{"ADCQ $300,BX", "ADCQ", []Operand{Imm(300), BX}, "4881d32c010000"},
|
||||||
|
{"ADCQ $300,AX", "ADCQ", []Operand{Imm(300), AX}, "48152c010000"},
|
||||||
|
{"ADCB $5,AL", "ADCB", []Operand{Imm(5), AL}, "1405"},
|
||||||
|
{"SBBQ $300,AX", "SBBQ", []Operand{Imm(300), AX}, "481d2c010000"},
|
||||||
|
{"ADCQ AX,(BX)", "ADCQ", []Operand{AX, Ptr(BX, 0, 8)}, "481103"},
|
||||||
|
{"ROLQ $3,AX", "ROLQ", []Operand{Imm(3), AX}, "48c1c003"},
|
||||||
|
{"ROLL CX,BX", "ROLL", []Operand{CL, BX}, "d3c3"},
|
||||||
|
{"RORQ CL,AX", "RORQ", []Operand{CL, AX}, "48d3c8"},
|
||||||
|
{"RCRQ $1,BX", "RCRQ", []Operand{Imm(1), BX}, "48d1db"},
|
||||||
|
{"RCLQ $3,AX", "RCLQ", []Operand{Imm(3), AX}, "48c1d003"},
|
||||||
|
{"RORB CL,BL", "RORB", []Operand{CL, BL}, "d2cb"},
|
||||||
|
{"SALQ $2,AX", "SALQ", []Operand{Imm(2), AX}, "48c1e002"},
|
||||||
|
{"ROLW $1,AX", "ROLW", []Operand{Imm(1), AX}, "66d1c0"},
|
||||||
|
{"MULQ CX", "MULQ", []Operand{CX}, "48f7e1"},
|
||||||
|
{"MULL CX", "MULL", []Operand{CX}, "f7e1"},
|
||||||
|
{"MULB CL", "MULB", []Operand{CL}, "f6e1"},
|
||||||
|
{"DIVL CX", "DIVL", []Operand{CX}, "f7f1"},
|
||||||
|
{"IDIVQ CX", "IDIVQ", []Operand{CX}, "48f7f9"},
|
||||||
|
{"MULW CX", "MULW", []Operand{CX}, "66f7e1"},
|
||||||
|
{"BTQ AX,DX", "BTQ", []Operand{AX, DX}, "480fa3c2"},
|
||||||
|
{"BTL AX,DX", "BTL", []Operand{AX, DX}, "0fa3c2"},
|
||||||
|
{"BTW AX,DX", "BTW", []Operand{AX, DX}, "660fa3c2"},
|
||||||
|
{"BTQ $3,BX", "BTQ", []Operand{Imm(3), BX}, "480fbae303"},
|
||||||
|
{"BTQ $3,(AX)", "BTQ", []Operand{Imm(3), Ptr(AX, 0, 8)}, "480fba2003"},
|
||||||
|
{"BTSQ $5,BX", "BTSQ", []Operand{Imm(5), BX}, "480fbaeb05"},
|
||||||
|
{"BTCQ AX,BX", "BTCQ", []Operand{AX, BX}, "480fbbc3"},
|
||||||
|
{"BTRQ $7,BX", "BTRQ", []Operand{Imm(7), BX}, "480fbaf307"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The bit-test immediate is an unsigned bit index with the negative
|
||||||
|
// spelling accepted, the shuffle convention: BTQ $300 must be rejected.
|
||||||
|
if _, err := Encode("BTQ", Imm(300), AX); err == nil {
|
||||||
|
t.Errorf("BTQ $300: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAtomicSystemGroundTruth pins the exchange/compare-exchange/accumulate
|
||||||
|
// family, the string primitives, the flag and system instructions, the MXCSR
|
||||||
|
// pair, the scalar float-to-int conversions and the x87 FMOVD byte for byte
|
||||||
|
// against go tool asm (see testdata/verify/atomics_amd64.s and
|
||||||
|
// testdata/verify/system_amd64.s).
|
||||||
|
func TestAtomicSystemGroundTruth(t *testing.T) {
|
||||||
|
r8 := Reg{idx: 8, size: 8}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"XCHGQ AX,BX", "XCHGQ", []Operand{AX, BX}, "4893"},
|
||||||
|
{"XCHGQ BX,AX", "XCHGQ", []Operand{BX, AX}, "4893"},
|
||||||
|
{"XCHGL AX,BX", "XCHGL", []Operand{AX, BX}, "93"},
|
||||||
|
{"XCHGB AL,BL", "XCHGB", []Operand{AL, BL}, "86c3"},
|
||||||
|
{"XCHGW AX,BX", "XCHGW", []Operand{AX, BX}, "6693"},
|
||||||
|
{"XCHGQ R8,R9", "XCHGQ", []Operand{r8, Reg{idx: 9, size: 8}}, "4d87c1"},
|
||||||
|
{"XCHGQ BX,(AX)", "XCHGQ", []Operand{BX, Ptr(AX, 0, 8)}, "488718"},
|
||||||
|
{"XCHGQ (AX),BX", "XCHGQ", []Operand{Ptr(AX, 0, 8), BX}, "488718"},
|
||||||
|
{"XCHGQ AX,(BX)", "XCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "488703"},
|
||||||
|
{"CMPXCHGL AX,BX", "CMPXCHGL", []Operand{AX, BX}, "0fb1c3"},
|
||||||
|
{"CMPXCHGQ AX,(BX)", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
|
||||||
|
{"CMPXCHGB AL,(BX)", "CMPXCHGB", []Operand{AL, Ptr(BX, 0, 1)}, "0fb003"},
|
||||||
|
{"CMPXCHGW AX,BX", "CMPXCHGW", []Operand{AX, BX}, "660fb1c3"},
|
||||||
|
{"XADDL AX,BX", "XADDL", []Operand{AX, BX}, "0fc1c3"},
|
||||||
|
{"XADDQ AX,(BX)", "XADDQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fc103"},
|
||||||
|
{"XADDB AL,(BX)", "XADDB", []Operand{AL, Ptr(BX, 0, 1)}, "0fc003"},
|
||||||
|
{"XADDW AX,BX", "XADDW", []Operand{AX, BX}, "660fc1c3"},
|
||||||
|
{"ADCXL AX,CX", "ADCXL", []Operand{AX, CX}, "660f38f6c8"},
|
||||||
|
{"ADCXQ AX,CX", "ADCXQ", []Operand{AX, CX}, "66480f38f6c8"},
|
||||||
|
{"ADOXL AX,CX", "ADOXL", []Operand{AX, CX}, "f30f38f6c8"},
|
||||||
|
{"ADOXQ AX,CX", "ADOXQ", []Operand{AX, CX}, "f3480f38f6c8"},
|
||||||
|
{"CRC32B AX,CX", "CRC32B", []Operand{AX, CX}, "f20f38f0c8"},
|
||||||
|
{"CRC32W AX,CX", "CRC32W", []Operand{AX, CX}, "66f20f38f1c8"},
|
||||||
|
{"CRC32L AX,CX", "CRC32L", []Operand{AX, CX}, "f20f38f1c8"},
|
||||||
|
{"CRC32Q AX,CX", "CRC32Q", []Operand{AX, CX}, "f2480f38f1c8"},
|
||||||
|
{"CRC32L (AX),CX", "CRC32L", []Operand{Ptr(AX, 0, 4), CX}, "f20f38f108"},
|
||||||
|
{"MOVSQ", "MOVSQ", []Operand{}, "48a5"},
|
||||||
|
{"MOVSL", "MOVSL", []Operand{}, "a5"},
|
||||||
|
{"MOVSB", "MOVSB", []Operand{}, "a4"},
|
||||||
|
{"MOVSW", "MOVSW", []Operand{}, "66a5"},
|
||||||
|
{"STOSB", "STOSB", []Operand{}, "aa"},
|
||||||
|
{"STOSQ", "STOSQ", []Operand{}, "48ab"},
|
||||||
|
{"STOSL", "STOSL", []Operand{}, "ab"},
|
||||||
|
{"STOSW", "STOSW", []Operand{}, "66ab"},
|
||||||
|
{"CLD", "CLD", []Operand{}, "fc"},
|
||||||
|
{"STD", "STD", []Operand{}, "fd"},
|
||||||
|
{"POPFQ", "POPFQ", []Operand{}, "9d"},
|
||||||
|
{"PUSHFQ", "PUSHFQ", []Operand{}, "9c"},
|
||||||
|
{"CPUID", "CPUID", []Operand{}, "0fa2"},
|
||||||
|
{"RDTSC", "RDTSC", []Operand{}, "0f31"},
|
||||||
|
{"RDTSCP", "RDTSCP", []Operand{}, "0f01f9"},
|
||||||
|
{"SYSCALL", "SYSCALL", []Operand{}, "0f05"},
|
||||||
|
{"XGETBV", "XGETBV", []Operand{}, "0f01d0"},
|
||||||
|
{"PAUSE", "PAUSE", []Operand{}, "f390"},
|
||||||
|
{"LFENCE", "LFENCE", []Operand{}, "0faee8"},
|
||||||
|
{"MFENCE", "MFENCE", []Operand{}, "0faef0"},
|
||||||
|
{"SFENCE", "SFENCE", []Operand{}, "0faef8"},
|
||||||
|
{"UNDEF", "UNDEF", []Operand{}, "0f0b"},
|
||||||
|
{"INT $3", "INT", []Operand{Imm(3)}, "cd03"},
|
||||||
|
{"LDMXCSR (AX)", "LDMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae10"},
|
||||||
|
{"STMXCSR (AX)", "STMXCSR", []Operand{Ptr(AX, 0, 4)}, "0fae18"},
|
||||||
|
{"CVTSD2SL X0,AX", "CVTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2dc0"},
|
||||||
|
{"CVTTSD2SQ X0,AX", "CVTTSD2SQ", []Operand{vreg(t, "X0"), AX}, "f2480f2cc0"},
|
||||||
|
{"CVTTSD2SL X0,AX", "CVTTSD2SL", []Operand{vreg(t, "X0"), AX}, "f20f2cc0"},
|
||||||
|
{"CVTSS2SQ X0,AX", "CVTSS2SQ", []Operand{vreg(t, "X0"), AX}, "f3480f2dc0"},
|
||||||
|
{"FMOVD (AX),F0", "FMOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "F0")}, "dd00"},
|
||||||
|
{"FMOVD F0,(AX)", "FMOVD", []Operand{vreg(t, "F0"), Ptr(AX, 0, 8)}, "dd10"},
|
||||||
|
{"FMOVD F0,F1", "FMOVD", []Operand{vreg(t, "F0"), vreg(t, "F1")}, "ddd1"},
|
||||||
|
{"MOVD AX,X0", "MOVD", []Operand{AX, vreg(t, "X0")}, "66480f6ec0"},
|
||||||
|
{"MOVD X0,AX", "MOVD", []Operand{vreg(t, "X0"), AX}, "66480f7ec0"},
|
||||||
|
{"MOVD X0,X1", "MOVD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f30f7ec8"},
|
||||||
|
{"MOVD (AX),X0", "MOVD", []Operand{Ptr(AX, 0, 8), vreg(t, "X0")}, "f30f7e00"},
|
||||||
|
{"MOVD X0,(AX)", "MOVD", []Operand{vreg(t, "X0"), Ptr(AX, 0, 8)}, "660fd600"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// LDMXCSR/STMXCSR take a memory operand only.
|
||||||
|
if _, err := Encode("LDMXCSR", AX); err == nil {
|
||||||
|
t.Errorf("LDMXCSR AX: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSSEGapsGroundTruth pins the legacy SSE gap families: the scalar
|
||||||
|
// compare and square root, the Plan 9 packed spellings, the imm8-controlled
|
||||||
|
// shuffles, the lane extracts and inserts, the packed integer shifts and the
|
||||||
|
// AES/SHA round instructions, byte for byte against go tool asm (see
|
||||||
|
// testdata/verify/crypto_amd64.s and testdata/verify/sse_amd64.s).
|
||||||
|
func TestSSEGapsGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"ANDNPD X0,X1", "ANDNPD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f55c8"},
|
||||||
|
{"ANDNPS X0,X1", "ANDNPS", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f55c8"},
|
||||||
|
{"COMISD X0,X1", "COMISD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f2fc8"},
|
||||||
|
{"SQRTSD X0,X1", "SQRTSD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "f20f51c8"},
|
||||||
|
{"PSHUFL $3,X0,X1", "PSHUFL", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f70c803"},
|
||||||
|
{"PALIGNR $2,X0,X1", "PALIGNR", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0fc802"},
|
||||||
|
{"PBLENDW $3,X0,X1", "PBLENDW", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1")}, "660f3a0ec803"},
|
||||||
|
{"PCMPESTRI $1,X0,X1", "PCMPESTRI", []Operand{Imm(1), vreg(t, "X0"), vreg(t, "X1")}, "660f3a61c801"},
|
||||||
|
{"PCLMULQDQ $0,X0,X1", "PCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3a44c800"},
|
||||||
|
{"PCLMULQDQ $0,(AX),X1", "PCLMULQDQ", []Operand{Imm(0), Ptr(AX, 0, 16), vreg(t, "X1")}, "660f3a440800"},
|
||||||
|
{"PEXTRB $1,X0,AX", "PEXTRB", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a14c001"},
|
||||||
|
{"PEXTRD $1,X0,AX", "PEXTRD", []Operand{Imm(1), vreg(t, "X0"), AX}, "660f3a16c001"},
|
||||||
|
{"PEXTRQ $1,X0,AX", "PEXTRQ", []Operand{Imm(1), vreg(t, "X0"), AX}, "66480f3a16c001"},
|
||||||
|
{"PEXTRW $1,X0,AX", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), AX}, "660fc5c001"},
|
||||||
|
{"PEXTRW $1,X0,(AX)", "PEXTRW", []Operand{Imm(1), vreg(t, "X0"), Ptr(AX, 0, 2)}, "660f3a150001"},
|
||||||
|
{"PINSRB $1,AX,X0", "PINSRB", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a20c001"},
|
||||||
|
{"PINSRD $1,AX,X0", "PINSRD", []Operand{Imm(1), AX, vreg(t, "X0")}, "660f3a22c001"},
|
||||||
|
{"PINSRQ $1,AX,X0", "PINSRQ", []Operand{Imm(1), AX, vreg(t, "X0")}, "66480f3a22c001"},
|
||||||
|
{"PINSRW $1,AX,X0", "PINSRW", []Operand{Imm(1), AX, vreg(t, "X0")}, "660fc4c001"},
|
||||||
|
{"PINSRW $1,(AX),X0", "PINSRW", []Operand{Imm(1), Ptr(AX, 0, 2), vreg(t, "X0")}, "660fc40001"},
|
||||||
|
{"PSLLL $2,X0", "PSLLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72f002"},
|
||||||
|
{"PSRAL $2,X0", "PSRAL", []Operand{Imm(2), vreg(t, "X0")}, "660f72e002"},
|
||||||
|
{"PSRLL $2,X0", "PSRLL", []Operand{Imm(2), vreg(t, "X0")}, "660f72d002"},
|
||||||
|
{"PSRLQ $2,X0", "PSRLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d002"},
|
||||||
|
{"PSLLQ $2,X0", "PSLLQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f002"},
|
||||||
|
{"PSLLW $2,X0", "PSLLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71f002"},
|
||||||
|
{"PSRLW $2,X0", "PSRLW", []Operand{Imm(2), vreg(t, "X0")}, "660f71d002"},
|
||||||
|
{"PSRAW $2,X0", "PSRAW", []Operand{Imm(2), vreg(t, "X0")}, "660f71e002"},
|
||||||
|
{"PSLLDQ $2,X0", "PSLLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73f802"},
|
||||||
|
{"PSRLDQ $2,X0", "PSRLDQ", []Operand{Imm(2), vreg(t, "X0")}, "660f73d802"},
|
||||||
|
{"PSLLL X0,X1", "PSLLL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ff2c8"},
|
||||||
|
{"PSRLQ X0,X1", "PSRLQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660fd3c8"},
|
||||||
|
{"PSLLL (AX),X1", "PSLLL", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660ff208"},
|
||||||
|
{"PSUBL X0,X1", "PSUBL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffac8"},
|
||||||
|
{"PADDL X0,X1", "PADDL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660ffec8"},
|
||||||
|
{"PCMPEQL X0,X1", "PCMPEQL", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f76c8"},
|
||||||
|
{"PUNPCKLBW X0,X1", "PUNPCKLBW", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f60c8"},
|
||||||
|
{"MOVOA X0,X1", "MOVOA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f6fc8"},
|
||||||
|
{"MOVOA (AX),X1", "MOVOA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f6f08"},
|
||||||
|
{"MOVOA X0,(AX)", "MOVOA", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "660f7f00"},
|
||||||
|
{"AESIMC X0,X1", "AESIMC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dbc8"},
|
||||||
|
{"AESIMC (AX),X1", "AESIMC", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "660f38db08"},
|
||||||
|
{"AESENC X0,X1", "AESENC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dcc8"},
|
||||||
|
{"AESENCLAST X0,X1", "AESENCLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38ddc8"},
|
||||||
|
{"AESDEC X0,X1", "AESDEC", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dec8"},
|
||||||
|
{"AESDECLAST X0,X1", "AESDECLAST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "660f38dfc8"},
|
||||||
|
{"AESKEYGENASSIST $0,X0,X1", "AESKEYGENASSIST", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "660f3adfc800"},
|
||||||
|
{"SHA1MSG1 X0,X1", "SHA1MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c9c8"},
|
||||||
|
{"SHA1MSG2 X0,X1", "SHA1MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cac8"},
|
||||||
|
{"SHA1NEXTE X0,X1", "SHA1NEXTE", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38c8c8"},
|
||||||
|
{"SHA1RNDS4 $0,X0,X1", "SHA1RNDS4", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1")}, "0f3accc800"},
|
||||||
|
{"SHA256MSG1 X0,X1", "SHA256MSG1", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38ccc8"},
|
||||||
|
{"SHA256MSG2 X0,X1", "SHA256MSG2", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "0f38cdc8"},
|
||||||
|
{"SHA256RNDS2 X0,X1,X2", "SHA256RNDS2", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "0f38cbd1"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// SHA256RNDS2's first operand must be the literal X0.
|
||||||
|
if _, err := Encode("SHA256RNDS2", vreg(t, "X1"), vreg(t, "X2"), vreg(t, "X3")); err == nil {
|
||||||
|
t.Errorf("SHA256RNDS2 X1,...: expected an error, got none")
|
||||||
|
}
|
||||||
|
// PSLLDQ has no variable-count form.
|
||||||
|
if _, err := Encode("PSLLDQ", vreg(t, "X0"), vreg(t, "X1")); err == nil {
|
||||||
|
t.Errorf("PSLLDQ X0,X1: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
// TestSSEBinGroundTruth checks the legacy packed/scalar binary family
|
||||||
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
// byte for byte (no prefix / 66 / F2 / F3 variants).
|
||||||
func TestSSEBinGroundTruth(t *testing.T) {
|
func TestSSEBinGroundTruth(t *testing.T) {
|
||||||
@@ -627,3 +928,296 @@ func TestMOVQXMMGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestPrefixStatements pins LOCK, REP and REPN. go tool asm encodes each as
|
||||||
|
// a standalone one-byte instruction with a PC of its own (F0, F3, F2), not a
|
||||||
|
// prefix field merged into the following instruction, and it validates
|
||||||
|
// nothing about the pairing (LOCK before NOP assembles). The prefixed
|
||||||
|
// atomic and string shapes are the bytes the runtime's own kernels need.
|
||||||
|
func TestPrefixStatements(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"LOCK", "LOCK", nil, "f0"},
|
||||||
|
{"REP", "REP", nil, "f3"},
|
||||||
|
{"REPN", "REPN", nil, "f2"},
|
||||||
|
// LOCK; CMPXCHGQ AX, (BX)
|
||||||
|
{"LOCK CMPXCHGQ", "CMPXCHGQ", []Operand{AX, Ptr(BX, 0, 8)}, "480fb103"},
|
||||||
|
// REP; MOVSQ
|
||||||
|
{"REP MOVSQ", "MOVSQ", nil, "48a5"},
|
||||||
|
// REPN; MOVSB
|
||||||
|
{"REPN MOVSB", "MOVSB", nil, "a4"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The prefix statements take no operands, as the toolchain reports for
|
||||||
|
// LOCK AX.
|
||||||
|
if _, err := Encode("LOCK", AX); err == nil {
|
||||||
|
t.Error("LOCK AX assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("REP", Imm(1)); err == nil {
|
||||||
|
t.Error("REP $1 assembled, want an error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDataEmission pins BYTE, WORD, LONG and QUAD: the immediate lands in
|
||||||
|
// the text stream as 1, 2, 4 or 8 little-endian bytes with no opcode
|
||||||
|
// lookup, truncated to the width rather than range-checked (go tool asm
|
||||||
|
// emits FF for BYTE $0x1FF and 45 23 for WORD $0x12345, both silently).
|
||||||
|
func TestDataEmission(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
imm Imm
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"BYTE", "BYTE", 0x0f, "0f"},
|
||||||
|
{"BYTE negative", "BYTE", -1, "ff"},
|
||||||
|
{"BYTE truncated", "BYTE", 0x1ff, "ff"},
|
||||||
|
{"WORD", "WORD", 0x1234, "3412"},
|
||||||
|
{"WORD negative", "WORD", -1, "ffff"},
|
||||||
|
{"WORD truncated", "WORD", 0x12345, "4523"},
|
||||||
|
{"LONG", "LONG", 0x11223344, "44332211"},
|
||||||
|
{"LONG negative", "LONG", -1, "ffffffff"},
|
||||||
|
{"QUAD", "QUAD", 0x1122334455667788, "8877665544332211"},
|
||||||
|
{"QUAD negative", "QUAD", -2, "feffffffffffffff"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.imm)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("%s = %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Exactly one immediate: the toolchain rejects BYTE $1, $2, $3, and a
|
||||||
|
// register or a missing operand is no immediate at all.
|
||||||
|
if _, err := Encode("BYTE"); err == nil {
|
||||||
|
t.Error("BYTE with no operand assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("BYTE", Imm(1), Imm(2)); err == nil {
|
||||||
|
t.Error("BYTE $1, $2 assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("WORD", AX); err == nil {
|
||||||
|
t.Error("WORD AX assembled, want an error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEndIgnored pins END: go tool asm drops the statement entirely, so it
|
||||||
|
// encodes to zero bytes and takes any operands without complaint (the
|
||||||
|
// toolchain accepts END $0 and END AX alike).
|
||||||
|
func TestEndIgnored(t *testing.T) {
|
||||||
|
for _, ops := range [][]Operand{nil, {Imm(0)}, {AX}} {
|
||||||
|
code, err := Encode("END", ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("END: %v", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(code) != 0 {
|
||||||
|
t.Errorf("END = %x, want no bytes", code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAdjsp pins ADJSP: a positive immediate is SUBQ $imm, SP, a negative
|
||||||
|
// one ADDQ $-imm, SP, in the imm8 or imm32 form the magnitude picks; $0
|
||||||
|
// has no encoding (go tool asm refuses ADJSP $0 outright).
|
||||||
|
func TestAdjsp(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
imm Imm
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"imm8", 112, "4883ec70"},
|
||||||
|
{"imm8 negative", -112, "4883c470"},
|
||||||
|
{"imm32", 200, "4881ecc8000000"},
|
||||||
|
{"imm32 negative", -200, "4881c4c8000000"},
|
||||||
|
{"small", 8, "4883ec08"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode("ADJSP", c.imm)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := fmt.Sprintf("%x", code); got != c.want {
|
||||||
|
t.Errorf("ADJSP %d = %s, want %s", int64(c.imm), got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if _, err := Encode("ADJSP", Imm(0)); err == nil {
|
||||||
|
t.Error("ADJSP $0 assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("ADJSP"); err == nil {
|
||||||
|
t.Error("ADJSP with no operand assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := Encode("ADJSP", AX); err == nil {
|
||||||
|
t.Error("ADJSP AX assembled, want an error")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestFloatImmediateGroundTruth pins the floating-point immediate rewrite
|
||||||
|
// byte for byte against go tool asm: the scalar moves and the scalar
|
||||||
|
// arithmetic read the constant from a synthesised read-only pool symbol
|
||||||
|
// ($f64.<hex>, $f32.<hex>) RIP-relative with the displacement left to the
|
||||||
|
// relocation, and a positive zero on the moves collapses to XORPS dst, dst.
|
||||||
|
func TestFloatImmediateGroundTruth(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"MOVSD -1.0", "MOVSD", []Operand{FloatImm{Text: "1.0", Neg: true}, vreg(t, "X2")}, "f20f101500000000"},
|
||||||
|
{"MOVSD 1.5", "MOVSD", []Operand{FloatImm{Text: "1.5"}, vreg(t, "X3")}, "f20f101d00000000"},
|
||||||
|
{"MOVSS 2.5", "MOVSS", []Operand{FloatImm{Text: "2.5"}, vreg(t, "X4")}, "f30f102500000000"},
|
||||||
|
{"MOVSS -0.5", "MOVSS", []Operand{FloatImm{Text: "0.5", Neg: true}, vreg(t, "X5")}, "f30f102d00000000"},
|
||||||
|
{"MOVSS +0.0 is XORPS", "MOVSS", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X10")}, "450f57d2"},
|
||||||
|
{"MOVSD +0.0 is XORPS", "MOVSD", []Operand{FloatImm{Text: "0.0"}, vreg(t, "X6")}, "0f57f6"},
|
||||||
|
{"ADDSD 1.0", "ADDSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f580500000000"},
|
||||||
|
{"ADDSS 0.5", "ADDSS", []Operand{FloatImm{Text: "0.5"}, vreg(t, "X1")}, "f30f580d00000000"},
|
||||||
|
{"SUBSD 2.0", "SUBSD", []Operand{FloatImm{Text: "2.0"}, vreg(t, "X3")}, "f20f5c1d00000000"},
|
||||||
|
{"MULSD -2.5", "MULSD", []Operand{FloatImm{Text: "2.5", Neg: true}, vreg(t, "X3")}, "f20f591d00000000"},
|
||||||
|
{"DIVSD 1.0", "DIVSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "f20f5e0500000000"},
|
||||||
|
{"COMISD 1.0", "COMISD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}, "660f2f0500000000"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The pool names carry the IEEE-754 bits, the float32 narrowing for the
|
||||||
|
// single spellings; negative zero keeps its sign bit and never takes the
|
||||||
|
// XORPS shortcut.
|
||||||
|
for _, c := range []struct {
|
||||||
|
mnem string
|
||||||
|
imm FloatImm
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"MOVSD", FloatImm{Text: "1.0", Neg: true}, "$f64.bff0000000000000"},
|
||||||
|
{"MOVSD", FloatImm{Text: "0.5"}, "$f64.3fe0000000000000"},
|
||||||
|
{"MOVSS", FloatImm{Text: "2.5"}, "$f32.40200000"},
|
||||||
|
{"MOVSS", FloatImm{Text: "0.5", Neg: true}, "$f32.bf000000"},
|
||||||
|
{"MOVSD", FloatImm{Text: "0.0", Neg: true}, "$f64.8000000000000000"},
|
||||||
|
} {
|
||||||
|
_, name, err := floatPoolValue(c.mnem, c.imm)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s %s: %v", c.mnem, c.imm.Text, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if name != c.want {
|
||||||
|
t.Errorf("%s $%s: pool name %s, want %s", c.mnem, c.imm.Text, name, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The shapes the toolchain's parser rejects: the packed and uniform
|
||||||
|
// forms, a non-vector destination, and the integer spellings.
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"MAXSD rejects the immediate", "MAXSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||||
|
{"MINSD rejects the immediate", "MINSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||||
|
{"SQRTSD rejects the immediate", "SQRTSD", []Operand{FloatImm{Text: "1.0"}, vreg(t, "X0")}},
|
||||||
|
{"integer destination", "MOVSD", []Operand{FloatImm{Text: "1.0"}, AX}},
|
||||||
|
} {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBookkeepingGroundTruth pins FUNCDATA and PCDATA as accept-and-ignore:
|
||||||
|
// go tool asm emits no text bytes for either, on every architecture.
|
||||||
|
func TestBookkeepingGroundTruth(t *testing.T) {
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"FUNCDATA", "FUNCDATA", []Operand{Imm(3), sbMem{name: "\u00b7f.arginfo0"}}},
|
||||||
|
{"PCDATA", "PCDATA", []Operand{Imm(1), Imm(-1)}},
|
||||||
|
} {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(code) != 0 {
|
||||||
|
t.Errorf("%s: emitted %x, want no bytes", c.name, code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"FUNCDATA arity", "FUNCDATA", []Operand{Imm(3)}},
|
||||||
|
{"FUNCDATA missing the count", "FUNCDATA", []Operand{sbMem{name: "x"}}},
|
||||||
|
{"FUNCDATA integer value", "FUNCDATA", []Operand{Imm(3), Imm(4)}},
|
||||||
|
{"PCDATA arity", "PCDATA", []Operand{Imm(1)}},
|
||||||
|
{"PCDATA register value", "PCDATA", []Operand{Imm(1), AX}},
|
||||||
|
} {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// At the statement level the bookkeeping lines sit between real
|
||||||
|
// instructions and contribute nothing to the body, symbol reference
|
||||||
|
// included: the FUNCDATA operand never needs file-level resolution.
|
||||||
|
f, errs := parser.Parse("t_amd64.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n\tNOP\n\tFUNCDATA $3, \u00b7f.arginfo0(SB)\n\tPCDATA $1, $-1\n\tFUNCDATA $0, x<>(SB)\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
want := "90c3"
|
||||||
|
if got := hexCompact(img.Code); got != want {
|
||||||
|
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tFUNCDATA $1, X0\n\tRET\n")); err == nil {
|
||||||
|
t.Error("FUNCDATA $1, X0 assembled, want an error")
|
||||||
|
}
|
||||||
|
if _, err := AssembleFile(mustParse(t, "TEXT \u00b7f(SB), NOSPLIT, $0\n\tPCDATA $1, X0\n\tRET\n")); err == nil {
|
||||||
|
t.Error("PCDATA $1, X0 assembled, want an error")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Encodable mirrors Encode for the names this work touched.
|
||||||
|
for _, mnem := range []string{"FUNCDATA", "PCDATA", "V4FMADDPS", "V4FMADDSS", "V4FNMADDPS", "V4FNMADDSS", "VP4DPWSSD", "VP4DPWSSDS"} {
|
||||||
|
if !Encodable(mnem) {
|
||||||
|
t.Errorf("Encodable(%s) = false, want true", mnem)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// mustParse parses src or fails the test.
|
||||||
|
func mustParse(t *testing.T, src string) *ast.File {
|
||||||
|
t.Helper()
|
||||||
|
f, errs := parser.Parse("t_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
return f
|
||||||
|
}
|
||||||
|
|||||||
+628
-42
@@ -5,6 +5,7 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"slices"
|
||||||
"strings"
|
"strings"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -91,7 +92,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
"VPSRAD": {1, 0x72, 0, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
|
// EVEX.128/256/512.66.0F.W1, variable shift with an XMM count (VPSRAQ;
|
||||||
// the W bit distinguishes it from VPSRAD's E2 form).
|
// the W bit distinguishes it from VPSRAD's E2 form).
|
||||||
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPSRAQ": {1, 0x72, 1, 1, 4, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
|
// EVEX.128/256/512.F3.0F.W1, signed qword to packed double (reg=dst,
|
||||||
// rm=src, no vvvv).
|
// rm=src, no vvvv).
|
||||||
@@ -138,7 +139,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
|
|
||||||
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
|
// EVEX.66.0F, the EVEX forms of the VEX two-source shuffle.
|
||||||
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
"VSHUFPD": {1, 0xC6, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
"VSHUFPS": {1, 0xC6, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
"VSHUFPS": {1, 0xC6, 0, 0, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
|
// EVEX.66.0F3A, lane insert ($imm, xsrc, zsrc1, zdst).
|
||||||
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
"VINSERTF32X4": {3, 0x18, 0, 1, -1, vexNDS3Imm, [3]int{0, 16, 32}},
|
||||||
@@ -180,10 +181,19 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
"VPCMPUQ": {3, 0x1E, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
// EVEX.66.0F38, permutes (NDS form).
|
// EVEX.66.0F38, permutes (NDS form).
|
||||||
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMB": {2, 0x8D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMW": {2, 0x8D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMI2B": {2, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMI2D": {2, 0x76, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2Q": {2, 0x76, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F38, population count (reg=dst, rm=src; W selects byte/word
|
||||||
|
// against dword/qword).
|
||||||
|
"VPOPCNTB": {2, 0x54, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTD": {2, 0x55, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTQ": {2, 0x55, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F.W1, the qword spelling of the packed OR (VPORQ has no VEX
|
||||||
|
// form in the Go assembler: it always encodes through EVEX).
|
||||||
|
"VPORQ": {1, 0xEB, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2D": {2, 0x7E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2Q": {2, 0x7E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPERMT2PD": {2, 0x7F, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
@@ -193,7 +203,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPMULHUW": {1, 0xE4, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPMADDUBSW": {2, 0x04, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPSLLVW": {2, 0x12, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPSRLVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPSRLVW": {2, 0x10, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPACKSSWB": {1, 0x63, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPACKUSWB": {1, 0x67, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
"VPACKSSDW": {1, 0x6B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
@@ -315,7 +325,7 @@ var evexTable = map[string]evexSpec{
|
|||||||
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
"VCVTPD2UQQ": {1, 0x79, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTPS2QQ": {1, 0x7B, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTUDQ2PD": {1, 0x7A, 0, 2, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
"VCVTUDQ2PS": {1, 0x7A, 0, 0, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTUDQ2PS": {1, 0x7A, 0, 3, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
// EVEX.66.0F38, half-precision convert (half-width source).
|
// EVEX.66.0F38, half-precision convert (half-width source).
|
||||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM, [3]int{8, 16, 32}},
|
||||||
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
|
// EVEX.66.0F3A, half-precision convert back ($imm, src, dst: reg=src,
|
||||||
@@ -485,6 +495,331 @@ var evexTable = map[string]evexSpec{
|
|||||||
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
// destination (VPMOVDW dword→word, VPMOVQD qword→dword).
|
||||||
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
"VPMOVDW": {2, 0x33, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
"VPMOVQD": {2, 0x35, 0, 2, -1, vexRMRev, [3]int{8, 16, 32}},
|
||||||
|
|
||||||
|
// --- the AVX-512 families the avx512enc corpus exercises, read off
|
||||||
|
// the toolchain opcodetables ---
|
||||||
|
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VALIGNQ": {3, 0x03, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VANDNPD": {1, 0x55, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VANDPD": {1, 0x54, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VBLENDMPD": {2, 0x65, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VBLENDMPS": {2, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VBROADCASTF32X2": {2, 0x19, 0, 1, -1, vexRM, [3]int{0, 8, 8}},
|
||||||
|
"VBROADCASTF32X4": {2, 0x1A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTF32X8": {2, 0x1B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VBROADCASTF64X2": {2, 0x1A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTF64X4": {2, 0x1B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VBROADCASTI32X2": {2, 0x59, 0, 1, -1, vexRM, [3]int{8, 8, 8}},
|
||||||
|
"VBROADCASTI32X4": {2, 0x5A, 0, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTI32X8": {2, 0x5B, 0, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VBROADCASTI64X2": {2, 0x5A, 1, 1, -1, vexRM, [3]int{0, 16, 16}},
|
||||||
|
"VBROADCASTI64X4": {2, 0x5B, 1, 1, -1, vexRM, [3]int{0, 0, 32}},
|
||||||
|
"VCOMISD": {1, 0x2F, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
|
||||||
|
"VCVTSD2SS": {1, 0x5A, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VDBPSADBW": {3, 0x42, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VEXP2PD": {2, 0xC8, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VEXP2PS": {2, 0xC8, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||||
|
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VMOVNTPD": {1, 0x2B, 1, 1, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||||
|
"VORPD": {1, 0x56, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMB": {2, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMD": {2, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMQ": {2, 0x64, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBLENDMW": {2, 0x66, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPBROADCASTMB2Q": {2, 0x2A, 1, 2, -1, vexRM, [3]int{0, 0, 0}},
|
||||||
|
"VPBROADCASTMW2D": {2, 0x3A, 0, 2, -1, vexRM, [3]int{0, 0, 0}},
|
||||||
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPEQQ": {2, 0x29, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTQ": {2, 0x37, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPCOMPRESSB": {2, 0x63, 0, 1, -1, vexRMRev, [3]int{1, 1, 1}},
|
||||||
|
"VPCOMPRESSW": {2, 0x63, 1, 1, -1, vexRMRev, [3]int{2, 2, 2}},
|
||||||
|
"VPCONFLICTD": {2, 0xC4, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPCONFLICTQ": {2, 0xC4, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPDPBUSD": {2, 0x50, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPDPBUSDS": {2, 0x51, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPDPWSSD": {2, 0x52, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPDPWSSDS": {2, 0x53, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2PD": {2, 0x77, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2PS": {2, 0x77, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMI2W": {2, 0x75, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3, [3]int{0, 32, 64}},
|
||||||
|
"VPERMT2B": {2, 0x7D, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMT2PS": {2, 0x7F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMT2W": {2, 0x7D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPEXPANDB": {2, 0x62, 0, 1, -1, vexRM, [3]int{1, 1, 1}},
|
||||||
|
"VPEXPANDW": {2, 0x62, 1, 1, -1, vexRM, [3]int{2, 2, 2}},
|
||||||
|
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm, [3]int{4, 0, 0}},
|
||||||
|
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm, [3]int{8, 0, 0}},
|
||||||
|
"VPLZCNTD": {2, 0x44, 0, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPLZCNTQ": {2, 0x44, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPMADD52HUQ": {2, 0xB5, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMADD52LUQ": {2, 0xB4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULDQ": {2, 0x28, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULTISHIFTQB": {2, 0x83, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPMULUDQ": {1, 0xF4, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPOPCNTW": {2, 0x54, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VPORD": {1, 0xEB, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPROLVD": {2, 0x15, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPROLVQ": {2, 0x15, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPRORVD": {2, 0x14, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPRORVQ": {2, 0x14, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDD": {3, 0x71, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDQ": {3, 0x71, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDVD": {2, 0x71, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDVQ": {2, 0x71, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDVW": {2, 0x70, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHLDW": {3, 0x70, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDD": {3, 0x73, 0, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDQ": {3, 0x73, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDVD": {2, 0x73, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDVQ": {2, 0x73, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDVW": {2, 0x72, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSHRDW": {3, 0x72, 1, 1, -1, vexNDS3Imm, [3]int{16, 32, 64}},
|
||||||
|
"VPSHUFBITQMB": {2, 0x8F, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSRAVW": {2, 0x11, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSRLD": {1, 0x72, 0, 1, 2, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F73 /7, the byte-quad shift left (the count is always an
|
||||||
|
// immediate; there is no register-count twin).
|
||||||
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128/256/512.0F.W0, the plain-prefix (no 66) packed spellings
|
||||||
|
// whose EVEX form drops the legacy prefix entirely.
|
||||||
|
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VCOMISS": {1, 0x2F, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
|
||||||
|
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM, [3]int{4, 0, 0}},
|
||||||
|
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMB": {2, 0x26, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMD": {2, 0x27, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMQ": {2, 0x27, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTMW": {2, 0x26, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMB": {2, 0x26, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMD": {2, 0x27, 0, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMQ": {2, 0x27, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPTESTNMW": {2, 0x26, 1, 2, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKHQDQ": {1, 0x6D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKLQDQ": {1, 0x6C, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VRCP28PD": {2, 0xCA, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRCP28PS": {2, 0xCA, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRCP28SD": {2, 0xCB, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VRCP28SS": {2, 0xCB, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VRSQRT28PD": {2, 0xCC, 1, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRSQRT28PS": {2, 0xCC, 0, 1, -1, vexRM, [3]int{0, 0, 64}},
|
||||||
|
"VRSQRT28SD": {2, 0xCD, 1, 1, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VRSQRT28SS": {2, 0xCD, 0, 1, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VSQRTPD": {1, 0x51, 1, 1, -1, vexRM, [3]int{16, 32, 64}},
|
||||||
|
"VSQRTSD": {1, 0x51, 1, 3, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3, [3]int{4, 0, 0}},
|
||||||
|
"VUCOMISD": {1, 0x2E, 1, 1, -1, vexRM, [3]int{8, 0, 0}},
|
||||||
|
"VXORPD": {1, 0x57, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128/256/512.0F.F3/F2.W0, word shuffles with an immediate
|
||||||
|
// ($imm, src, dst: reg = dst, rm = src, imm8). The F3/F2 prefixes
|
||||||
|
// split the high/low lane spellings.
|
||||||
|
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128.66.0F3A, lane extract to a general-purpose register or
|
||||||
|
// memory ($imm, xsrc, GPR/mem dst: reg = source, rm = destination).
|
||||||
|
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtractGPR, [3]int{1, 1, 1}},
|
||||||
|
"VPEXTRW": {3, 0x15, 0, 1, -1, vexExtractGPR, [3]int{2, 2, 2}},
|
||||||
|
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtractGPR, [3]int{4, 4, 4}},
|
||||||
|
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtractGPR, [3]int{8, 8, 8}},
|
||||||
|
|
||||||
|
// EVEX.66.0F3A.W1, the qword permutes with an immediate control
|
||||||
|
// ($imm, src, dst: reg = dst, rm = src, imm8); the register-count
|
||||||
|
// forms live in evexRegFormTable.
|
||||||
|
"VPERMQ": {3, 0x00, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VPERMPD": {3, 0x01, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.66.0F3A, the packed permute shuffles with an immediate control.
|
||||||
|
"VPERMILPS": {3, 0x04, 0, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
"VPERMILPD": {3, 0x05, 1, 1, -1, vexImmRM, [3]int{16, 32, 64}},
|
||||||
|
|
||||||
|
// EVEX.128.0F.W0, high/low half moves. VMOVHPS carries the
|
||||||
|
// three-operand insert form (rm = m64 source, vvvv = preserved,
|
||||||
|
// reg = dst) and the two-operand store (reg = source, rm = m64);
|
||||||
|
// the encoder splits on the operand count. VMOVLHPS is the
|
||||||
|
// three-operand form alone.
|
||||||
|
"VMOVHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
"VMOVLHPS": {1, 0x16, 0, 0, -1, vexNDS3, [3]int{8, 0, 0}},
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexQuad describes one quad-register instruction: the opcode under
|
||||||
|
// EVEX.0F38.W0 with the F2 mandatory prefix, and the width of the vector
|
||||||
|
// registers the bracketed list and the destination take (512-bit ZMM for
|
||||||
|
// the packed forms, 128-bit XMM for the scalar ones).
|
||||||
|
type evexQuad struct {
|
||||||
|
opcode byte
|
||||||
|
width int // register width in bytes: 64 (ZMM) or 16 (XMM)
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexQuadTable maps the quad-register instructions (the 4FMAPS and 4VNNIW
|
||||||
|
// families) to their encoding. The operand shape is fixed: a single memory
|
||||||
|
// source in r/m, the bracketed register list whose LOW register travels the
|
||||||
|
// inverted 5-bit V'VVVV field, an optional opmask in aaa and the vector
|
||||||
|
// destination in reg. The vector length follows the destination (512-bit
|
||||||
|
// for the ZMM list forms, 128-bit for the scalar ones) while the disp8×N
|
||||||
|
// multiplier stays 16 for every member, the toolchain's own tuple choice.
|
||||||
|
var evexQuadTable = map[string]evexQuad{
|
||||||
|
"V4FMADDPS": {0x9A, 64},
|
||||||
|
"V4FMADDSS": {0x9B, 16},
|
||||||
|
"V4FNMADDPS": {0xAA, 64},
|
||||||
|
"V4FNMADDSS": {0xAB, 16},
|
||||||
|
"VP4DPWSSD": {0x52, 64},
|
||||||
|
"VP4DPWSSDS": {0x53, 64},
|
||||||
|
}
|
||||||
|
|
||||||
|
// isEvexQuad reports whether the mnemonic is a quad-register instruction.
|
||||||
|
func isEvexQuad(upper string) bool {
|
||||||
|
_, ok := evexQuadTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeEvexQuad encodes the quad-register form: OP mem, [Zn-Zn+3], (K), dst.
|
||||||
|
// The register list is the VVVV-side source: its low register fills the
|
||||||
|
// inverted V'VVVV bits, which is why an indexed memory source above Z15 (no
|
||||||
|
// spare EVEX.X bit once V' is taken) is refused. Masking rides the standard
|
||||||
|
// aaa field, zeroing keeps the usual requires-a-mask rule, and no other
|
||||||
|
// suffix applies.
|
||||||
|
func (e *enc) encodeEvexQuad(mnem string, q evexQuad, ops []Operand, sfx evexSuffix) error {
|
||||||
|
if len(ops) != 3 && len(ops) != 4 {
|
||||||
|
return fmt.Errorf("%s expects 3 or 4 operands (mem, [Zn-Zn+3], (K), dst), got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
mem, lst := ops[0], ops[1]
|
||||||
|
dst := ops[len(ops)-1]
|
||||||
|
mask := 0
|
||||||
|
if len(ops) == 4 {
|
||||||
|
k, ok := ops[2].(Reg)
|
||||||
|
if !ok || !k.mask {
|
||||||
|
return fmt.Errorf("%s: third operand must be an opmask register", mnem)
|
||||||
|
}
|
||||||
|
if k.idx == 0 {
|
||||||
|
return fmt.Errorf("k0 is not a usable mask register")
|
||||||
|
}
|
||||||
|
mask = k.idx
|
||||||
|
}
|
||||||
|
list, ok := lst.(RegList)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s: second operand must be a four-register list", mnem)
|
||||||
|
}
|
||||||
|
if list.Lo.size != q.width {
|
||||||
|
return fmt.Errorf("%s: the register list must hold %d-bit vector registers", mnem, q.width*8)
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
if dstReg.size != q.width {
|
||||||
|
return fmt.Errorf("%s: the destination must be a %d-bit vector register", mnem, q.width*8)
|
||||||
|
}
|
||||||
|
if !memOperand(mem) {
|
||||||
|
return fmt.Errorf("%s: the source must be a memory operand", mnem)
|
||||||
|
}
|
||||||
|
// The list owns V'VVVV; a scaled index in the EVEX-only half would fold
|
||||||
|
// its fifth bit into the same field the list's low register occupies.
|
||||||
|
if m, ok := mem.(Mem); ok && m.HasIndex && m.Index.idx >= 16 {
|
||||||
|
return fmt.Errorf("%s: an index register above Z15 has no EVEX bit free", mnem)
|
||||||
|
}
|
||||||
|
if sfx.zeroing && mask == 0 {
|
||||||
|
return fmt.Errorf("%s: zeroing (.Z) requires a mask register", mnem)
|
||||||
|
}
|
||||||
|
spec := evexSpec{mapSel: 2, opcode: q.opcode, w: 0, pp: 3, opdigit: -1, n: [3]int{16, 16, 16}}
|
||||||
|
// The vector length follows the destination (512-bit for the ZMM forms,
|
||||||
|
// 128-bit for the scalar ones), exactly as the oracle encodes it.
|
||||||
|
return e.emitEvexFields(spec, dstReg.vecLenBit(), dstReg.idx, list.Lo.idx, mem, mask, sfx)
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
// evexBcastSpec describes an EVEX broadcast (VPBROADCASTD/Q): the opcode
|
||||||
@@ -520,32 +855,38 @@ type evexMoveSpec struct {
|
|||||||
n [3]int
|
n [3]int
|
||||||
vecOK bool // the non-memory operand may be a vector register
|
vecOK bool // the non-memory operand may be a vector register
|
||||||
xmmOnly bool // wider than XMM registers are rejected
|
xmmOnly bool // wider than XMM registers are rejected
|
||||||
|
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
|
||||||
var evexMoveTable = map[string]evexMoveSpec{
|
var evexMoveTable = map[string]evexMoveSpec{
|
||||||
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
|
||||||
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
|
||||||
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
|
||||||
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
|
||||||
// semantics).
|
// semantics).
|
||||||
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
|
||||||
// encoding).
|
// encoding).
|
||||||
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512, aligned packed moves.
|
// EVEX.128/256/512, aligned packed moves.
|
||||||
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128/256/512.66.0F, aligned integer moves.
|
// EVEX.128/256/512.66.0F, aligned integer moves.
|
||||||
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false},
|
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false},
|
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
|
||||||
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
|
||||||
// three-operand register form is not supported).
|
// three-operand register form is not supported).
|
||||||
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true},
|
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
|
||||||
|
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
|
||||||
|
// three-operand register form (VMOVSD dst, src1, src2).
|
||||||
|
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
|
||||||
|
// EVEX.128/256/512.0F.W0, unaligned packed single move.
|
||||||
|
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
|
||||||
}
|
}
|
||||||
|
|
||||||
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
|
||||||
@@ -556,8 +897,10 @@ func isEvex(mnemUpper string) bool {
|
|||||||
if _, ok := evexBcastTable[mnemUpper]; ok {
|
if _, ok := evexBcastTable[mnemUpper]; ok {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
_, ok := evexMoveTable[mnemUpper]
|
if _, ok := evexMoveTable[mnemUpper]; ok {
|
||||||
return ok
|
return true
|
||||||
|
}
|
||||||
|
return isEvexQuad(mnemUpper)
|
||||||
}
|
}
|
||||||
|
|
||||||
// evexRequired reports whether the operands force the EVEX encoding of a
|
// evexRequired reports whether the operands force the EVEX encoding of a
|
||||||
@@ -570,6 +913,13 @@ func evexRequired(upper string, ops []Operand) bool {
|
|||||||
if !inVex && !inVexMove {
|
if !inVex && !inVexMove {
|
||||||
return true // EVEX-only mnemonic
|
return true // EVEX-only mnemonic
|
||||||
}
|
}
|
||||||
|
// The byte-quad shifts have VEX register forms but EVEX-only memory
|
||||||
|
// forms: a memory count source forces the EVEX encoding.
|
||||||
|
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
|
||||||
|
if slices.ContainsFunc(ops, memOperand) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
for _, op := range ops {
|
for _, op := range ops {
|
||||||
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
if r, ok := op.(Reg); ok && (r.size == 64 || r.mask || (r.isVec() && r.idx >= 16)) {
|
||||||
return true
|
return true
|
||||||
@@ -732,6 +1082,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
return e.encodeEvexRM(spec, ops, 0, sfx)
|
return e.encodeEvexRM(spec, ops, 0, sfx)
|
||||||
}
|
}
|
||||||
spec, inTable := evexTable[mnemUpper]
|
spec, inTable := evexTable[mnemUpper]
|
||||||
|
if q, ok := evexQuadTable[mnemUpper]; ok {
|
||||||
|
// The quad-register family carries no rounding, SAE or broadcast;
|
||||||
|
// only masking and zeroing apply.
|
||||||
|
if sfx.sae || sfx.bcst || sfx.rounding >= 0 {
|
||||||
|
return fmt.Errorf("%s takes no rounding/SAE/broadcast suffix", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexQuad(mnemUpper, q, ops, sfx)
|
||||||
|
}
|
||||||
if inTable {
|
if inTable {
|
||||||
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
if (sfx.rounding >= 0 || sfx.sae) && !evexRound[mnemUpper] {
|
||||||
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
|
return fmt.Errorf("%s: rounding/SAE is not supported for this instruction", mnemUpper)
|
||||||
@@ -743,6 +1101,27 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
}
|
}
|
||||||
spec.n = [3]int{n, n, n}
|
spec.n = [3]int{n, n, n}
|
||||||
}
|
}
|
||||||
|
// A mnemonic with an immediate and a register spelling (the
|
||||||
|
// variable-count shifts, the permutes) encodes the register one
|
||||||
|
// when the first operand is not an immediate.
|
||||||
|
if len(ops) > 0 {
|
||||||
|
if _, isImm := ops[0].(Imm); !isImm {
|
||||||
|
if alt, ok := evexRegFormTable[mnemUpper]; ok {
|
||||||
|
spec, inTable = alt, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The high/low half moves split by operand count: three operands
|
||||||
|
// insert, two store (VMOVHPS m64, X1).
|
||||||
|
if hs, ok := evexHptrTable[mnemUpper]; ok {
|
||||||
|
if len(ops) == 2 {
|
||||||
|
if hs.store.opcode == 0 {
|
||||||
|
return fmt.Errorf("%s has no two-operand form", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexRMRev(hs.store, ops, 0, sfx)
|
||||||
|
}
|
||||||
|
spec = hs.insert
|
||||||
|
}
|
||||||
} else if sfx.evexOnly() {
|
} else if sfx.evexOnly() {
|
||||||
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
return fmt.Errorf("%s: the instruction does not take rounding/SAE/broadcast suffixes", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -802,6 +1181,12 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
}
|
}
|
||||||
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
return e.encodeEvexMove(mnemUpper, ms, ops, mask, sfx)
|
||||||
}
|
}
|
||||||
|
if ps, ok := evexPrefGatherTable[mnemUpper]; ok {
|
||||||
|
if sfx.any() {
|
||||||
|
return fmt.Errorf("%s takes no EVEX suffixes", mnemUpper)
|
||||||
|
}
|
||||||
|
return e.encodeEvexPrefGather(mnemUpper, ps, ops, mask, sfx)
|
||||||
|
}
|
||||||
if !inTable {
|
if !inTable {
|
||||||
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
return fmt.Errorf("unsupported instruction %q for ZMM/K operands", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -820,6 +1205,8 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
|
|||||||
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
return e.encodeEvexNDS3Imm(spec, ops, mask, sfx)
|
||||||
case vexExtract:
|
case vexExtract:
|
||||||
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
return e.encodeEvexExtract(spec, ops, mask, sfx)
|
||||||
|
case vexExtractGPR:
|
||||||
|
return e.encodeEvexExtractGPR(spec, ops, mask, sfx)
|
||||||
case vexRMSrcLen:
|
case vexRMSrcLen:
|
||||||
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
return e.encodeEvexRMSrcLen(spec, ops, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -899,6 +1286,11 @@ func (e *enc) encodeEvexImmRM(spec evexSpec, ops []Operand, mask int, sfx evexSu
|
|||||||
if dstReg.mask {
|
if dstReg.mask {
|
||||||
if r, ok := src.(Reg); ok && r.isVec() {
|
if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
ll = r.vecLenBit()
|
ll = r.vecLenBit()
|
||||||
|
} else if l, err := soleLen(spec.n); err == nil {
|
||||||
|
// A memory source with a length-fixed mnemonic
|
||||||
|
// (VFPCLASSPDX/Y/Z): the length comes from the table's
|
||||||
|
// single valid slot, not from the operand.
|
||||||
|
ll = l
|
||||||
}
|
}
|
||||||
} else if r, ok := src.(Reg); ok && r.isVec() {
|
} else if r, ok := src.(Reg); ok && r.isVec() {
|
||||||
ll = r.vecLenBit()
|
ll = r.vecLenBit()
|
||||||
@@ -925,9 +1317,11 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
|
|||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("shift count must be an immediate")
|
return fmt.Errorf("shift count must be an immediate")
|
||||||
}
|
}
|
||||||
srcReg, ok := src.(Reg)
|
// The count source is a vector register or memory; the length the L'L
|
||||||
if !ok || !srcReg.isVec() {
|
// field and the disp8×N multiplier follow is the destination's either
|
||||||
return fmt.Errorf("shift source must be a vector register")
|
// way.
|
||||||
|
if !vecOrMem(src) {
|
||||||
|
return fmt.Errorf("shift source must be a vector register or memory")
|
||||||
}
|
}
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || !dstReg.isVec() {
|
||||||
@@ -937,7 +1331,7 @@ func (e *enc) encodeEvexShiftImm(spec evexSpec, ops []Operand, mask int, sfx eve
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, srcReg, mask, sfx); err != nil {
|
if err := e.emitEvexFields(spec, dstReg.vecLenBit(), spec.opdigit, dstReg.idx, src, mask, sfx); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
e.out = append(e.out, immByte)
|
e.out = append(e.out, immByte)
|
||||||
@@ -1009,10 +1403,77 @@ func (e *enc) encodeEvexExtract(spec evexSpec, ops []Operand, mask int, sfx evex
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeEvexExtractGPR encodes the lane extract to a general-purpose
|
||||||
|
// register or memory: OP $imm, xsrc, dst (reg = the XMM source, rm = the
|
||||||
|
// destination, imm8). The encoding is 128-bit regardless of register
|
||||||
|
// numbers, so L'L is fixed at 0 and the disp8×N multiplier is the extracted
|
||||||
|
// element size the table carries.
|
||||||
|
func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("extract expects 3 operands ($imm, xsrc, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("extract lane must be an immediate")
|
||||||
|
}
|
||||||
|
srcReg, ok := src.(Reg)
|
||||||
|
if !ok || !srcReg.isVec() {
|
||||||
|
return fmt.Errorf("extract source must be a vector register")
|
||||||
|
}
|
||||||
|
switch dst.(type) {
|
||||||
|
case Reg:
|
||||||
|
if dst.(Reg).isVec() {
|
||||||
|
return fmt.Errorf("extract destination must be a general-purpose register or memory")
|
||||||
|
}
|
||||||
|
case Mem, sbMem:
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("extract destination must be a general-purpose register or memory")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := e.emitEvexFields(spec, 0, srcReg.idx, -1, dst, mask, sfx); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
// encodeEvexMove encodes a two-operand EVEX move; a vector→vector move uses
|
||||||
// the store-form opcode (reg = source, rm = destination), matching the Go
|
// the store-form opcode (reg = source, rm = destination), matching the Go
|
||||||
// assembler.
|
// assembler. The scalar moves also carry a three-operand register form
|
||||||
|
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
|
||||||
|
// opens.
|
||||||
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
|
if len(ops) == 3 {
|
||||||
|
if !ms.nds3 {
|
||||||
|
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
// The masked scalar register form keeps the Go assembler's own
|
||||||
|
// layout: the store opcode with reg = op0, vvvv = op1 and the
|
||||||
|
// destination in r/m (op2) — the bytes go tool asm emits, not
|
||||||
|
// the manual's NDS reading.
|
||||||
|
src, src1, dst := ops[0], ops[1], ops[2]
|
||||||
|
reg, ok := src.(Reg)
|
||||||
|
if !ok || !reg.isVec() {
|
||||||
|
return fmt.Errorf("%s: first operand must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || !vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: second operand must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s: destination must be a vector register", mnem)
|
||||||
|
}
|
||||||
|
if ms.xmmOnly && (reg.size != 16 || vvvvReg.size != 16 || dstReg.size != 16) {
|
||||||
|
return fmt.Errorf("%s operates on XMM registers only", mnem)
|
||||||
|
}
|
||||||
|
spec := evexSpec{mapSel: ms.mapSel, opcode: ms.store, w: ms.w, pp: ms.pp, opdigit: -1, n: ms.n}
|
||||||
|
return e.emitEvexFields(spec, dstReg.vecLenBit(), reg.idx, vvvvReg.idx, dst, mask, sfx)
|
||||||
|
}
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("EVEX move expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
@@ -1132,12 +1593,20 @@ func (e *enc) encodeEvexBcast(bs evexBcastSpec, ops []Operand, mask int, sfx eve
|
|||||||
return fmt.Errorf("broadcast destination must be a vector register")
|
return fmt.Errorf("broadcast destination must be a vector register")
|
||||||
}
|
}
|
||||||
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
spec := evexSpec{mapSel: bs.mapSel, w: bs.w, pp: 1, opdigit: -1}
|
||||||
switch src.(type) {
|
switch r := src.(type) {
|
||||||
case Mem, sbMem:
|
case Mem, sbMem:
|
||||||
spec.opcode = bs.opMem
|
spec.opcode = bs.opMem
|
||||||
spec.n = [3]int{bs.n, bs.n, bs.n}
|
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||||||
case Reg:
|
case Reg:
|
||||||
spec.opcode = bs.opReg
|
// A GPR source uses the register broadcast opcode; a vector
|
||||||
|
// source shares the xmm/mem one (the low byte is copied from
|
||||||
|
// the lane or from the memory operand).
|
||||||
|
if r.isVec() {
|
||||||
|
spec.opcode = bs.opMem
|
||||||
|
spec.n = [3]int{bs.n, bs.n, bs.n}
|
||||||
|
} else {
|
||||||
|
spec.opcode = bs.opReg
|
||||||
|
}
|
||||||
default:
|
default:
|
||||||
return fmt.Errorf("broadcast source must be a register or memory")
|
return fmt.Errorf("broadcast source must be a register or memory")
|
||||||
}
|
}
|
||||||
@@ -1340,6 +1809,102 @@ func isScatter(upper string) bool {
|
|||||||
return ok
|
return ok
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// isEvexPrefGather reports whether the mnemonic is a gather/scatter
|
||||||
|
// prefetch hint.
|
||||||
|
func isEvexPrefGather(upper string) bool {
|
||||||
|
_, ok := evexPrefGatherTable[upper]
|
||||||
|
return ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexRegFormTable holds the register-count twin of the immediate-form
|
||||||
|
// entries in evexTable. Several mnemonics name two encodings: an immediate
|
||||||
|
// count or control ($imm, src, dst …) and a register-count one whose second
|
||||||
|
// operand is a vector register or memory (count, src2, src1, dst). The
|
||||||
|
// immediate spelling lives in evexTable, this table carries the register
|
||||||
|
// spelling, and encodeEvex picks by whether the first operand is an
|
||||||
|
// immediate, the way vexVarShift does on the VEX side.
|
||||||
|
var evexRegFormTable = map[string]evexSpec{
|
||||||
|
"VPSLLD": {1, 0xF2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSLLQ": {1, 0xF3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSLLW": {1, 0xF1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRAD": {1, 0xE2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRAQ": {1, 0xE2, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRAW": {1, 0xE1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRLD": {1, 0xD2, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRLQ": {1, 0xD3, 1, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
"VPSRLW": {1, 0xD1, 0, 1, -1, vexNDS3, [3]int{16, 16, 16}},
|
||||||
|
// EVEX.NDS.0F38.W1, the register-count permutes (the immediate
|
||||||
|
// controls live in evexTable under 0F3A).
|
||||||
|
"VPERMQ": {2, 0x36, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMPD": {2, 0x16, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
// EVEX.NDS.0F38, the register-count permil shuffles.
|
||||||
|
"VPERMILPS": {2, 0x0C, 0, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
"VPERMILPD": {2, 0x0D, 1, 1, -1, vexNDS3, [3]int{16, 32, 64}},
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexPrefGatherSpec describes a gather/scatter prefetch hint: one memory
|
||||||
|
// operand with a VSIB index and an opmask register, no destination. The
|
||||||
|
// ModRM.reg field carries a fixed /digit, the L'L field is fixed at 512, and
|
||||||
|
// the mask register is the instruction's only register operand.
|
||||||
|
type evexPrefGatherSpec struct {
|
||||||
|
mapSel int
|
||||||
|
opcode byte
|
||||||
|
w int
|
||||||
|
pp int
|
||||||
|
opdigit int
|
||||||
|
n int
|
||||||
|
}
|
||||||
|
|
||||||
|
var evexPrefGatherTable = map[string]evexPrefGatherSpec{
|
||||||
|
"VGATHERPF0DPD": {2, 0xC6, 1, 1, 1, 8},
|
||||||
|
"VGATHERPF0DPS": {2, 0xC6, 0, 1, 1, 4},
|
||||||
|
"VGATHERPF0QPD": {2, 0xC7, 1, 1, 1, 8},
|
||||||
|
"VGATHERPF0QPS": {2, 0xC7, 0, 1, 1, 4},
|
||||||
|
"VGATHERPF1DPD": {2, 0xC6, 1, 1, 2, 8},
|
||||||
|
"VGATHERPF1DPS": {2, 0xC6, 0, 1, 2, 4},
|
||||||
|
"VGATHERPF1QPD": {2, 0xC7, 1, 1, 2, 8},
|
||||||
|
"VGATHERPF1QPS": {2, 0xC7, 0, 1, 2, 4},
|
||||||
|
"VSCATTERPF0DPD": {2, 0xC6, 1, 1, 5, 8},
|
||||||
|
"VSCATTERPF0DPS": {2, 0xC6, 0, 1, 5, 4},
|
||||||
|
"VSCATTERPF0QPD": {2, 0xC7, 1, 1, 5, 8},
|
||||||
|
"VSCATTERPF0QPS": {2, 0xC7, 0, 1, 5, 4},
|
||||||
|
"VSCATTERPF1DPD": {2, 0xC6, 1, 1, 6, 8},
|
||||||
|
"VSCATTERPF1DPS": {2, 0xC6, 0, 1, 6, 4},
|
||||||
|
"VSCATTERPF1QPD": {2, 0xC7, 1, 1, 6, 8},
|
||||||
|
"VSCATTERPF1QPS": {2, 0xC7, 0, 1, 6, 4},
|
||||||
|
}
|
||||||
|
|
||||||
|
// evexHptrSpec describes the high/low half moves (VMOVHPS family): the
|
||||||
|
// three-operand insert shares an opcode with a two-operand store whose
|
||||||
|
// source is the vector register and whose destination is m64.
|
||||||
|
type evexHptrSpec struct {
|
||||||
|
insert evexSpec
|
||||||
|
store evexSpec // store.opcode == 0 when the mnemonic has no store form
|
||||||
|
}
|
||||||
|
|
||||||
|
var evexHptrTable = map[string]evexHptrSpec{
|
||||||
|
"VMOVHPS": {
|
||||||
|
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
|
||||||
|
store: evexSpec{mapSel: 1, opcode: 0x17, w: 0, pp: 0, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
|
||||||
|
},
|
||||||
|
"VMOVLHPS": {
|
||||||
|
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
|
||||||
|
func (e *enc) encodeEvexPrefGather(upper string, ps evexPrefGatherSpec, ops []Operand, mask int, sfx evexSuffix) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands (K, vsib memory), got %d", upper, len(ops)+1)
|
||||||
|
}
|
||||||
|
m, ok := ops[0].(Mem)
|
||||||
|
if !ok || !m.HasIndex || !m.Index.isVec() {
|
||||||
|
return fmt.Errorf("%s: operand must be a VSIB memory reference with a vector index", upper)
|
||||||
|
}
|
||||||
|
spec := evexSpec{mapSel: ps.mapSel, opcode: ps.opcode, w: ps.w, pp: ps.pp, opdigit: ps.opdigit, n: [3]int{ps.n, ps.n, ps.n}}
|
||||||
|
return e.emitEvexFields(spec, 2, ps.opdigit, -1, m, mask, sfx)
|
||||||
|
}
|
||||||
|
|
||||||
// vsibLen validates a VSIB memory operand (the index must be a vector
|
// vsibLen validates a VSIB memory operand (the index must be a vector
|
||||||
// register) and returns it with the vector length the index selects, the
|
// register) and returns it with the vector length the index selects, the
|
||||||
// EVEX L'L field follows the index register, not the data register.
|
// EVEX L'L field follows the index register, not the data register.
|
||||||
@@ -1361,7 +1926,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if mask != 0 || sfx.any() {
|
if mask != 0 || sfx.any() {
|
||||||
// EVEX form: OP vsib, K, dst.
|
// EVEX form: OP vsib, K, dst. The L'L field is the wider of the
|
||||||
|
// index and the data register lengths (the Go assembler's
|
||||||
|
// layout); the disp8×N multiplier stays the index element size.
|
||||||
if len(rest) != 2 {
|
if len(rest) != 2 {
|
||||||
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
return fmt.Errorf("%s expects 3 operands (vsib, K, dst), got %d", upper, len(ops))
|
||||||
}
|
}
|
||||||
@@ -1373,6 +1940,9 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
|
|||||||
if !ok || !dst.isVec() {
|
if !ok || !dst.isVec() {
|
||||||
return fmt.Errorf("%s: destination must be a vector register", upper)
|
return fmt.Errorf("%s: destination must be a vector register", upper)
|
||||||
}
|
}
|
||||||
|
if d := dst.vecLenBit(); d > ll {
|
||||||
|
ll = d
|
||||||
|
}
|
||||||
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
evex := evexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1, n: [3]int{gs.n, gs.n, gs.n}}
|
||||||
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
return e.emitEvexFields(evex, ll, dst.idx, -1, vsib, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -1422,6 +1992,11 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
// The L'L field is the wider of the data register and the VSIB index
|
||||||
|
// lengths, the bytes go tool asm emits.
|
||||||
|
if d := src.vecLenBit(); d > ll {
|
||||||
|
ll = d
|
||||||
|
}
|
||||||
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
evex := evexSpec{mapSel: 2, opcode: ss.opcode, w: ss.w, pp: 1, opdigit: -1, n: [3]int{ss.n, ss.n, ss.n}}
|
||||||
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
return e.emitEvexFields(evex, ll, src.idx, -1, vsib, mask, sfx)
|
||||||
}
|
}
|
||||||
@@ -1432,21 +2007,25 @@ func (e *enc) encodeScatter(upper string, ss gatherSpec, ops []Operand, sfx evex
|
|||||||
var evexKOperand = map[string]bool{
|
var evexKOperand = map[string]bool{
|
||||||
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
"VPMOVM2B": true, "VPMOVM2W": true, "VPMOVM2D": true, "VPMOVM2Q": true,
|
||||||
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
"VPMOVB2M": true, "VPMOVW2M": true, "VPMOVD2M": true, "VPMOVQ2M": true,
|
||||||
|
// The K-to-vector broadcast reads its opmask source from r/m.
|
||||||
|
"VPBROADCASTMB2Q": true, "VPBROADCASTMW2D": true,
|
||||||
}
|
}
|
||||||
|
|
||||||
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
// kmovSpec describes a KMOV width: the opcode depends on the operand
|
||||||
// direction, kk (k/mem → K is 90, k → k uses the same), kmem (K → mem),
|
// direction, kk (k → k), kmem (k → mem), gprk (GPR/mem → k) and kgpr
|
||||||
// gprk (GPR/mem → K), kgpr (K → GPR), and the GPR forms carry a mandatory
|
// (k → GPR). Each direction group carries its own mandatory prefix and W:
|
||||||
// prefix and W for the wider widths.
|
// the k-destination/source forms share one pair, the GPR forms another.
|
||||||
type kmovSpec struct {
|
type kmovSpec struct {
|
||||||
kk, kmem, gprk, kgpr byte
|
kk, kmem, gprk, kgpr byte
|
||||||
gprPP int
|
kPP, kW int // prefix and VEX.W for the k forms
|
||||||
w int
|
gprPP, gprW int // prefix and VEX.W for the GPR forms
|
||||||
}
|
}
|
||||||
|
|
||||||
var kmovTable = map[string]kmovSpec{
|
var kmovTable = map[string]kmovSpec{
|
||||||
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0},
|
"KMOVW": {0x90, 0x91, 0x92, 0x93, 0, 0, 0, 0},
|
||||||
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 3, 1},
|
"KMOVB": {0x90, 0x91, 0x92, 0x93, 1, 0, 1, 0},
|
||||||
|
"KMOVD": {0x90, 0x91, 0x92, 0x93, 1, 1, 3, 0},
|
||||||
|
"KMOVQ": {0x90, 0x91, 0x92, 0x93, 0, 1, 3, 1},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
// encodeKmov encodes a KMOV width, selecting the opcode by direction.
|
||||||
@@ -1460,14 +2039,14 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
|||||||
dstReg, dstIsReg := dst.(Reg)
|
dstReg, dstIsReg := dst.(Reg)
|
||||||
srcK := srcIsReg && srcReg.mask
|
srcK := srcIsReg && srcReg.mask
|
||||||
dstK := dstIsReg && dstReg.mask
|
dstK := dstIsReg && dstReg.mask
|
||||||
spec := vexSpec{mapSel: 1, w: ks.w, pp: 0, opdigit: -1}
|
|
||||||
switch {
|
switch {
|
||||||
case srcK && dstK:
|
case srcK && dstK:
|
||||||
spec.opcode = ks.kk // k ← k: reg = dst, rm = src
|
// k ← k: reg = dst, rm = src.
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: ks.kk, w: ks.kW, pp: ks.kPP, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||||
case srcK && dstIsReg:
|
case srcK && dstIsReg:
|
||||||
spec.opcode = ks.kgpr // GPR ← k: reg = dst, rm = src
|
// GPR ← k: reg = dst, rm = src.
|
||||||
spec.pp = ks.gprPP
|
spec := vexSpec{mapSel: 1, opcode: ks.kgpr, w: ks.gprW, pp: ks.gprPP, opdigit: -1}
|
||||||
rBit := 0
|
rBit := 0
|
||||||
if dstReg.idx >= 8 {
|
if dstReg.idx >= 8 {
|
||||||
rBit = 1
|
rBit = 1
|
||||||
@@ -1477,11 +2056,17 @@ func (e *enc) encodeKmov(upper string, ops []Operand) error {
|
|||||||
if _, ok := dst.(Mem); !ok {
|
if _, ok := dst.(Mem); !ok {
|
||||||
return fmt.Errorf("%s: invalid destination operand", upper)
|
return fmt.Errorf("%s: invalid destination operand", upper)
|
||||||
}
|
}
|
||||||
spec.opcode = ks.kmem // mem ← k: reg = src, rm = dst
|
// mem ← k: reg = src, rm = dst.
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: ks.kmem, w: ks.kW, pp: ks.kPP, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
return e.emitVexFields(spec, 0, srcReg.idx&7, 0, 15, dst)
|
||||||
case dstK:
|
case dstK:
|
||||||
spec.opcode = ks.gprk // k ← GPR/mem: reg = dst, rm = src
|
// k ← GPR: reg = dst, rm = src. A memory source shares the k ← k
|
||||||
spec.pp = ks.gprPP
|
// opcode and prefix group (the ykmovb layout the Go assembler uses).
|
||||||
|
opcode, w, pp := ks.gprk, ks.gprW, ks.gprPP
|
||||||
|
if memOperand(src) {
|
||||||
|
opcode, w, pp = ks.kk, ks.kW, ks.kPP
|
||||||
|
}
|
||||||
|
spec := vexSpec{mapSel: 1, opcode: opcode, w: w, pp: pp, opdigit: -1}
|
||||||
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
return e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("%s requires a K register operand", upper)
|
return fmt.Errorf("%s requires a K register operand", upper)
|
||||||
@@ -1524,6 +2109,7 @@ var kOpsTable = map[string]kOpSpec{
|
|||||||
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
|
"KXORD": {1, 0x47, 1, 1, 1, vexNDS3},
|
||||||
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
|
"KXORQ": {1, 0x47, 1, 0, 1, vexNDS3},
|
||||||
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
"KUNPCKBW": {1, 0x4B, 0, 1, 1, vexNDS3},
|
||||||
|
"KUNPCKWD": {1, 0x4B, 0, 0, 1, vexNDS3},
|
||||||
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
"KUNPCKDQ": {1, 0x4B, 1, 0, 1, vexNDS3},
|
||||||
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
"KADDB": {1, 0x4A, 0, 1, 1, vexNDS3},
|
||||||
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
"KADDW": {1, 0x4A, 0, 0, 1, vexNDS3},
|
||||||
|
|||||||
@@ -38,6 +38,15 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
{"VADDPD Z11,Z10,Z10", "VADDPD", []Operand{vreg(t, "Z11"), vreg(t, "Z10"), vreg(t, "Z10")}, "6251ad4858d3"},
|
||||||
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
{"VMULPD Z13,Z12,Z12", "VMULPD", []Operand{vreg(t, "Z13"), vreg(t, "Z12"), vreg(t, "Z12")}, "62519d4859e5"},
|
||||||
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
{"VFMADD231PD Z14,Z12,Z10", "VFMADD231PD", []Operand{vreg(t, "Z14"), vreg(t, "Z12"), vreg(t, "Z10")}, "62529d48b8d6"},
|
||||||
|
// The qword OR spelling always encodes through EVEX.
|
||||||
|
{"VPORQ Y0,Y1,Y2", "VPORQ", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "62f1f528ebd0"},
|
||||||
|
{"VPORQ X0,X1,X2", "VPORQ", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f1f508ebd0"},
|
||||||
|
// Byte permute and population count.
|
||||||
|
{"VPERMI2B X0,X1,X2", "VPERMI2B", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "62f2750875d0"},
|
||||||
|
{"VPOPCNTB X0,X1", "VPOPCNTB", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0854c8"},
|
||||||
|
{"VPOPCNTD X0,X1", "VPOPCNTD", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f27d0855c8"},
|
||||||
|
{"VPOPCNTD Y0,Y1", "VPOPCNTD", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "62f27d2855c8"},
|
||||||
|
{"VPOPCNTQ X0,X1", "VPOPCNTQ", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "62f2fd0855c8"},
|
||||||
// Align (NDS + imm8).
|
// Align (NDS + imm8).
|
||||||
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
{"VALIGND $12,Z12,Z0,Z1", "VALIGND", []Operand{Imm(12), vreg(t, "Z12"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803cc0c"},
|
||||||
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
{"VALIGND $15,Z9,Z0,Z1", "VALIGND", []Operand{Imm(15), vreg(t, "Z9"), vreg(t, "Z0"), vreg(t, "Z1")}, "62d37d4803c90f"},
|
||||||
@@ -52,6 +61,16 @@ func TestEvexGroundTruth(t *testing.T) {
|
|||||||
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
{"KMOVW K1,CX", "KMOVW", []Operand{vreg(t, "K1"), CX}, "c5f893c9"},
|
||||||
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
{"KMOVW K1,R12", "KMOVW", []Operand{vreg(t, "K1"), vreg(t, "R12")}, "c57893e1"},
|
||||||
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
{"KTESTW K1,K1", "KTESTW", []Operand{vreg(t, "K1"), vreg(t, "K1")}, "c5f899c9"},
|
||||||
|
{"KMOVB K1,K2", "KMOVB", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c5f990d1"},
|
||||||
|
{"KMOVB AX,K1", "KMOVB", []Operand{AX, vreg(t, "K1")}, "c5f992c8"},
|
||||||
|
{"KMOVB K1,AX", "KMOVB", []Operand{vreg(t, "K1"), AX}, "c5f993c1"},
|
||||||
|
{"KMOVB K1,(AX)", "KMOVB", []Operand{vreg(t, "K1"), Ptr(AX, 0, 1)}, "c5f99108"},
|
||||||
|
{"KMOVD K1,K2", "KMOVD", []Operand{vreg(t, "K1"), vreg(t, "K2")}, "c4e1f990d1"},
|
||||||
|
{"KMOVD AX,K1", "KMOVD", []Operand{AX, vreg(t, "K1")}, "c5fb92c8"},
|
||||||
|
{"KMOVD K1,AX", "KMOVD", []Operand{vreg(t, "K1"), AX}, "c5fb93c1"},
|
||||||
|
{"KMOVD K1,(AX)", "KMOVD", []Operand{vreg(t, "K1"), Ptr(AX, 0, 4)}, "c4e1f99108"},
|
||||||
|
{"KMOVB (AX),K1", "KMOVB", []Operand{Ptr(AX, 0, 1), vreg(t, "K1")}, "c5f99008"},
|
||||||
|
{"KMOVQ (AX),K1", "KMOVQ", []Operand{Ptr(AX, 0, 8), vreg(t, "K1")}, "c4e1f89008"},
|
||||||
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
// Moves, incl. disp8×N (64 for a 512-bit operand).
|
||||||
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
{"VMOVDQU32 (SI)(R15*4),Z3", "VMOVDQU32", []Operand{Idx(SI, vreg(t, "R15"), 4, 0, 64), vreg(t, "Z3")}, "62b17e486f1cbe"},
|
||||||
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
{"VMOVDQU32 4(SI)(AX*1),Z4", "VMOVDQU32", []Operand{Idx(SI, AX, 1, 4, 64), vreg(t, "Z4")}, "62f17e486fa40604000000"},
|
||||||
@@ -702,3 +721,216 @@ func hexCompact(b []byte) string {
|
|||||||
}
|
}
|
||||||
return string(out)
|
return string(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestAvx512CorpusFamilies pins representative encodings of the AVX-512
|
||||||
|
// families the toolchain's avx512enc corpus exercises: the bytes are the
|
||||||
|
// go tool asm output for exactly these operands, and the same families are
|
||||||
|
// covered end to end by the avx512_amd64.s differential kernel.
|
||||||
|
func TestAvx512CorpusFamilies(t *testing.T) {
|
||||||
|
vsib := func(base, idx string, scale int) Operand {
|
||||||
|
return Idx(vreg(t, base), vreg(t, idx), scale, 0, 0)
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
// AES rounds (EVEX NDS, VEX twin routed by operand width).
|
||||||
|
{"VAESDEC Z", "VAESDEC", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f26d48ded9"},
|
||||||
|
// Integer VNNI and the bit algorithm group.
|
||||||
|
{"VPDPBUSD", "VPDPBUSD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K2"), vreg(t, "Z3")}, "62f26d4a50d9"},
|
||||||
|
{"VPOPCNTW", "VPOPCNTW", []Operand{vreg(t, "Z1"), vreg(t, "K3"), vreg(t, "Z2")}, "62f2fd4b54d1"},
|
||||||
|
{"VPCONFLICTD", "VPCONFLICTD", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d49c4d1"},
|
||||||
|
{"VPLZCNTQ masked", "VPLZCNTQ", []Operand{vreg(t, "Z7"), vreg(t, "K1"), vreg(t, "Z8")}, "6272fd4944c7"},
|
||||||
|
{"VPERMT2B", "VPERMT2B", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f26d497dd9"},
|
||||||
|
{"VPMULTISHIFTQB", "VPMULTISHIFTQB", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z4")}, "62f2ed4b83e1"},
|
||||||
|
{"VDBPSADBW", "VDBPSADBW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K3"), vreg(t, "Z3")}, "62f36d4b42d903"},
|
||||||
|
{"VPSHUFBITQMB", "VPSHUFBITQMB", []Operand{vreg(t, "Z9"), vreg(t, "Z10"), vreg(t, "K3")}, "62d22d488fd9"},
|
||||||
|
{"VPTESTNMQ", "VPTESTNMQ", []Operand{vreg(t, "Z13"), vreg(t, "Z14"), vreg(t, "K5")}, "62d28e4827ed"},
|
||||||
|
// Permutations: immediate and register counts.
|
||||||
|
{"VALIGNQ", "VALIGNQ", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f3ed4903d903"},
|
||||||
|
{"VPERMQ imm", "VPERMQ", []Operand{Imm(1), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f3fd4a00d101"},
|
||||||
|
{"VPERMQ reg", "VPERMQ", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K2"), vreg(t, "Z5")}, "62f2dd4a36eb"},
|
||||||
|
{"VPERMPD reg", "VPERMPD", []Operand{vreg(t, "Z1"), vreg(t, "Z2"), vreg(t, "Z3")}, "62f2ed4816d9"},
|
||||||
|
{"VPERMILPS imm", "VPERMILPS", []Operand{Imm(5), vreg(t, "Z9"), vreg(t, "K2"), vreg(t, "Z10")}, "62537d4a04d105"},
|
||||||
|
{"VPERMILPS reg", "VPERMILPS", []Operand{vreg(t, "Z11"), vreg(t, "Z12"), vreg(t, "K2"), vreg(t, "Z13")}, "62521d4a0ceb"},
|
||||||
|
// Shifts: immediate, register-count and memory-count forms; the
|
||||||
|
// count source carries its own XMM tuple width.
|
||||||
|
{"VPSLLW imm mask", "VPSLLW", []Operand{Imm(3), vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z2")}, "62f16d4a71f103"},
|
||||||
|
{"VPSLLD reg count", "VPSLLD", []Operand{vreg(t, "X1"), vreg(t, "Z2"), vreg(t, "K1"), vreg(t, "Z3")}, "62f16d49f2d9"},
|
||||||
|
{"VPSLLDQ", "VPSLLDQ", []Operand{Imm(9), vreg(t, "Z7"), vreg(t, "Z8")}, "62f13d4873ff09"},
|
||||||
|
{"VPSRLDQ mem", "VPSRLDQ", []Operand{Imm(11), Ptr(SI, 16, 16), vreg(t, "Z4")}, "62f15d48739e100000000b"},
|
||||||
|
{"VPSRLVW", "VPSRLVW", []Operand{vreg(t, "Z3"), vreg(t, "Z4"), vreg(t, "K1"), vreg(t, "Z5")}, "62f2dd4910eb"},
|
||||||
|
// Conversions and shuffles with the F2 prefix and no prefix.
|
||||||
|
{"VCVTUDQ2PS", "VCVTUDQ2PS", []Operand{vreg(t, "Z1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f17f497ad1"},
|
||||||
|
{"VSHUFPS", "VSHUFPS", []Operand{Imm(2), vreg(t, "Z4"), vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f15449c6f402"},
|
||||||
|
// Gather and scatter prefetch hints (memory-only, /digit in reg).
|
||||||
|
{"VGATHERPF0DPD", "VGATHERPF0DPD", []Operand{vreg(t, "K5"), vsib("R10", "Y29", 8)}, "6292fd45c60cea"},
|
||||||
|
{"VSCATTERPF1DPS", "VSCATTERPF1DPS", []Operand{vreg(t, "K2"), vsib("R10", "Z28", 4)}, "62927d42c634a2"},
|
||||||
|
// Opmask broadcasts and the K logic.
|
||||||
|
{"VPBROADCASTMB2Q", "VPBROADCASTMB2Q", []Operand{vreg(t, "K1"), vreg(t, "Z2")}, "62f2fe482ad1"},
|
||||||
|
{"VPBROADCASTMW2D", "VPBROADCASTMW2D", []Operand{vreg(t, "K3"), vreg(t, "Z4")}, "62f27e483ae3"},
|
||||||
|
{"KUNPCKWD", "KUNPCKWD", []Operand{vreg(t, "K6"), vreg(t, "K4"), vreg(t, "K1")}, "c5dc4bce"},
|
||||||
|
{"KADDB", "KADDB", []Operand{vreg(t, "K2"), vreg(t, "K3"), vreg(t, "K5")}, "c5e54aea"},
|
||||||
|
// Lane extracts to general registers (EVEX and VEX routes).
|
||||||
|
{"VPEXTRB", "VPEXTRB", []Operand{Imm(3), vreg(t, "X26"), AX}, "62637d0814d003"},
|
||||||
|
{"VPEXTRD", "VPEXTRD", []Operand{Imm(1), vreg(t, "X26"), vreg(t, "R9")}, "62437d0816d101"},
|
||||||
|
{"VPEXTRD vex", "VPEXTRD", []Operand{Imm(1), vreg(t, "X2"), DI}, "c4e37916d701"},
|
||||||
|
{"VPINSRQ", "VPINSRQ", []Operand{Imm(1), DI, vreg(t, "X3"), vreg(t, "X4")}, "c4e3e122e701"},
|
||||||
|
// Moves: masked unaligned, masked scalar register form, half moves
|
||||||
|
// and non-temporal stores.
|
||||||
|
{"VMOVUPS mask", "VMOVUPS", []Operand{vreg(t, "Z1"), vreg(t, "K2"), vreg(t, "Z3")}, "62f17c4a11cb"},
|
||||||
|
{"VMOVSD 3op", "VMOVSD", []Operand{vreg(t, "X14"), vreg(t, "X5"), vreg(t, "K3"), vreg(t, "X22")}, "6231d70b11f6"},
|
||||||
|
{"VMOVSS 3op", "VMOVSS", []Operand{vreg(t, "X18"), vreg(t, "X3"), vreg(t, "K2"), vreg(t, "X25")}, "6281660a11d1"},
|
||||||
|
{"VMOVHPS insert", "VMOVHPS", []Operand{Ptr(SI, 0, 8), vreg(t, "X18"), vreg(t, "X19")}, "62e16c00161e"},
|
||||||
|
{"VMOVHPS store", "VMOVHPS", []Operand{vreg(t, "X20"), Ptr(SI, 8, 8)}, "62e17c08176601"},
|
||||||
|
{"VMOVLHPS", "VMOVLHPS", []Operand{vreg(t, "X16"), vreg(t, "X5"), vreg(t, "X17")}, "62a1540816c8"},
|
||||||
|
{"VMOVNTDQ", "VMOVNTDQ", []Operand{vreg(t, "Z7"), Ptr(SI, 0, 64)}, "62f17d48e73e"},
|
||||||
|
{"VMOVNTDQA", "VMOVNTDQA", []Operand{Ptr(SI, 64, 64), vreg(t, "Z8")}, "62727d482a4601"},
|
||||||
|
{"VMOVNTPS", "VMOVNTPS", []Operand{vreg(t, "Z9"), Ptr(SI, 0, 64)}, "62717c482b0e"},
|
||||||
|
// Scalar compares with and without the 66 prefix.
|
||||||
|
{"VCOMISD", "VCOMISD", []Operand{vreg(t, "X5"), vreg(t, "X6")}, "c5f92ff5"},
|
||||||
|
{"VUCOMISS", "VUCOMISS", []Operand{vreg(t, "X7"), vreg(t, "X8")}, "c5782ec7"},
|
||||||
|
// Floating point helpers.
|
||||||
|
{"VSQRTSD", "VSQRTSD", []Operand{vreg(t, "X1"), vreg(t, "X2"), vreg(t, "K1"), vreg(t, "X3")}, "62f1ef0951d9"},
|
||||||
|
{"VEXP2PD", "VEXP2PD", []Operand{vreg(t, "Z5"), vreg(t, "K1"), vreg(t, "Z6")}, "62f2fd49c8f5"},
|
||||||
|
{"VRCP28SD", "VRCP28SD", []Operand{vreg(t, "X9"), vreg(t, "X8"), vreg(t, "K1"), vreg(t, "X10")}, "6252bd09cbd1"},
|
||||||
|
{"VBROADCASTF32X2", "VBROADCASTF32X2", []Operand{vreg(t, "X1"), vreg(t, "K1"), vreg(t, "Z2")}, "62f27d4919d1"},
|
||||||
|
{"VPCOMPRESSB", "VPCOMPRESSB", []Operand{vreg(t, "Z1"), vreg(t, "K1"), Ptr(SI, 0, 64)}, "62f27d49630e"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexQuadRegisterGroundTruth pins the quad-register instructions (the
|
||||||
|
// 4FMAPS and 4VNNIW families) byte for byte against go tool asm: the memory
|
||||||
|
// source keeps r/m, the bracketed list's LOW register travels the inverted
|
||||||
|
// 5-bit V'VVVV field, the destination sits in reg, the opmask rides aaa and
|
||||||
|
// the vector length follows the destination (L'L=512 for the ZMM forms,
|
||||||
|
// 128 for the scalar ones) while the disp8×N multiplier stays 16 for every
|
||||||
|
// member. The x86 decoder has no view of these forms, so no decode check
|
||||||
|
// runs.
|
||||||
|
func TestEvexQuadRegisterGroundTruth(t *testing.T) {
|
||||||
|
sp := vreg(t, "RSP")
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
want string
|
||||||
|
}{
|
||||||
|
{"V4FMADDPS 17(SP) [Z0-Z3] K2 Z0", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4a9a842411000000"},
|
||||||
|
{"V4FMADDPS [Z10-Z13]", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z10"), vreg(t, "Z13")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f22f4a9a842411000000"},
|
||||||
|
{"V4FMADDPS [Z20-Z23]", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z20"), vreg(t, "Z23")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f25f429a842411000000"},
|
||||||
|
{"V4FMADDPS Z8 dst", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z8")},
|
||||||
|
"62727f4a9a842411000000"},
|
||||||
|
{"V4FMADDPS disp8x16", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 64, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4a9a442404"},
|
||||||
|
{"V4FMADDPS unmasked", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||||
|
"62f27f489a842411000000"},
|
||||||
|
{"V4FMADDSS 7(AX) [X0-X3] K5 X22", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e27f0d9bb007000000"},
|
||||||
|
{"V4FMADDSS (DI)", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(DI, 0, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e27f0d9b37"},
|
||||||
|
{"V4FMADDSS [X10-X13]", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X10"), vreg(t, "X13")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e22f0d9bb007000000"},
|
||||||
|
{"V4FMADDSS [X20-X23]", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e25f059bb007000000"},
|
||||||
|
{"V4FMADDSS X30 dst", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||||
|
"62627f0d9bb007000000"},
|
||||||
|
{"V4FMADDSS X3 dst", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X3")},
|
||||||
|
"62f27f0d9b9807000000"},
|
||||||
|
{"V4FMADDSS disp8x16", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 16, 8), RegList{vreg(t, "X20"), vreg(t, "X23")}, vreg(t, "K5"), vreg(t, "X30")},
|
||||||
|
"62625f059b7001"},
|
||||||
|
{"V4FNMADDPS", "V4FNMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4aaa842411000000"},
|
||||||
|
{"V4FNMADDSS", "V4FNMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 7, 8), RegList{vreg(t, "X0"), vreg(t, "X3")}, vreg(t, "K5"), vreg(t, "X22")},
|
||||||
|
"62e27f0dabb007000000"},
|
||||||
|
{"VP4DPWSSD", "VP4DPWSSD",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "K2"), vreg(t, "Z0")},
|
||||||
|
"62f27f4a52842411000000"},
|
||||||
|
{"VP4DPWSSDS unmasked", "VP4DPWSSDS",
|
||||||
|
[]Operand{Ptr(sp, 17, 8), RegList{vreg(t, "Z0"), vreg(t, "Z3")}, vreg(t, "Z0")},
|
||||||
|
"62f27f4853842411000000"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
code, err := Encode(c.mnem, c.ops...)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%s: Encode: %v", c.name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got := hexCompact(code); got != c.want {
|
||||||
|
t.Errorf("%s: got %s, want %s", c.name, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvexQuadRegisterErrors pins the operand shapes the toolchain rejects:
|
||||||
|
// the register class the list and the destination take is fixed per
|
||||||
|
// instruction, the source is memory only, the opmask slot is positional and
|
||||||
|
// the list's low register owns V'VVVV.
|
||||||
|
func TestEvexQuadRegisterErrors(t *testing.T) {
|
||||||
|
sp := vreg(t, "RSP")
|
||||||
|
list := func(lo, hi string) RegList {
|
||||||
|
return RegList{vreg(t, lo), vreg(t, hi)}
|
||||||
|
}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
mnem string
|
||||||
|
ops []Operand
|
||||||
|
}{
|
||||||
|
{"X list on the PS form", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("X0", "X3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"Z list on the SS form", "V4FMADDSS",
|
||||||
|
[]Operand{Ptr(AX, 0, 8), list("Z0", "Z3"), vreg(t, "K5"), vreg(t, "X22")}},
|
||||||
|
{"Y destination", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Y0")}},
|
||||||
|
{"register source", "V4FMADDPS",
|
||||||
|
[]Operand{vreg(t, "Z1"), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"non-mask third operand", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z4"), vreg(t, "Z0")}},
|
||||||
|
{"k0 mask", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K0"), vreg(t, "Z0")}},
|
||||||
|
{"K after the destination", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0"), vreg(t, "K2")}},
|
||||||
|
{"zeroing without a mask", "V4FMADDPS.Z",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "Z0")}},
|
||||||
|
{"SAE suffix", "V4FMADDPS.SAE",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"high index source", "VP4DPWSSD",
|
||||||
|
[]Operand{Idx(DI, vreg(t, "X16"), 1, 0, 8), list("Z0", "Z3"), vreg(t, "K2"), vreg(t, "Z0")}},
|
||||||
|
{"short operand list", "V4FMADDPS",
|
||||||
|
[]Operand{Ptr(sp, 0, 8), list("Z0", "Z3")}},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
if _, err := Encode(c.mnem, c.ops...); err == nil {
|
||||||
|
t.Errorf("%s: expected an error, got none", c.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -463,6 +463,61 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
|
|||||||
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
symRelocs[si] = append(symRelocs[si], rec[:]...)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The data symbols' own relocations: the symbol-valued DATA fields
|
||||||
|
// ("DATA s+0(SB)/8, $other(SB)"). The toolchain patches each field
|
||||||
|
// with the target's absolute address through an R_ADDR of the DATA
|
||||||
|
// line's width, on every architecture (the code relocations are
|
||||||
|
// per-architecture PC-relative shapes; a data pointer word is not), so
|
||||||
|
// this mapping bypasses relocField. The definitions were appended in
|
||||||
|
// DataSyms order, so data symbol i is definition index i.
|
||||||
|
for i, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
if r.Kind != RelAddr {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: data symbol %q carries a non-data relocation", d.Name)
|
||||||
|
}
|
||||||
|
var rec [23]byte
|
||||||
|
binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off)))
|
||||||
|
rec[4] = r.Siz
|
||||||
|
binary.LittleEndian.PutUint16(rec[5:], relocAddr)
|
||||||
|
binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend))
|
||||||
|
switch {
|
||||||
|
case r.External && r.Name == goobjBuiltinMorestack:
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxBuiltin)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], goobjBuiltinMorestackNoctxt)
|
||||||
|
case r.External:
|
||||||
|
pkg, name := splitQualified(r.Name)
|
||||||
|
if pkg == "" {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: external symbol %q has no package prefix", r.Name)
|
||||||
|
}
|
||||||
|
pIdx, ok := extPkgIdx[pkg]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: package %q not resolved", pkg)
|
||||||
|
}
|
||||||
|
sIdx, ok := extSymIdx[pkg+"·"+name]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: symbol %s·%s not resolved", pkg, name)
|
||||||
|
}
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx))
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx))
|
||||||
|
default:
|
||||||
|
if di, ok := defIdx[r.Name]; ok {
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(di))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
// A DATA field may hold the address of a TEXT function of
|
||||||
|
// the same file (the rt0 lib entry spelling), which is a
|
||||||
|
// non-package definition.
|
||||||
|
ni, isText := textNpIdx[r.Name]
|
||||||
|
if !isText {
|
||||||
|
return nil, fmt.Errorf("GOOBJ emission: reference to unknown symbol %q", r.Name)
|
||||||
|
}
|
||||||
|
binary.LittleEndian.PutUint32(rec[15:], pkgIdxNone)
|
||||||
|
binary.LittleEndian.PutUint32(rec[19:], uint32(ni))
|
||||||
|
}
|
||||||
|
symRelocs[i] = append(symRelocs[i], rec[:]...)
|
||||||
|
}
|
||||||
|
}
|
||||||
// The DWARF symbols' own relocations (the function address references).
|
// The DWARF symbols' own relocations (the function address references).
|
||||||
for _, ds := range dwarfRelocs {
|
for _, ds := range dwarfRelocs {
|
||||||
for _, r := range ds.relocs {
|
for _, r := range ds.relocs {
|
||||||
|
|||||||
+1
-1
@@ -12,7 +12,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
|
||||||
|
|||||||
+6
-6
@@ -9,8 +9,8 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
|
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
|
||||||
@@ -307,12 +307,12 @@ func TestStackGuardBytesLOONG64(t *testing.T) {
|
|||||||
func TestStackGuardGOObjInternalCall(t *testing.T) {
|
func TestStackGuardGOObjInternalCall(t *testing.T) {
|
||||||
for _, tt := range []struct {
|
for _, tt := range []struct {
|
||||||
src string
|
src string
|
||||||
assemble func(*ast.File) (*Image, error)
|
assemble func(*ast.File, ...AssembleOption) (*Image, error)
|
||||||
}{
|
}{
|
||||||
{"g_amd64.s", AssembleFile},
|
{"g_amd64.s", AssembleFile},
|
||||||
{"g_arm64.s", AssembleFileARM64},
|
{"g_arm64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileARM64(f) }},
|
||||||
{"g_riscv64.s", AssembleFileRISCV},
|
{"g_riscv64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileRISCV(f) }},
|
||||||
{"g_loong64.s", AssembleFileLOONG64},
|
{"g_loong64.s", func(f *ast.File, _ ...AssembleOption) (*Image, error) { return AssembleFileLOONG64(f) }},
|
||||||
} {
|
} {
|
||||||
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
f, errs := parser.Parse(tt.src, "TEXT \u00b7callsmall(SB), $16-0\n\tCALL \u00b7other(SB)\n\tRET\nTEXT \u00b7other(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
|
|||||||
+780
-35
@@ -14,30 +14,82 @@ var aluOp = map[string]struct {
|
|||||||
}{
|
}{
|
||||||
"ADD": {0x01, 0},
|
"ADD": {0x01, 0},
|
||||||
"OR": {0x09, 1},
|
"OR": {0x09, 1},
|
||||||
|
"ADC": {0x11, 2},
|
||||||
|
"SBB": {0x19, 3},
|
||||||
"AND": {0x21, 4},
|
"AND": {0x21, 4},
|
||||||
"SUB": {0x29, 5},
|
"SUB": {0x29, 5},
|
||||||
"XOR": {0x31, 6},
|
"XOR": {0x31, 6},
|
||||||
"CMP": {0x39, 7},
|
"CMP": {0x39, 7},
|
||||||
}
|
}
|
||||||
|
|
||||||
// unaryOp maps INC/DEC/NEG/NOT to their /digit and base opcode. INC/DEC use
|
// unaryOp maps INC/DEC/NEG/NOT/MUL/DIV/IDIV to their /digit and base opcode.
|
||||||
// the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes in 64-bit
|
// INC/DEC use the 0xFE/0xFF group (the short 0x40-0x4F forms are REX prefixes
|
||||||
// mode); NEG/NOT use the 0xF6/0xF7 group.
|
// in 64-bit mode); NEG/NOT/MUL/DIV/IDIV use the 0xF6/0xF7 group (MUL /4,
|
||||||
|
// DIV /6, IDIV /7; the accumulator is the implicit other operand).
|
||||||
var unaryOp = map[string]struct {
|
var unaryOp = map[string]struct {
|
||||||
digit int
|
digit int
|
||||||
op byte
|
op byte
|
||||||
}{
|
}{
|
||||||
"INC": {0, 0xFF},
|
"INC": {0, 0xFF},
|
||||||
"DEC": {1, 0xFF},
|
"DEC": {1, 0xFF},
|
||||||
"NOT": {2, 0xF7},
|
"NOT": {2, 0xF7},
|
||||||
"NEG": {3, 0xF7},
|
"NEG": {3, 0xF7},
|
||||||
|
"MUL": {4, 0xF7},
|
||||||
|
"DIV": {6, 0xF7},
|
||||||
|
"IDIV": {7, 0xF7},
|
||||||
}
|
}
|
||||||
|
|
||||||
// shiftOp maps SHL/SHR/SAR to their /digit in the 0xC0/0xC1/0xD0-0xD3 group.
|
// shiftOp maps SHL/SAL/SHR/SAR/ROL/ROR/RCL/RCR to their /digit in the
|
||||||
|
// 0xC0/0xC1/0xD0-0xD3 group. SAL is the same encoding as SHL (/4).
|
||||||
var shiftOp = map[string]int{
|
var shiftOp = map[string]int{
|
||||||
"SHL": 4,
|
"SHL": 4,
|
||||||
|
"SAL": 4,
|
||||||
"SHR": 5,
|
"SHR": 5,
|
||||||
"SAR": 7,
|
"SAR": 7,
|
||||||
|
"ROL": 0,
|
||||||
|
"ROR": 1,
|
||||||
|
"RCL": 2,
|
||||||
|
"RCR": 3,
|
||||||
|
}
|
||||||
|
|
||||||
|
// bitTestOp maps BT/BTS/BTR/BTC to their /digit in the 0F BA immediate form;
|
||||||
|
// the register form is 0F A3/AB/B3/BB, the same digit in the low nibble's
|
||||||
|
// opcode row.
|
||||||
|
var bitTestOp = map[string]int{
|
||||||
|
"BT": 4,
|
||||||
|
"BTS": 5,
|
||||||
|
"BTR": 6,
|
||||||
|
"BTC": 7,
|
||||||
|
}
|
||||||
|
|
||||||
|
// noOperandTable maps a fixed no-operand mnemonic to its opcode bytes. The
|
||||||
|
// fence names carry their opcode inside the 0F AE /digit group spelled out in
|
||||||
|
// full (E8/F0/F8), and PAUSE is F3 90.
|
||||||
|
//
|
||||||
|
// LOCK, REP and REPN are the prefix statements. go tool asm encodes each as
|
||||||
|
// a standalone one-byte instruction with a PC of its own (F0, F3 and F2
|
||||||
|
// respectively), not as a prefix field merged into the next instruction: the
|
||||||
|
// statement that follows is encoded unaware of it, and nothing validates
|
||||||
|
// that the pairing is a legal one (LOCK before NOP assembles without
|
||||||
|
// complaint, each byte pinned against the toolchain). Because the bytes
|
||||||
|
// land in the stream before the following statement anyway, a LOCKed
|
||||||
|
// CMPXCHGQ encodes identically to a prefixed form.
|
||||||
|
var noOperandTable = map[string][]byte{
|
||||||
|
"CPUID": {0x0F, 0xA2},
|
||||||
|
"RDTSC": {0x0F, 0x31},
|
||||||
|
"RDTSCP": {0x0F, 0x01, 0xF9},
|
||||||
|
"SYSCALL": {0x0F, 0x05},
|
||||||
|
"XGETBV": {0x0F, 0x01, 0xD0},
|
||||||
|
"CLD": {0xFC},
|
||||||
|
"STD": {0xFD},
|
||||||
|
"PAUSE": {0xF3, 0x90},
|
||||||
|
"LFENCE": {0x0F, 0xAE, 0xE8},
|
||||||
|
"MFENCE": {0x0F, 0xAE, 0xF0},
|
||||||
|
"SFENCE": {0x0F, 0xAE, 0xF8},
|
||||||
|
"UNDEF": {0x0F, 0x0B},
|
||||||
|
"LOCK": {0xF0},
|
||||||
|
"REP": {0xF3},
|
||||||
|
"REPN": {0xF2},
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- MOV --------------------------------------------------------------------
|
// --- MOV --------------------------------------------------------------------
|
||||||
@@ -140,6 +192,30 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
|||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
|
|
||||||
|
case TLSMem:
|
||||||
|
if !dstIsReg {
|
||||||
|
return fmt.Errorf("MOV: two memory operands")
|
||||||
|
}
|
||||||
|
// MOV r, off(TLS): the segment-prefixed absolute load, reg=dst,
|
||||||
|
// rm=src(tlsMem) through the SIB escape; the disp32 is the TLS slot
|
||||||
|
// offset with its R_TLSLE patch site.
|
||||||
|
i := newInstr(size, []byte{movRR(size)})
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
|
||||||
|
case SegAbs:
|
||||||
|
if !dstIsReg {
|
||||||
|
return fmt.Errorf("MOV: two memory operands")
|
||||||
|
}
|
||||||
|
// MOV r, 0x30(GS): the segment-absolute load.
|
||||||
|
i := newInstr(size, []byte{movRR(size)})
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
|
||||||
case Imm:
|
case Imm:
|
||||||
if dstIsReg {
|
if dstIsReg {
|
||||||
v := int64(src)
|
v := int64(src)
|
||||||
@@ -180,11 +256,24 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
|
|||||||
i.imm = imm
|
i.imm = imm
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0.
|
// MOV r/m, imm: 0xC6 (8-bit) / 0xC7 /0. An immediate in the
|
||||||
|
// destination slot is the absolute-address crash-store spelling,
|
||||||
|
// MOVL $0xf1, 0xf1: the parser reads the trailing bare constant
|
||||||
|
// as an immediate, and the store's disp32 carries the address.
|
||||||
op := byte(0xC7)
|
op := byte(0xC7)
|
||||||
if size == 1 {
|
if size == 1 {
|
||||||
op = 0xC6
|
op = 0xC6
|
||||||
}
|
}
|
||||||
|
if d, ok := dst.(Imm); ok {
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
setSegAbs(i, 0, SegAbs{Disp: int64(d)})
|
||||||
|
immBytes, err := immediate(int64(src), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
i := newInstr(size, []byte{op})
|
i := newInstr(size, []byte{op})
|
||||||
if err := setRMDigit(i, 0, dst, size); err != nil {
|
if err := setRMDigit(i, 0, dst, size); err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -309,6 +398,13 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
|
||||||
|
// the destination is AL, the form the Go assembler prefers here.
|
||||||
|
if r, ok := dst.(Reg); ok && r.idx == 0 {
|
||||||
|
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
|
||||||
|
i.imm = immBytes
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
i := newInstr(1, []byte{0x80})
|
i := newInstr(1, []byte{0x80})
|
||||||
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
if err := setRMDigit(i, digit, dst, 1); err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -451,13 +547,34 @@ func (e *enc) encodeUnary(op struct {
|
|||||||
|
|
||||||
// --- SHL/SHR/SAR ------------------------------------------------------------
|
// --- SHL/SHR/SAR ------------------------------------------------------------
|
||||||
|
|
||||||
func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
// doubleShiftOp maps the two mnemonics whose three-operand form go tool asm
|
||||||
|
// accepts to the SHLD/SHRD opcode pair (imm8 form, CL form). SAR, SAL and
|
||||||
|
// the rotates have no such form: the oracle rejects SARQ/ROLQ with three
|
||||||
|
// operands, and so do we.
|
||||||
|
var doubleShiftOp = map[string][2]byte{
|
||||||
|
"SHL": {0xA4, 0xA5}, // SHLD
|
||||||
|
"SHR": {0xAC, 0xAD}, // SHRD
|
||||||
|
}
|
||||||
|
|
||||||
|
// isShiftCountCL reports whether a count operand is the CL register or its
|
||||||
|
// CX spelling: go tool asm accepts both (CX names the same low byte) and
|
||||||
|
// rejects ECX/RCX.
|
||||||
|
func isShiftCountCL(o Operand) bool {
|
||||||
|
reg, ok := o.(Reg)
|
||||||
|
return ok && reg.idx == 1 && (reg.size == 1 || reg.size == 2)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (e *enc) encodeShift(base string, ops []Operand, size int) error {
|
||||||
|
digit := shiftOp[base]
|
||||||
|
if len(ops) == 3 {
|
||||||
|
return e.encodeDoubleShift(base, ops, size)
|
||||||
|
}
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
|
return fmt.Errorf("shift expects 2 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
count, dst := ops[0], ops[1]
|
count, dst := ops[0], ops[1]
|
||||||
// Count is $1, %CL, or an imm8.
|
// Count is $1, CL (or its CX spelling), or an imm8.
|
||||||
if reg, ok := count.(Reg); ok && reg.idx == 1 && reg.size <= 1 {
|
if isShiftCountCL(count) {
|
||||||
// CL: 0xD2 (8-bit) / 0xD3.
|
// CL: 0xD2 (8-bit) / 0xD3.
|
||||||
op := byte(0xD3)
|
op := byte(0xD3)
|
||||||
if size == 1 {
|
if size == 1 {
|
||||||
@@ -504,12 +621,59 @@ func (e *enc) encodeShift(digit int, ops []Operand, size int) error {
|
|||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeDoubleShift emits the three-operand SHL/SHR form, which the Go
|
||||||
|
// assembler spells as a shift but encodes as SHLD/SHRD (0F A4/A5, 0F AC/AD):
|
||||||
|
// the first operand is the count ($imm or CL), the second feeds the vacated
|
||||||
|
// bits (the reg field) and the third is the shifted value (the r/m field),
|
||||||
|
// matching go tool asm byte for byte. The W/L/Q widths exist; the oracle
|
||||||
|
// rejects the three-operand B form and every SAR/rotate one.
|
||||||
|
func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
|
||||||
|
opc, ok := doubleShiftOp[base]
|
||||||
|
if !ok || size == 1 {
|
||||||
|
return fmt.Errorf("%s: shift expects 2 operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
count, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
srcReg, ok := src.(Reg)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s: middle operand must be a register, like go tool asm", base)
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, opc[0]})
|
||||||
|
if isShiftCountCL(count) {
|
||||||
|
// CL (or CX) form: 0F A5/AD.
|
||||||
|
i.opcode[1] = opc[1]
|
||||||
|
} else {
|
||||||
|
imm, ok := count.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("shift count must be $1, CL or an immediate")
|
||||||
|
}
|
||||||
|
// The count is an unsigned imm8: the same range convention as the
|
||||||
|
// two-operand shift above.
|
||||||
|
if imm < 0 || imm > 255 {
|
||||||
|
return fmt.Errorf("shift count $%d is out of the 0..255 range", int64(imm))
|
||||||
|
}
|
||||||
|
i.imm = []byte{byte(imm)}
|
||||||
|
}
|
||||||
|
if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
// --- IMUL -------------------------------------------------------------------
|
// --- IMUL -------------------------------------------------------------------
|
||||||
|
|
||||||
func (e *enc) encodeImul(ops []Operand, size int) error {
|
func (e *enc) encodeImul(ops []Operand, size int) error {
|
||||||
switch len(ops) {
|
switch len(ops) {
|
||||||
case 2:
|
case 2:
|
||||||
// IMUL r, r/m: 0x0F 0xAF.
|
// Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
|
||||||
|
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
|
||||||
|
// Otherwise IMUL r, r/m: 0x0F 0xAF.
|
||||||
|
if imm, ok := ops[0].(Imm); ok {
|
||||||
|
dstReg, isReg := ops[1].(Reg)
|
||||||
|
if !isReg {
|
||||||
|
return fmt.Errorf("IMUL: destination must be a register")
|
||||||
|
}
|
||||||
|
return e.encodeImulImm(imm, dstReg, dstReg, size)
|
||||||
|
}
|
||||||
dstReg, ok := ops[1].(Reg)
|
dstReg, ok := ops[1].(Reg)
|
||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("IMUL: destination must be a register")
|
return fmt.Errorf("IMUL: destination must be a register")
|
||||||
@@ -529,29 +693,36 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("IMUL: immediate operand expected first")
|
return fmt.Errorf("IMUL: immediate operand expected first")
|
||||||
}
|
}
|
||||||
// Plan 9 order: IMUL $imm, src, dst.
|
// Plan 9 order: IMUL $imm, src, dst; the source stays a general
|
||||||
if fits8(int64(imm)) {
|
// r/m operand (setRM takes registers and memory alike).
|
||||||
i := newInstr(size, []byte{0x6B})
|
return e.encodeImulImm(imm, ops[1], dstReg, size)
|
||||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
i.imm = []byte{byte(int8(imm))}
|
|
||||||
return e.emit(i)
|
|
||||||
}
|
|
||||||
i := newInstr(size, []byte{0x69})
|
|
||||||
if err := setRM(i, dstReg, ops[1], size); err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
immBytes, err := immediate(int64(imm), size, false)
|
|
||||||
if err != nil {
|
|
||||||
return err
|
|
||||||
}
|
|
||||||
i.imm = immBytes
|
|
||||||
return e.emit(i)
|
|
||||||
}
|
}
|
||||||
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
|
||||||
|
// when the value fits, 0x69 with a 32-bit immediate otherwise.
|
||||||
|
func (e *enc) encodeImulImm(imm Imm, rm Operand, dst Reg, size int) error {
|
||||||
|
if fits8(int64(imm)) {
|
||||||
|
i := newInstr(size, []byte{0x6B})
|
||||||
|
if err := setRM(i, dst, rm, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{byte(int8(imm))}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x69})
|
||||||
|
if err := setRM(i, dst, rm, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
immBytes, err := immediate(int64(imm), size, false)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = immBytes
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
// --- PUSH / POP -------------------------------------------------------------
|
// --- PUSH / POP -------------------------------------------------------------
|
||||||
|
|
||||||
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
|
||||||
@@ -913,6 +1084,7 @@ type sseMove struct {
|
|||||||
var sseMoveTable = map[string]sseMove{
|
var sseMoveTable = map[string]sseMove{
|
||||||
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
"MOVOU": {0xF3, 0x6F, 0x7F}, // MOVDQU, unaligned octa
|
||||||
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
"MOVO": {0x66, 0x6F, 0x7F}, // MOVDQA, aligned octa
|
||||||
|
"MOVOA": {0x66, 0x6F, 0x7F}, // MOVDQA, the aligned octa alias
|
||||||
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
"MOVUPS": {0x00, 0x10, 0x11}, // unaligned packed single
|
||||||
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
"MOVAPS": {0x00, 0x28, 0x29}, // aligned packed single
|
||||||
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
"MOVUPD": {0x66, 0x10, 0x11}, // unaligned packed double
|
||||||
@@ -938,12 +1110,12 @@ func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
|
|||||||
op = m.load
|
op = m.load
|
||||||
reg, rm = dstReg, src
|
reg, rm = dstReg, src
|
||||||
case srcVec:
|
case srcVec:
|
||||||
if _, ok := dst.(Mem); !ok {
|
if !isX86Mem(dst) {
|
||||||
return fmt.Errorf("SSE move: invalid destination operand")
|
return fmt.Errorf("SSE move: invalid destination operand")
|
||||||
}
|
}
|
||||||
reg, rm = srcReg, dst
|
reg, rm = srcReg, dst
|
||||||
case dstVec:
|
case dstVec:
|
||||||
if _, ok := src.(Mem); !ok {
|
if !isX86Mem(src) {
|
||||||
return fmt.Errorf("SSE move: invalid source operand")
|
return fmt.Errorf("SSE move: invalid source operand")
|
||||||
}
|
}
|
||||||
op = m.load
|
op = m.load
|
||||||
@@ -1000,10 +1172,120 @@ var sseBinTable = map[string]sseBin{
|
|||||||
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
"PSUBB": {0x66, 0xF8, false}, "PSUBW": {0x66, 0xF9, false},
|
||||||
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
"PSUBD": {0x66, 0xFA, false}, "PSUBQ": {0x66, 0xFB, false},
|
||||||
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
"PCMPEQB": {0x66, 0x74, false}, "PCMPEQW": {0x66, 0x75, false},
|
||||||
"PCMPEQD": {0x66, 0x76, false},
|
"PCMPEQD": {0x66, 0x76, false}, "PCMPEQL": {0x66, 0x76, false},
|
||||||
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
"PCMPGTB": {0x66, 0x64, false}, "PCMPGTW": {0x66, 0x65, false},
|
||||||
"PCMPGTD": {0x66, 0x66, false},
|
"PCMPGTD": {0x66, 0x66, false},
|
||||||
"PSHUFB": {0x66, 0x00, true},
|
"PSHUFB": {0x66, 0x00, true},
|
||||||
|
// Scalar compares and square root, packed adds/subtracts and the byte
|
||||||
|
// unpack, the spellings the Plan 9 table uses (COMISD orders the
|
||||||
|
// operands like every other two-operand form).
|
||||||
|
"ANDNPD": {0x66, 0x55, false},
|
||||||
|
"ANDNPS": {0x00, 0x55, false},
|
||||||
|
"COMISD": {0x66, 0x2F, false},
|
||||||
|
"SQRTSD": {0xF2, 0x51, false},
|
||||||
|
"PADDL": {0x66, 0xFE, false},
|
||||||
|
"PSUBL": {0x66, 0xFA, false},
|
||||||
|
"PUNPCKLBW": {0x66, 0x60, false},
|
||||||
|
// AES round functions (66 0F38) and the SHA message schedule helpers
|
||||||
|
// (no prefix, 0F38).
|
||||||
|
"AESENC": {0x66, 0xDC, true},
|
||||||
|
"AESENCLAST": {0x66, 0xDD, true},
|
||||||
|
"AESDEC": {0x66, 0xDE, true},
|
||||||
|
"AESDECLAST": {0x66, 0xDF, true},
|
||||||
|
"AESIMC": {0x66, 0xDB, true},
|
||||||
|
"SHA1MSG1": {0x00, 0xC9, true},
|
||||||
|
"SHA1MSG2": {0x00, 0xCA, true},
|
||||||
|
"SHA1NEXTE": {0x00, 0xC8, true},
|
||||||
|
"SHA256MSG1": {0x00, 0xCC, true},
|
||||||
|
"SHA256MSG2": {0x00, 0xCD, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseImm3 describes a legacy SSE instruction taking a leading imm8 and two
|
||||||
|
// further operands: OP $imm, src, dst with reg = dst, rm = src. map38 and
|
||||||
|
// map3A select the opcode map the same way as sseBin's.
|
||||||
|
type sseImm3 struct {
|
||||||
|
prefix byte
|
||||||
|
op byte
|
||||||
|
map3A bool // opcode lives under 0F3A instead of 0F38
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseImm3Table covers the imm8-controlled legacy instructions: the SSSE3
|
||||||
|
// align/blend shuffles, the string compare, carry-less multiply and the AES
|
||||||
|
// key assistant. SHA1RNDS4 carries no prefix, unlike its 0F3A siblings.
|
||||||
|
var sseImm3Table = map[string]sseImm3{
|
||||||
|
"PALIGNR": {0x66, 0x0F, true},
|
||||||
|
"PBLENDW": {0x66, 0x0E, true},
|
||||||
|
"PCMPESTRI": {0x66, 0x61, true},
|
||||||
|
"PCLMULQDQ": {0x66, 0x44, true},
|
||||||
|
"AESKEYGENASSIST": {0x66, 0xDF, true},
|
||||||
|
"SHA1RNDS4": {0x00, 0xCC, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseExtract describes a lane extract: OP $imm, xsrc, dst with reg = the XMM
|
||||||
|
// source and rm = the destination (GPR or memory). PEXTRW's GPR destination
|
||||||
|
// uses the older 0F C5 form; its memory destination the SSE4.1 0F3A 15 one,
|
||||||
|
// so it carries both opcodes.
|
||||||
|
type sseExtract struct {
|
||||||
|
op []byte
|
||||||
|
opMem []byte // used when the destination is memory; nil shares op
|
||||||
|
rexW bool // PEXTRQ's REX.W
|
||||||
|
}
|
||||||
|
|
||||||
|
var sseExtractTable = map[string]sseExtract{
|
||||||
|
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
|
||||||
|
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
|
||||||
|
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
|
||||||
|
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
|
||||||
|
// destination and rm = the source (GPR or memory).
|
||||||
|
type sseInsert struct {
|
||||||
|
op []byte
|
||||||
|
rexW bool // PINSRQ's REX.W
|
||||||
|
}
|
||||||
|
|
||||||
|
var sseInsertTable = map[string]sseInsert{
|
||||||
|
"PINSRB": {[]byte{0x0F, 0x3A, 0x20}, false},
|
||||||
|
"PINSRD": {[]byte{0x0F, 0x3A, 0x22}, false},
|
||||||
|
"PINSRQ": {[]byte{0x0F, 0x3A, 0x22}, true},
|
||||||
|
"PINSRW": {[]byte{0x0F, 0xC4}, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShiftImm maps the legacy packed integer shifts' immediate form:
|
||||||
|
// OP $imm, dst (66 0F 71/72/73 /digit). The Plan 9 dword spellings end in L
|
||||||
|
// (PSLLL/PSRAL/PSRLL) and the octa byte shifts are PSLLDQ/PSRLDQ.
|
||||||
|
var sseShiftImm = map[string]sseShift{
|
||||||
|
"PSLLW": {0x71, 6},
|
||||||
|
"PSRLW": {0x71, 2},
|
||||||
|
"PSRAW": {0x71, 4},
|
||||||
|
"PSLLL": {0x72, 6},
|
||||||
|
"PSRLL": {0x72, 2},
|
||||||
|
"PSRAL": {0x72, 4},
|
||||||
|
"PSLLQ": {0x73, 6},
|
||||||
|
"PSRLQ": {0x73, 2},
|
||||||
|
"PSLLDQ": {0x73, 7},
|
||||||
|
"PSRLDQ": {0x73, 3},
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShiftVar maps the variable-count forms (the count comes from an XMM
|
||||||
|
// register or memory): OP count, dst (66 0F D1-F3). PSLLDQ/PSRLDQ have no
|
||||||
|
// variable form.
|
||||||
|
var sseShiftVar = map[string]byte{
|
||||||
|
"PSLLW": 0xF1,
|
||||||
|
"PSRLW": 0xD1,
|
||||||
|
"PSRAW": 0xE1,
|
||||||
|
"PSLLL": 0xF2,
|
||||||
|
"PSRLL": 0xD2,
|
||||||
|
"PSRAL": 0xE2,
|
||||||
|
"PSLLQ": 0xF3,
|
||||||
|
"PSRLQ": 0xD3,
|
||||||
|
}
|
||||||
|
|
||||||
|
// sseShift is one /digit selector in the 0F 71/72/73 immediate group.
|
||||||
|
type sseShift struct {
|
||||||
|
op byte
|
||||||
|
digit int
|
||||||
}
|
}
|
||||||
|
|
||||||
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
// sseShuf describes a legacy SSE shuffle taking a trailing imm8
|
||||||
@@ -1016,6 +1298,7 @@ type sseShuf struct {
|
|||||||
var sseShufTable = map[string]sseShuf{
|
var sseShufTable = map[string]sseShuf{
|
||||||
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
"SHUFPS": {0, 0xC6}, "SHUFPD": {0x66, 0xC6},
|
||||||
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
"PSHUFD": {0x66, 0x70}, "PSHUFHW": {0xF3, 0x70}, "PSHUFLW": {0xF2, 0x70},
|
||||||
|
"PSHUFL": {0x66, 0x70},
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
// encodeSSEBin encodes reg = reg op rm (memory allowed for rm).
|
||||||
@@ -1090,3 +1373,465 @@ func (e *enc) encodeCvtsi2sd(quad bool, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
return e.emit(i)
|
return e.emit(i)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- carry, bit test, exchange and accumulate -------------------------------
|
||||||
|
|
||||||
|
// encodeBitTest encodes BT/BTS/BTR/BTC. The bit index goes first in Plan 9
|
||||||
|
// order (BTQ AX, BX tests BX at the offset in AX, encoding 0F A3 with
|
||||||
|
// reg = index, rm = target); an immediate index uses 0F BA /digit with imm8.
|
||||||
|
func (e *enc) encodeBitTest(name string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
digit := bitTestOp[name]
|
||||||
|
index, target := ops[0], ops[1]
|
||||||
|
if reg, ok := index.(Reg); ok {
|
||||||
|
// Register index: 0F A3 (BT) / 0F AB (BTS) / 0F B3 (BTR) / 0F BB (BTC),
|
||||||
|
// the /digit base plus eight per step.
|
||||||
|
i := newInstr(size, []byte{0x0F, 0xA3 + byte(digit-4)<<3})
|
||||||
|
if err := setRM(i, reg, target, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
imm, ok := index.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s index must be a register or an immediate", name)
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, 0xBA})
|
||||||
|
if err := setRMDigit(i, digit, target, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeExchange encodes XCHG. A register-to-register exchange where either
|
||||||
|
// operand is AX uses the 0x90+r accumulator form (with REX.W for the quad
|
||||||
|
// form, as the Go assembler emits it); everything else uses 0x86/0x87 with
|
||||||
|
// the register operand in ModRM.reg, the memory (or second register) in r/m.
|
||||||
|
func (e *enc) encodeExchange(ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("XCHG expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
srcReg, srcIsReg := src.(Reg)
|
||||||
|
dstReg, dstIsReg := dst.(Reg)
|
||||||
|
if srcIsReg && dstIsReg && size > 1 && (srcReg.idx == 0 || dstReg.idx == 0) {
|
||||||
|
// 0x90+r: r is the non-AX register, whichever side it sits on.
|
||||||
|
r := dstReg
|
||||||
|
if srcReg.idx == 0 {
|
||||||
|
r = dstReg
|
||||||
|
} else {
|
||||||
|
r = srcReg
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x90 + byte(r.idx&7)})
|
||||||
|
i.rexB = r.idx >= 8
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
op := byte(0x87)
|
||||||
|
if size == 1 {
|
||||||
|
op = 0x86
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case srcIsReg:
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
if err := setRM(i, srcReg, dst, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case dstIsReg:
|
||||||
|
i := newInstr(size, []byte{op})
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("XCHG: at least one operand must be a register")
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeRegRegOp encodes the two-operand read-modify-write pair CMPXCHG
|
||||||
|
// (0F B0/B1) and XADD (0F C0/C1): reg = source, rm = destination, with the
|
||||||
|
// destination writable (register or memory).
|
||||||
|
func (e *enc) encodeRegRegOp(op8, op byte, name string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s source must be a register", name)
|
||||||
|
}
|
||||||
|
opc := op
|
||||||
|
if size == 1 {
|
||||||
|
opc = op8
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, opc})
|
||||||
|
if err := setRM(i, srcReg, ops[1], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCrc32 encodes the CRC32 family: F2 0F38 F0 for the byte form, F1 for
|
||||||
|
// the rest; the word form carries a 0x66 operand-size prefix (66 F2, the
|
||||||
|
// prefix order the Go assembler emits) and the quad form REX.W. reg = GPR
|
||||||
|
// accumulator, rm = the data source.
|
||||||
|
func (e *enc) encodeCrc32(ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("CRC32 expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("CRC32 destination must be a general register")
|
||||||
|
}
|
||||||
|
i := &instr{opSize16: size == 2, prefix: 0xF2, opcode: []byte{0x0F, 0x38, 0xF0}, modrm: -1, sib: -1}
|
||||||
|
if size > 1 {
|
||||||
|
i.opcode[2] = 0xF1
|
||||||
|
}
|
||||||
|
i.rexW = size == 8
|
||||||
|
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCarryExt encodes ADCX (66 0F38 F6) and ADOX (F3 0F38 F6): reg =
|
||||||
|
// destination, rm = source, the carry/overflow flag as the carry-in.
|
||||||
|
func (e *enc) encodeCarryExt(prefix byte, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("ADCX/ADOX expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("ADCX/ADOX destination must be a general register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0x38, 0xF6}, modrm: -1, sib: -1, rexW: size == 8}
|
||||||
|
if err := setRM(i, dstReg, ops[0], size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- string primitives, flags and INT ----------------------------------------
|
||||||
|
|
||||||
|
// encodeStringOp encodes the no-operand string primitives MOVS (A4/A5) and
|
||||||
|
// STOS (AA/AB); the size suffix picks the byte form and supplies the 0x66 or
|
||||||
|
// REX.W prefix.
|
||||||
|
func (e *enc) encodeStringOp(base string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s takes no operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
var op byte
|
||||||
|
switch base {
|
||||||
|
case "MOVS":
|
||||||
|
op = 0xA5
|
||||||
|
if size == 1 {
|
||||||
|
op = 0xA4
|
||||||
|
}
|
||||||
|
case "STOS":
|
||||||
|
op = 0xAB
|
||||||
|
if size == 1 {
|
||||||
|
op = 0xAA
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("unsupported string instruction %q", base)
|
||||||
|
}
|
||||||
|
return e.emit(newInstr(size, []byte{op}))
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeInt encodes INT with its single imm8 operand. The field takes the
|
||||||
|
// low byte silently inside the 32-bit span, matching the scalar convention
|
||||||
|
// (go tool asm encodes INT $256 as CD 00).
|
||||||
|
func (e *enc) encodeInt(ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("INT expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("INT operand must be an immediate")
|
||||||
|
}
|
||||||
|
if imm < -(1<<31) || imm > (1<<32)-1 {
|
||||||
|
return fmt.Errorf("immediate $%d does not fit in 32 bits", int64(imm))
|
||||||
|
}
|
||||||
|
return e.emit(&instr{opcode: []byte{0xCD}, modrm: -1, sib: -1, imm: []byte{byte(imm)}})
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeMxcsr encodes LDMXCSR (0F AE /2) and STMXCSR (0F AE /3); both take a
|
||||||
|
// single 32-bit memory operand.
|
||||||
|
func (e *enc) encodeMxcsr(digit int, ops []Operand) error {
|
||||||
|
if len(ops) != 1 {
|
||||||
|
return fmt.Errorf("MXCSR instruction expects 1 operand, got %d", len(ops))
|
||||||
|
}
|
||||||
|
m, ok := ops[0].(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("MXCSR instruction requires a memory operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0xAE}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, digit, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// cvtIntOp maps the scalar float-to-integer conversions to their mandatory
|
||||||
|
// prefix and opcode: 0F 2D (CVTSD2S, CVTSS2S) and 0F 2C (their truncating
|
||||||
|
// CVTT forms). The mnemonic's Q/L suffix fixes the GPR destination width.
|
||||||
|
var cvtIntOp = map[string]struct {
|
||||||
|
prefix byte
|
||||||
|
op byte
|
||||||
|
}{
|
||||||
|
"CVTSD2S": {0xF2, 0x2D},
|
||||||
|
"CVTTSD2S": {0xF2, 0x2C},
|
||||||
|
"CVTSS2S": {0xF3, 0x2D},
|
||||||
|
"CVTTSS2S": {0xF3, 0x2C},
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCvtInt encodes a scalar float-to-integer conversion: F2/F3 0F 2D/2C
|
||||||
|
// with reg = GPR destination, rm = XMM (or memory) source; REX.W follows the
|
||||||
|
// quad spellings.
|
||||||
|
func (e *enc) encodeCvtInt(base string, ops []Operand, size int) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", base, len(ops))
|
||||||
|
}
|
||||||
|
spec := cvtIntOp[base]
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s destination must be a general register", base)
|
||||||
|
}
|
||||||
|
i := newInstr(size, []byte{0x0F, spec.op})
|
||||||
|
i.prefix = spec.prefix
|
||||||
|
if err := setRM(i, dstReg, src, size); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeFmov encodes the x87 double move. The memory forms are DD /0
|
||||||
|
// (FMOVD mem, F: load) and DD /2 (FMOVD F, mem: store); a register-to-register
|
||||||
|
// move is DD C0+dst (FLD st(dst)), the form the Go assembler emits.
|
||||||
|
func (e *enc) encodeFmov(ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("FMOVD expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
srcReg, srcIsF := src.(Reg)
|
||||||
|
dstReg, dstIsF := dst.(Reg)
|
||||||
|
srcF := srcIsF && srcReg.fp
|
||||||
|
dstF := dstIsF && dstReg.fp
|
||||||
|
switch {
|
||||||
|
case srcF && dstF:
|
||||||
|
// The register form is DD /2 with rm = the destination (FST st(dst)).
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, 2, dstReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case dstF:
|
||||||
|
m, ok := src.(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("FMOVD: invalid source operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, 0, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
case srcF:
|
||||||
|
m, ok := dst.(Mem)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("FMOVD: invalid destination operand")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0xDD}, modrm: -1, sib: -1}
|
||||||
|
if err := setMem(i, 2, m); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
return fmt.Errorf("FMOVD needs an x87 register operand")
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- legacy SSE imm8, extract, insert and packed shift families --------------
|
||||||
|
|
||||||
|
// encodeSSEImm3 encodes an imm8-controlled three-operand form: OP $imm, src,
|
||||||
|
// dst with reg = dst, rm = src and the immediate appended last (PALIGNR,
|
||||||
|
// PBLENDW, PCMPESTRI, PCLMULQDQ, AESKEYGENASSIST, SHA1RNDS4).
|
||||||
|
func (e *enc) encodeSSEImm3(m sseImm3, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
src, dst := ops[1], ops[2]
|
||||||
|
dstReg, ok2 := dst.(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("SSE imm8 instruction destination must be a vector register")
|
||||||
|
}
|
||||||
|
opcode := []byte{0x0F, 0x38, m.op}
|
||||||
|
if m.map3A {
|
||||||
|
opcode = []byte{0x0F, 0x3A, m.op}
|
||||||
|
}
|
||||||
|
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, src, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEExtract encodes a lane extract: OP $imm, xsrc, dst with reg = the
|
||||||
|
// XMM source, rm = the GPR or memory destination (PEXTRB/PEXTRD/PEXTRQ and
|
||||||
|
// PEXTRW, whose GPR form is the older 0F C5 opcode and whose memory form the
|
||||||
|
// SSE4.1 0F3A 15 one).
|
||||||
|
func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("extract expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("extract needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
srcReg, srcVec := vecReg(ops[1])
|
||||||
|
if !srcVec {
|
||||||
|
return fmt.Errorf("extract source must be an XMM register")
|
||||||
|
}
|
||||||
|
opcode := m.op
|
||||||
|
if m.opMem != nil && memOperand(ops[2]) {
|
||||||
|
opcode = m.opMem
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
|
||||||
|
if err := setRM(i, srcReg, ops[2], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEInsert encodes a lane insert: OP $imm, src, xdst with reg = the
|
||||||
|
// XMM destination and rm = the GPR or memory source (PINSRB/PINSRD/PINSRQ and
|
||||||
|
// PINSRW).
|
||||||
|
func (e *enc) encodeSSEInsert(m sseInsert, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("insert expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[0].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("insert needs an immediate first operand")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dstReg, dstVec := vecReg(ops[2])
|
||||||
|
if !dstVec {
|
||||||
|
return fmt.Errorf("insert destination must be an XMM register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: m.op, modrm: -1, sib: -1, rexW: m.rexW}
|
||||||
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSSEShift encodes the legacy packed integer shifts. The immediate
|
||||||
|
// form is OP $imm, dst (66 0F 71/72/73 /digit); the variable form
|
||||||
|
// OP count, dst carries the count in an XMM register (or memory) on the
|
||||||
|
// 66 0F D1-F3 opcodes. The destination is always the register written.
|
||||||
|
func (e *enc) encodeSSEShift(name string, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("%s expects 2 operands, got %d", name, len(ops))
|
||||||
|
}
|
||||||
|
dstReg, ok := ops[1].(Reg)
|
||||||
|
if !ok || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("%s destination must be the second, vector operand", name)
|
||||||
|
}
|
||||||
|
if imm, isImm := ops[0].(Imm); isImm {
|
||||||
|
spec := sseShiftImm[name]
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
if !vecOrMem(ops[0]) {
|
||||||
|
return fmt.Errorf("%s count must be an immediate, a vector register or memory", name)
|
||||||
|
}
|
||||||
|
op, ok := sseShiftVar[name]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("%s has no variable-count form", name)
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0x66, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeCmpsd encodes CMPSD, the scalar double compare with its predicate
|
||||||
|
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
|
||||||
|
// F2 0F C2 with reg = dst, rm = src.
|
||||||
|
func (e *enc) encodeCmpsd(ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, ok := ops[2].(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("CMPSD predicate must be an immediate")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(imm))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
dstReg, ok2 := ops[1].(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("CMPSD destination must be a vector register")
|
||||||
|
}
|
||||||
|
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[0], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
i.imm = []byte{immByte}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeSha256rnds2 encodes SHA256RNDS2, whose first operand must be the
|
||||||
|
// literal X0 carrying the round constant: OP X0, src, dst (0F38 CB, no
|
||||||
|
// prefix, reg = dst, rm = src; X0 is implicit on the wire).
|
||||||
|
func (e *enc) encodeSha256rnds2(ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 expects 3 operands (X0, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
x0, ok := ops[0].(Reg)
|
||||||
|
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 first operand must be X0")
|
||||||
|
}
|
||||||
|
dstReg, ok2 := ops[2].(Reg)
|
||||||
|
if !ok2 || !dstReg.isVec() {
|
||||||
|
return fmt.Errorf("SHA256RNDS2 destination must be a vector register")
|
||||||
|
}
|
||||||
|
i := &instr{opcode: []byte{0x0F, 0x38, 0xCB}, modrm: -1, sib: -1}
|
||||||
|
if err := setRM(i, dstReg, ops[1], 8); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
return e.emit(i)
|
||||||
|
}
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ import (
|
|||||||
|
|
||||||
"golang.org/x/arch/x86/x86asm"
|
"golang.org/x/arch/x86/x86asm"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
|
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
|
||||||
@@ -25,7 +25,7 @@ import (
|
|||||||
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
||||||
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
path := "../../go-libraries/go-flac/avx2_amd64.s"
|
||||||
if _, err := os.Stat(path); err != nil {
|
if _, err := os.Stat(path); err != nil {
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
t.Skip("go-libraries repository not present next to gasm-sdk")
|
||||||
}
|
}
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -86,7 +86,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
|
|||||||
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
|
||||||
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
path := "../../go-libraries/go-flac/avx512_amd64.s"
|
||||||
if _, err := os.Stat(path); err != nil {
|
if _, err := os.Stat(path); err != nil {
|
||||||
t.Skip("go-libraries repository not present next to gasm-devkit")
|
t.Skip("go-libraries repository not present next to gasm-sdk")
|
||||||
}
|
}
|
||||||
src, err := os.ReadFile(path)
|
src, err := os.ReadFile(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -0,0 +1,219 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package asm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The differential kernels for the DATA-path and front-end gaps are kept in
|
||||||
|
// testdata/verify beside the campaign's other kernels; the verify package's
|
||||||
|
// suites are not open to the asm package, so this test is their runner: each
|
||||||
|
// kernel assembles through gasm and through go tool asm, and the functions'
|
||||||
|
// bytes must agree with the relocation sites masked on both sides.
|
||||||
|
|
||||||
|
// toolAsmObject assembles path with the installed toolchain's assembler for
|
||||||
|
// goarch ("" = the host) and returns the object bytes.
|
||||||
|
func toolAsmObject(t *testing.T, path, goarch string) []byte {
|
||||||
|
t.Helper()
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
out, err := exec.Command(goBin, "env", "GOROOT").Output()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("go env GOROOT: %v", err)
|
||||||
|
}
|
||||||
|
includeDir := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||||
|
|
||||||
|
pkg := strings.TrimSuffix(filepath.Base(path), ".s")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_amd64")
|
||||||
|
pkg = strings.TrimSuffix(pkg, "_arm64")
|
||||||
|
|
||||||
|
objPath := filepath.Join(t.TempDir(), "oracle.o")
|
||||||
|
cmd := exec.Command(goBin, "tool", "asm", "-I", includeDir, "-p", pkg, "-o", objPath, path)
|
||||||
|
if goarch != "" {
|
||||||
|
environ := os.Environ()
|
||||||
|
env := make([]string, 0, len(environ)+1)
|
||||||
|
for _, e := range environ {
|
||||||
|
if !strings.HasPrefix(e, "GOARCH=") {
|
||||||
|
env = append(env, e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
cmd.Env = append(env, "GOARCH="+goarch)
|
||||||
|
}
|
||||||
|
if out, err := cmd.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("go tool asm %s: %v\n%s", filepath.Base(path), err, out)
|
||||||
|
}
|
||||||
|
obj, err := os.ReadFile(objPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return obj
|
||||||
|
}
|
||||||
|
|
||||||
|
// oracleFuncCode extracts the non-package TEXT functions' code bytes from a
|
||||||
|
// toolchain object, keyed by the name the object records (pkg.name). Each
|
||||||
|
// function's span is its own symbol size: a toolchain object that follows
|
||||||
|
// the text with data symbols (the synthesised float-constant pool) would
|
||||||
|
// otherwise fold them into the last function's bytes.
|
||||||
|
func oracleFuncCode(t *testing.T, obj []byte) map[string][]byte {
|
||||||
|
t.Helper()
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
le := binary.LittleEndian
|
||||||
|
const symSize = 21
|
||||||
|
nps := v.syms(blkNonpkgdef)
|
||||||
|
data := v.blk(blkData)
|
||||||
|
didx := v.blk(blkDataIdx)
|
||||||
|
preceding := 0
|
||||||
|
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef} {
|
||||||
|
preceding += len(v.blk(bi)) / symSize
|
||||||
|
}
|
||||||
|
out := make(map[string][]byte, len(nps))
|
||||||
|
for i, s := range nps {
|
||||||
|
if s.typ != kindSTEXT {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
start := le.Uint32(didx[4*(preceding+i):])
|
||||||
|
out[s.name] = data[start : start+s.size]
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// maskCode zeroes every relocation field, the way the toolchain's object
|
||||||
|
// leaves them for the linker.
|
||||||
|
func maskCode(code []byte, relocs []Reloc) []byte {
|
||||||
|
for _, r := range relocs {
|
||||||
|
for j := r.Off; j < r.Off+4 && j < len(code); j++ {
|
||||||
|
code[j] = 0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return code
|
||||||
|
}
|
||||||
|
|
||||||
|
// code assembles src for amd64 and returns the image's code bytes.
|
||||||
|
func code(path, src string) []byte {
|
||||||
|
f, errs := parser.Parse(path, src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return img.Code
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDifferentialKernels pins the new kernels against the oracle.
|
||||||
|
func TestDifferentialKernels(t *testing.T) {
|
||||||
|
if runtime.GOARCH != "amd64" {
|
||||||
|
t.Skip("the amd64 kernels assume an amd64 host assembler default")
|
||||||
|
}
|
||||||
|
for _, k := range []struct {
|
||||||
|
path string
|
||||||
|
goarch string
|
||||||
|
arm64 bool
|
||||||
|
}{
|
||||||
|
{filepath.Join("..", "testdata", "verify", "datarel_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "divslash_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "semicolons_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "quadreg_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "floatimm_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "bookkeep_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "forms_amd64.s"), "", false},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "datarel_arm64.s"), "arm64", true},
|
||||||
|
{filepath.Join("..", "testdata", "verify", "divslash_arm64.s"), "arm64", true},
|
||||||
|
} {
|
||||||
|
t.Run(filepath.Base(k.path), func(t *testing.T) {
|
||||||
|
src, err := os.ReadFile(k.path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read: %v", err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse(k.path, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
var img *Image
|
||||||
|
if k.arm64 {
|
||||||
|
img, err = AssembleFileARM64(f)
|
||||||
|
} else {
|
||||||
|
img, err = AssembleFile(f)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
gt := oracleFuncCode(t, toolAsmObject(t, k.path, k.goarch))
|
||||||
|
// The oracle keys its functions by the qualified object name
|
||||||
|
// (pkg.name); match on the local part.
|
||||||
|
byLocal := make(map[string][]byte, len(gt))
|
||||||
|
for name, code := range gt {
|
||||||
|
if _, after, ok := strings.Cut(name, "."); ok {
|
||||||
|
name = after
|
||||||
|
}
|
||||||
|
byLocal[name] = code
|
||||||
|
}
|
||||||
|
|
||||||
|
matched := 0
|
||||||
|
for _, fn := range img.Funcs {
|
||||||
|
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
|
||||||
|
goCode, ok := byLocal[fn.Name]
|
||||||
|
if !ok {
|
||||||
|
t.Errorf("%s: not in ground truth (%d functions: %v)", fn.Name, len(gt), keysOf(byLocal))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
goCode = maskCode(append([]byte(nil), goCode...), fn.Relocs)
|
||||||
|
cmpLen := min(len(goCode), len(gasmCode))
|
||||||
|
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
|
||||||
|
t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\ngasm %x\ngo %x", fn.Name, len(gasmCode), len(goCode), gasmCode, goCode)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, b := range goCode[len(gasmCode):] {
|
||||||
|
if b != 0 {
|
||||||
|
t.Errorf("%s: non-zero trailing bytes in go tool asm output", fn.Name)
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
matched++
|
||||||
|
t.Logf("%s: MATCH (%d bytes)", fn.Name, len(gasmCode))
|
||||||
|
}
|
||||||
|
if matched == 0 {
|
||||||
|
t.Fatal("no functions matched")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func keysOf(m map[string][]byte) []string {
|
||||||
|
out := make([]string, 0, len(m))
|
||||||
|
for k := range m {
|
||||||
|
out = append(out, k)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSemicolonSpellingParity pins that the ';' statement separator changes
|
||||||
|
// nothing about the encoding: the one-line spelling assembles to exactly the
|
||||||
|
// bytes of the same statements written one per line.
|
||||||
|
func TestSemicolonSpellingParity(t *testing.T) {
|
||||||
|
for _, tt := range []struct{ one, two string }{
|
||||||
|
{"\tROLQ $3, DI; ROLQ $13, DI\n", "\tROLQ $3, DI\n\tROLQ $13, DI\n"},
|
||||||
|
{"\tREP; MOVSQ\n", "\tREP\n\tMOVSQ\n"},
|
||||||
|
{"\tXORQ AX, AX; XORQ CX, CX\n", "\tXORQ AX, AX\n\tXORQ CX, CX\n"},
|
||||||
|
} {
|
||||||
|
one := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.one+"\tRET\n")
|
||||||
|
two := code("t.s", "TEXT \u00b7f(SB), NOSPLIT, $0\n"+tt.two+"\tRET\n")
|
||||||
|
if !bytes.Equal(one, two) {
|
||||||
|
t.Errorf("semicolon spelling %q: %x, want the two-line bytes %x", tt.one, one, two)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -12,7 +12,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
|
||||||
|
|||||||
+177
-12
@@ -5,10 +5,11 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"math"
|
||||||
"sort"
|
"sort"
|
||||||
"strconv"
|
"strconv"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Image is an assembled file: the function bodies laid out in source order,
|
// Image is an assembled file: the function bodies laid out in source order,
|
||||||
@@ -103,6 +104,7 @@ const (
|
|||||||
RelArm64Branch // R_CALLARM64 (BL instruction)
|
RelArm64Branch // R_CALLARM64 (BL instruction)
|
||||||
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
|
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
|
||||||
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
|
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
|
||||||
|
RelAddr // R_ADDR: the absolute address of a symbol held in a DATA field
|
||||||
)
|
)
|
||||||
|
|
||||||
type Reloc struct {
|
type Reloc struct {
|
||||||
@@ -112,13 +114,17 @@ type Reloc struct {
|
|||||||
// Addend select the target: the symbol plus the byte offset. An
|
// Addend select the target: the symbol plus the byte offset. An
|
||||||
// External relocation names a symbol no GLOBL in the file defines;
|
// External relocation names a symbol no GLOBL in the file defines;
|
||||||
// the object-file emitters carry it into the output's relocation
|
// the object-file emitters carry it into the output's relocation
|
||||||
// table.
|
// table. Siz is the width of the patched field and is set only for
|
||||||
|
// data-field relocations (RelAddr, Off relative to the data symbol),
|
||||||
|
// whose width is the DATA line's; code relocations take their width
|
||||||
|
// from the architecture's instruction encoding.
|
||||||
Off int
|
Off int
|
||||||
After int
|
After int
|
||||||
Name string
|
Name string
|
||||||
Addend int64
|
Addend int64
|
||||||
External bool
|
External bool
|
||||||
Kind RelocKind
|
Kind RelocKind
|
||||||
|
Siz uint8
|
||||||
}
|
}
|
||||||
|
|
||||||
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
// DataSymbol describes one GLOBL symbol laid out in the data section.
|
||||||
@@ -130,6 +136,11 @@ type DataSymbol struct {
|
|||||||
Static bool // the <> marker: file-local, not exported
|
Static bool // the <> marker: file-local, not exported
|
||||||
Rodata bool // the RODATA flag: read-only data
|
Rodata bool // the RODATA flag: read-only data
|
||||||
Dupok bool // the DUPOK flag: duplicate-OK
|
Dupok bool // the DUPOK flag: duplicate-OK
|
||||||
|
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
|
||||||
|
// $other(SB)"): fields of this symbol's data that hold another symbol's
|
||||||
|
// address, resolved by the linker. Off is relative to the symbol's
|
||||||
|
// data start.
|
||||||
|
Relocs []Reloc
|
||||||
}
|
}
|
||||||
|
|
||||||
// Bytes returns the whole image: code, then data.
|
// Bytes returns the whole image: code, then data.
|
||||||
@@ -139,6 +150,18 @@ func (img *Image) Bytes() []byte {
|
|||||||
return append(out, img.Data...)
|
return append(out, img.Data...)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// AssembleOption adjusts the file-level assembly context.
|
||||||
|
type AssembleOption func(*linkInfo)
|
||||||
|
|
||||||
|
// WithGOOS selects the target operating system for the forms that depend on
|
||||||
|
// it, the TLS access shape above all: linux and freebsd take the
|
||||||
|
// one-instruction form, windows and plan9 keep the two-instruction load.
|
||||||
|
func WithGOOS(goos string) AssembleOption {
|
||||||
|
return func(l *linkInfo) {
|
||||||
|
l.goos = goos
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
// AssembleFile assembles every TEXT function of a parsed file and lays out
|
||||||
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
// its static symbols (GLOBL/DATA) in a data section behind the code. Each
|
||||||
// reference to a file-local static symbol becomes a RIP-relative load whose
|
// reference to a file-local static symbol becomes a RIP-relative load whose
|
||||||
@@ -146,7 +169,7 @@ func (img *Image) Bytes() []byte {
|
|||||||
// GLOBL defines is recorded as an external relocation (Externals) with its
|
// GLOBL defines is recorded as an external relocation (Externals) with its
|
||||||
// displacement left zero, the object-file emitters resolve it at link
|
// displacement left zero, the object-file emitters resolve it at link
|
||||||
// time, while the raw image (Bytes) cannot represent it.
|
// time, while the raw image (Bytes) cannot represent it.
|
||||||
func AssembleFile(f *ast.File) (*Image, error) {
|
func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
|
||||||
dataSyms, err := collectData(f)
|
dataSyms, err := collectData(f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -155,7 +178,18 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
for _, d := range dataSyms {
|
for _, d := range dataSyms {
|
||||||
known[d.name] = true
|
known[d.name] = true
|
||||||
}
|
}
|
||||||
|
// TEXT symbols are file-level definitions too: a symbol immediate
|
||||||
|
// ($fn(SB)) may name one, exactly as a data reference names a GLOBL.
|
||||||
|
for _, d := range f.Decls {
|
||||||
|
if t, ok := d.(*ast.Text); ok {
|
||||||
|
known[t.Name.Name] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
link := &linkInfo{symbols: known, allowExternal: true}
|
link := &linkInfo{symbols: known, allowExternal: true}
|
||||||
|
for _, o := range opts {
|
||||||
|
o(link)
|
||||||
|
}
|
||||||
|
poolSeen := map[string]bool{}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
textOff := map[string]int{}
|
textOff := map[string]int{}
|
||||||
@@ -169,7 +203,26 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
code, patches, labels, steps, lines, err := assemble(t, link)
|
code, patches, labels, steps, lines, pool, err := assemble(t, link)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
|
}
|
||||||
|
// The pooled floating-point constants join the declared data as
|
||||||
|
// read-only symbols, deduplicated across the file (the toolchain
|
||||||
|
// synthesises the same symbols into its rodata).
|
||||||
|
for _, entry := range pool {
|
||||||
|
if poolSeen[entry.name] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
poolSeen[entry.name] = true
|
||||||
|
dataSyms = append(dataSyms, dataSym{
|
||||||
|
name: entry.name,
|
||||||
|
buf: entry.data,
|
||||||
|
size: len(entry.data),
|
||||||
|
rodata: true,
|
||||||
|
dupok: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
}
|
}
|
||||||
@@ -257,6 +310,22 @@ func AssembleFile(f *ast.File) (*Image, error) {
|
|||||||
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
img.Funcs[i].Relocs = append(img.Funcs[i].Relocs, reloc)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The data symbols' symbol-valued DATA fields resolve the same way the
|
||||||
|
// code references do: a name the file defines (GLOBL or TEXT) stays an
|
||||||
|
// internal reference the emitters resolve, anything else is external.
|
||||||
|
// img.DataSyms was laid out in dataSyms order, so the indexes line up.
|
||||||
|
for i := range img.DataSyms {
|
||||||
|
for _, r := range dataSyms[i].relocs {
|
||||||
|
reloc := r
|
||||||
|
if _, ok := img.Symbols[reloc.Name]; !ok {
|
||||||
|
if _, ok := textOff[reloc.Name]; !ok {
|
||||||
|
reloc.External = true
|
||||||
|
externals[reloc.Name] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
|
||||||
|
}
|
||||||
|
}
|
||||||
for name := range externals {
|
for name := range externals {
|
||||||
img.Externals = append(img.Externals, name)
|
img.Externals = append(img.Externals, name)
|
||||||
}
|
}
|
||||||
@@ -273,6 +342,10 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
// The pooled $i64 constants the wide MOV immediate loads refer to join
|
||||||
|
// the declared data as read-only symbols, deduplicated across the file
|
||||||
|
// (the toolchain synthesises the same symbols into its rodata).
|
||||||
|
litSeen := map[string]bool{}
|
||||||
|
|
||||||
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
@@ -280,10 +353,23 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
code, labels, relocs, lines, spadj, err := assembleRISCV(t)
|
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
||||||
}
|
}
|
||||||
|
for _, lit := range lits {
|
||||||
|
if litSeen[lit.Name] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
litSeen[lit.Name] = true
|
||||||
|
dataSyms = append(dataSyms, dataSym{
|
||||||
|
name: lit.Name,
|
||||||
|
buf: lit.Data,
|
||||||
|
size: len(lit.Data),
|
||||||
|
rodata: true,
|
||||||
|
dupok: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
fl := FuncLayout{
|
fl := FuncLayout{
|
||||||
Name: t.Name.Name,
|
Name: t.Name.Name,
|
||||||
Pkg: t.Name.Pkg,
|
Pkg: t.Name.Pkg,
|
||||||
@@ -429,6 +515,23 @@ func markExternals(img *Image, dataSyms []dataSym) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The declared data symbols carry the file's own relocations (the
|
||||||
|
// symbol-valued DATA fields); the layouts appended img.DataSyms in
|
||||||
|
// dataSyms order, so the indexes line up. The trailing entries (the
|
||||||
|
// pooled arm64 literals) have no source relocations.
|
||||||
|
for i := range img.DataSyms {
|
||||||
|
if i >= len(dataSyms) {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
for _, r := range dataSyms[i].relocs {
|
||||||
|
reloc := r
|
||||||
|
if !known[reloc.Name] {
|
||||||
|
reloc.External = true
|
||||||
|
externals[reloc.Name] = true
|
||||||
|
}
|
||||||
|
img.DataSyms[i].Relocs = append(img.DataSyms[i].Relocs, reloc)
|
||||||
|
}
|
||||||
|
}
|
||||||
for name := range externals {
|
for name := range externals {
|
||||||
img.Externals = append(img.Externals, name)
|
img.Externals = append(img.Externals, name)
|
||||||
}
|
}
|
||||||
@@ -444,6 +547,9 @@ type dataSym struct {
|
|||||||
static bool
|
static bool
|
||||||
rodata bool
|
rodata bool
|
||||||
dupok bool
|
dupok bool
|
||||||
|
// relocs are the symbol-valued DATA fields, in declaration order; Off
|
||||||
|
// is relative to the symbol's data start.
|
||||||
|
relocs []Reloc
|
||||||
}
|
}
|
||||||
|
|
||||||
// collectData gathers the file's static symbols (GLOBL) and their initial
|
// collectData gathers the file's static symbols (GLOBL) and their initial
|
||||||
@@ -511,20 +617,79 @@ func collectData(f *ast.File) ([]dataSym, error) {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
return nil, fmt.Errorf("DATA %q: no matching GLOBL", dd.Name.Name)
|
||||||
}
|
}
|
||||||
if dd.Value == nil || !dd.Value.Imm.HasVal {
|
if dd.Value == nil {
|
||||||
return nil, fmt.Errorf("DATA %q: value must be an integer immediate", dd.Name.Name)
|
return nil, fmt.Errorf("DATA %q: missing value", dd.Name.Name)
|
||||||
}
|
}
|
||||||
w := dd.Width
|
w := dd.Width
|
||||||
switch w {
|
|
||||||
case 1, 2, 4, 8:
|
|
||||||
default:
|
|
||||||
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
|
||||||
}
|
|
||||||
off := dd.Name.Offset
|
off := dd.Name.Offset
|
||||||
buf := syms[i].buf
|
buf := syms[i].buf
|
||||||
if off < 0 || off+int64(w) > int64(len(buf)) {
|
if off < 0 || off+int64(w) > int64(len(buf)) {
|
||||||
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
return nil, fmt.Errorf("DATA %q+%d/%d exceeds GLOBL size %d", dd.Name.Name, off, w, len(buf))
|
||||||
}
|
}
|
||||||
|
// A symbol value ("DATA s+0(SB)/8, $other(SB)", the rt0 spelling)
|
||||||
|
// leaves the field zero and records a relocation against the named
|
||||||
|
// symbol: the linker patches the absolute address at this data
|
||||||
|
// offset. The toolchain emits the same shape, an R_ADDR of the
|
||||||
|
// DATA width with the value's offset as the addend, on every
|
||||||
|
// architecture.
|
||||||
|
if sym := dd.Value.Imm.Sym; !dd.Value.Imm.HasVal && sym != nil {
|
||||||
|
syms[i].relocs = append(syms[i].relocs, Reloc{
|
||||||
|
Off: int(off),
|
||||||
|
Name: sym.Name,
|
||||||
|
Addend: sym.Offset,
|
||||||
|
Kind: RelAddr,
|
||||||
|
Siz: uint8(w),
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// A string or rune value ("DATA s+0(SB)/20, $"text"") writes its
|
||||||
|
// bytes into the field and leaves the rest zero, the toolchain's
|
||||||
|
// WriteString: the declared width must hold every byte, and any
|
||||||
|
// width is legal.
|
||||||
|
if s := dd.Value.Imm.Str; s != "" && !dd.Value.Imm.HasVal {
|
||||||
|
text, err := strconv.Unquote(s)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid string value %s", dd.Name.Name, s)
|
||||||
|
}
|
||||||
|
if len(text) > w {
|
||||||
|
return nil, fmt.Errorf("DATA %q: string of %d bytes does not fit width %d", dd.Name.Name, len(text), w)
|
||||||
|
}
|
||||||
|
copy(buf[off:], text)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// A floating-point value stores its IEEE-754 bits: /4 the float32
|
||||||
|
// rounding of the parsed double, /8 the full 64 bits, the
|
||||||
|
// toolchain's WriteFloat32 and WriteFloat64.
|
||||||
|
if f := dd.Value.Imm.Float; f != "" && !dd.Value.Imm.HasVal {
|
||||||
|
num, err := strconv.ParseFloat(f, 64)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid floating-point value %q", dd.Name.Name, f)
|
||||||
|
}
|
||||||
|
if dd.Value.Imm.Neg {
|
||||||
|
num = -num
|
||||||
|
}
|
||||||
|
var v uint64
|
||||||
|
switch w {
|
||||||
|
case 4:
|
||||||
|
v = uint64(math.Float32bits(float32(num)))
|
||||||
|
case 8:
|
||||||
|
v = math.Float64bits(num)
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid width %d for a float (want 4 or 8)", dd.Name.Name, w)
|
||||||
|
}
|
||||||
|
for j := range w {
|
||||||
|
buf[off+int64(j)] = byte(v >> (8 * j))
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if !dd.Value.Imm.HasVal {
|
||||||
|
return nil, fmt.Errorf("DATA %q: value must be an integer immediate or a symbol address", dd.Name.Name)
|
||||||
|
}
|
||||||
|
switch w {
|
||||||
|
case 1, 2, 4, 8:
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("DATA %q: invalid width %d (want 1, 2, 4 or 8)", dd.Name.Name, w)
|
||||||
|
}
|
||||||
v := dd.Value.Imm.Val
|
v := dd.Value.Imm.Val
|
||||||
if dd.Value.Imm.Neg {
|
if dd.Value.Imm.Neg {
|
||||||
v = -v
|
v = -v
|
||||||
|
|||||||
+326
-1
@@ -4,10 +4,15 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestAssembleFileStaticData checks the whole-image layout; code, padding
|
// TestAssembleFileStaticData checks the whole-image layout; code, padding
|
||||||
@@ -166,3 +171,323 @@ func TestCollectDataNumericFlags(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestCollectDataSymbolValue covers the symbol-valued DATA field ("DATA
|
||||||
|
// s+0(SB)/8, $other(SB)", the rt0 spelling): the field stays zero in the
|
||||||
|
// image and the relocation is recorded against the named symbol, whatever
|
||||||
|
// the file defines (a TEXT function, a GLOBL) or leaves external.
|
||||||
|
func TestCollectDataSymbolValue(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ target+0(FP), AX
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $32
|
||||||
|
DATA holder+0(SB)/8, $·Keep(SB)
|
||||||
|
DATA holder+8(SB)/8, $·Keep+5(SB)
|
||||||
|
DATA holder+16(SB)/8, $holder(SB)
|
||||||
|
GLOBL spare(SB), NOPTR, $8
|
||||||
|
DATA spare+0(SB)/8, $extvar(SB)
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("f_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
byName := map[string]DataSymbol{}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
byName[d.Name] = d
|
||||||
|
}
|
||||||
|
want := []struct {
|
||||||
|
sym string
|
||||||
|
off int
|
||||||
|
name string
|
||||||
|
addend int64
|
||||||
|
ext bool
|
||||||
|
}{
|
||||||
|
{"holder", 0, "Keep", 0, false},
|
||||||
|
{"holder", 8, "Keep", 5, false},
|
||||||
|
{"holder", 16, "holder", 0, false},
|
||||||
|
{"spare", 0, "extvar", 0, true},
|
||||||
|
}
|
||||||
|
var flat []struct {
|
||||||
|
sym string
|
||||||
|
r Reloc
|
||||||
|
}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
for _, r := range d.Relocs {
|
||||||
|
flat = append(flat, struct {
|
||||||
|
sym string
|
||||||
|
r Reloc
|
||||||
|
}{d.Name, r})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(flat) != len(want) {
|
||||||
|
t.Fatalf("data relocations = %d, want %d", len(flat), len(want))
|
||||||
|
}
|
||||||
|
for i, w := range want {
|
||||||
|
g := flat[i]
|
||||||
|
r := g.r
|
||||||
|
if g.sym != w.sym {
|
||||||
|
t.Errorf("relocation %d sits on %q, want %q", i, g.sym, w.sym)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if r.Off != w.off || r.Name != w.name || r.Addend != w.addend || r.External != w.ext {
|
||||||
|
t.Errorf("relocation %d = {+%d %q addend %d ext %v}, want {+%d %q addend %d ext %v}",
|
||||||
|
i, r.Off, r.Name, r.Addend, r.External, w.off, w.name, w.addend, w.ext)
|
||||||
|
}
|
||||||
|
if r.Kind != RelAddr {
|
||||||
|
t.Errorf("relocation %d kind = %v, want RelAddr", i, r.Kind)
|
||||||
|
}
|
||||||
|
if r.Siz != 8 {
|
||||||
|
t.Errorf("relocation %d siz = %d, want 8", i, r.Siz)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The fields themselves stay zero: only the linker fills them.
|
||||||
|
for _, b := range img.Data {
|
||||||
|
if b != 0 {
|
||||||
|
t.Fatal("data section is not all zero before relocation")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(img.Externals) != 1 || img.Externals[0] != "extvar" {
|
||||||
|
t.Errorf("Externals = %v, want [extvar]", img.Externals)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectDataSymbolReloc pins the GOOBJ record a symbol-valued DATA
|
||||||
|
// field produces, against the shape the toolchain emits for the same
|
||||||
|
// source: an R_ADDR of the DATA width at the field offset, pkgIdxNone plus
|
||||||
|
// the non-package definition index when the target is the file's own TEXT
|
||||||
|
// function (the rt0 lib entry spelling).
|
||||||
|
func TestGOObjectDataSymbolReloc(t *testing.T) {
|
||||||
|
f, errs := parser.Parse("f_amd64.s", `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
RET
|
||||||
|
GLOBL holder(SB), NOPTR, $16
|
||||||
|
DATA holder+0(SB)/8, $·Keep+5(SB)
|
||||||
|
`)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
obj, err := img.GOObject("main", "f_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObject: %v", err)
|
||||||
|
}
|
||||||
|
v := openGoobj(t, obj)
|
||||||
|
// Walk every relocation record; the data record is the one of Siz 8
|
||||||
|
// and type R_ADDR.
|
||||||
|
var off, add int64
|
||||||
|
var pkg, sym uint32
|
||||||
|
found := false
|
||||||
|
for data := v.blk(blkReloc); len(data) >= 23; data = data[23:] {
|
||||||
|
if data[4] != 8 || binary.LittleEndian.Uint16(data[5:]) != relocAddr {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
found = true
|
||||||
|
off = int64(int32(binary.LittleEndian.Uint32(data[0:])))
|
||||||
|
add = int64(binary.LittleEndian.Uint64(data[7:]))
|
||||||
|
pkg = binary.LittleEndian.Uint32(data[15:])
|
||||||
|
sym = binary.LittleEndian.Uint32(data[19:])
|
||||||
|
break
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Fatal("no data relocation record in the object")
|
||||||
|
}
|
||||||
|
if off != 0 || add != 5 {
|
||||||
|
t.Errorf("data reloc = {off %d addend %d}, want {off 0 addend 5}", off, add)
|
||||||
|
}
|
||||||
|
if pkg != pkgIdxNone {
|
||||||
|
t.Errorf("data reloc pkg = %#x, want pkgIdxNone (the TEXT function)", pkg)
|
||||||
|
}
|
||||||
|
// The function's non-package definition index: the four pc tables
|
||||||
|
// precede it, so index 4.
|
||||||
|
if sym != 4 {
|
||||||
|
t.Errorf("data reloc sym = %d, want 4", sym)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGOObjectDataSymbolLink is the end-to-end proof for symbol-valued DATA
|
||||||
|
// fields: the gasm object is substituted for the toolchain's and re-linked,
|
||||||
|
// then executed, and the linked data word must hold the real address of the
|
||||||
|
// function the DATA line named (runtime.FuncForPC identifies it).
|
||||||
|
func TestGOObjectDataSymbolLink(t *testing.T) {
|
||||||
|
goBin, err := exec.LookPath("go")
|
||||||
|
if err != nil {
|
||||||
|
t.Skip("no Go toolchain available")
|
||||||
|
}
|
||||||
|
dir := t.TempDir()
|
||||||
|
asmSrc := `#include "textflag.h"
|
||||||
|
GLOBL entry(SB), NOPTR, $8
|
||||||
|
DATA entry+0(SB)/8, $·keepme(SB)
|
||||||
|
|
||||||
|
TEXT ·keepme(SB), NOSPLIT, $0-0
|
||||||
|
RET
|
||||||
|
|
||||||
|
TEXT ·entryptr(SB), NOSPLIT, $0-8
|
||||||
|
MOVQ entry+0(SB), AX
|
||||||
|
MOVQ AX, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main_amd64.s"), []byte(asmSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
mainSrc := `package main
|
||||||
|
|
||||||
|
import "runtime"
|
||||||
|
|
||||||
|
func keepme()
|
||||||
|
func entryptr() uintptr
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
pc := entryptr()
|
||||||
|
fn := runtime.FuncForPC(pc)
|
||||||
|
if fn == nil {
|
||||||
|
panic("the entry word does not point at a function")
|
||||||
|
}
|
||||||
|
if fn.Name() != "main.keepme" {
|
||||||
|
panic("the entry word points at " + fn.Name())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
`
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module dlink\n\ngo 1.21\n"), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Capture the build: the package archive's asm object and the link line.
|
||||||
|
build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".")
|
||||||
|
build.Dir = dir
|
||||||
|
buildLog, err := build.CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("baseline build: %v\n%s", err, buildLog)
|
||||||
|
}
|
||||||
|
var work, linkLine, asmObj string
|
||||||
|
for line := range strings.SplitSeq(string(buildLog), "\n") {
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(line, "WORK="):
|
||||||
|
work = strings.TrimPrefix(line, "WORK=")
|
||||||
|
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
|
||||||
|
asmObj = fieldAfter(line, "-o")
|
||||||
|
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
|
||||||
|
linkLine = line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if work == "" || asmObj == "" || linkLine == "" {
|
||||||
|
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
|
||||||
|
}
|
||||||
|
defer os.RemoveAll(work)
|
||||||
|
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
|
||||||
|
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
|
||||||
|
|
||||||
|
// Assemble the same source with gasm and substitute the object.
|
||||||
|
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f, errs := parser.Parse("main_amd64.s", string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
gasmObj, err := img.GOObject("dlink", "main_amd64.s")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GOObject: %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
|
||||||
|
t.Fatalf("write gasm object: %v", err)
|
||||||
|
}
|
||||||
|
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
|
||||||
|
if out, err := linkCmd.CombinedOutput(); err != nil {
|
||||||
|
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The linked program must run and find the right function behind the
|
||||||
|
// data word.
|
||||||
|
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("linked program failed: %v\n%s", err, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCollectDataFloatAndStringValues covers the non-integer DATA values the
|
||||||
|
// runtime's math and asm files use: floating-point initialisers store their
|
||||||
|
// IEEE-754 bits (/4 the float32 rounding, /8 the full double) and string
|
||||||
|
// initialisers write their bytes zero-padded within the declared width.
|
||||||
|
func TestCollectDataFloatAndStringValues(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·Keep(SB), NOSPLIT, $0-8
|
||||||
|
RET
|
||||||
|
GLOBL vals<>(SB), RODATA, $44
|
||||||
|
DATA vals<>+0(SB)/8, $0.5
|
||||||
|
DATA vals<>+8(SB)/8, $-1.0
|
||||||
|
DATA vals<>+16(SB)/4, $1.5
|
||||||
|
DATA vals<>+20(SB)/16, $"call frame too "
|
||||||
|
DATA vals<>+36(SB)/4, $"hi"
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("fvals_amd64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFile(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFile: %v", err)
|
||||||
|
}
|
||||||
|
byName := map[string]DataSymbol{}
|
||||||
|
for _, d := range img.DataSyms {
|
||||||
|
byName[d.Name] = d
|
||||||
|
}
|
||||||
|
d := byName["vals"]
|
||||||
|
if d.Size != 44 {
|
||||||
|
t.Fatalf("vals size = %d, want 44", d.Size)
|
||||||
|
}
|
||||||
|
buf := img.Data[d.Offset : d.Offset+44]
|
||||||
|
// 0.5 = 0x3FE0000000000000, -1.0 = 0xBFF0000000000000 (float64);
|
||||||
|
// 1.5 = 0x3FC00000 (float32).
|
||||||
|
for _, c := range []struct {
|
||||||
|
off int
|
||||||
|
want []byte
|
||||||
|
}{
|
||||||
|
{0, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xE0, 0x3F}},
|
||||||
|
{8, []byte{0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xF0, 0xBF}},
|
||||||
|
{16, []byte{0x00, 0x00, 0xC0, 0x3F}},
|
||||||
|
{20, []byte("call frame too ")},
|
||||||
|
{36, []byte{'h', 'i', 0x00, 0x00}},
|
||||||
|
} {
|
||||||
|
if string(buf[c.off:c.off+len(c.want)]) != string(c.want) {
|
||||||
|
t.Errorf("vals+%d: got % x, want % x", c.off, buf[c.off:c.off+len(c.want)], c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCollectDataValueErrors pins the value-kind width rules: a float needs
|
||||||
|
// width 4 or 8, a string must fit its declared width, and a bad float
|
||||||
|
// literal is diagnosed rather than stored.
|
||||||
|
func TestCollectDataValueErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
`GLOBL v<>(SB), RODATA, $4
|
||||||
|
DATA v<>+0(SB)/1, $0.5`,
|
||||||
|
`GLOBL v<>(SB), RODATA, $2
|
||||||
|
DATA v<>+0(SB)/2, $"toolarge"`,
|
||||||
|
}
|
||||||
|
for i, src := range cases {
|
||||||
|
full := "#include \"textflag.h\"\nTEXT ·Keep(SB), NOSPLIT, $0-8\n\tRET\n" + src
|
||||||
|
f, errs := parser.Parse(fmt.Sprintf("verr%d_amd64.s", i), full)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("case %d parse: %v", i, errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFile(f); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+873
-46
File diff suppressed because it is too large
Load Diff
+608
-6
@@ -30,7 +30,10 @@ package asm
|
|||||||
// of the immediate and register fields), mirroring the toolchain's OP_*
|
// of the immediate and register fields), mirroring the toolchain's OP_*
|
||||||
// helpers, so each l64* function only ORs its fields in.
|
// helpers, so each l64* function only ORs its fields in.
|
||||||
|
|
||||||
import "maps"
|
import (
|
||||||
|
"maps"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
||||||
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
||||||
@@ -103,7 +106,12 @@ func loong64RegNum(name string) int {
|
|||||||
case "R31", "S8":
|
case "R31", "S8":
|
||||||
return 31
|
return 31
|
||||||
}
|
}
|
||||||
// F0-F31, FCC0-FCC7, FCSR0-FCSR31.
|
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
|
||||||
|
// X0-X31) are deliberately NOT accepted here: they are a separate
|
||||||
|
// register class, and the toolchain rejects V/X names wherever an
|
||||||
|
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
|
||||||
|
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
|
||||||
|
// resolved only through loong64VecRegNum.
|
||||||
if len(name) >= 4 && name[:4] == "FCSR" {
|
if len(name) >= 4 && name[:4] == "FCSR" {
|
||||||
return loong64RegSpecial(name[4:], 31)
|
return loong64RegSpecial(name[4:], 31)
|
||||||
}
|
}
|
||||||
@@ -148,6 +156,19 @@ func loong64RegSpecial(digits string, max int) int {
|
|||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
|
||||||
|
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
|
||||||
|
// of their own: the toolchain accepts them only in the vector operands of the
|
||||||
|
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
|
||||||
|
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
|
||||||
|
// spellings never reach the integer/FP resolver.
|
||||||
|
func loong64VecRegNum(name string) int {
|
||||||
|
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
|
||||||
|
return -1
|
||||||
|
}
|
||||||
|
return loong64RegSpecial(name[1:], 31)
|
||||||
|
}
|
||||||
|
|
||||||
// ---- format helpers ----
|
// ---- format helpers ----
|
||||||
|
|
||||||
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
||||||
@@ -247,7 +268,7 @@ const (
|
|||||||
l64Firr14 // 2RI14 (ldptr/stptr)
|
l64Firr14 // 2RI14 (ldptr/stptr)
|
||||||
l64Firr16 // 2RI16 (addu16i.d)
|
l64Firr16 // 2RI16 (addu16i.d)
|
||||||
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
||||||
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub)
|
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
|
||||||
l64Firir // bstrins/bstrpick
|
l64Firir // bstrins/bstrpick
|
||||||
l64Firrr // alsl
|
l64Firrr // alsl
|
||||||
l64Fi15 // syscall/break/dbar
|
l64Fi15 // syscall/break/dbar
|
||||||
@@ -255,6 +276,9 @@ const (
|
|||||||
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
||||||
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
||||||
l64Fpreld // preld (2RI12 + 5-bit hint)
|
l64Fpreld // preld (2RI12 + 5-bit hint)
|
||||||
|
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
|
||||||
|
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
|
||||||
|
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
|
||||||
)
|
)
|
||||||
|
|
||||||
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
@@ -277,10 +301,72 @@ type l64DualEnc struct {
|
|||||||
var l64DualTable = map[string]l64DualEnc{}
|
var l64DualTable = map[string]l64DualEnc{}
|
||||||
|
|
||||||
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
||||||
// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered
|
// to their encoding.
|
||||||
// yet; the base integer, memory and floating-point ISA is complete.
|
|
||||||
var l64InstrTable = map[string]l64Enc{}
|
var l64InstrTable = map[string]l64Enc{}
|
||||||
|
|
||||||
|
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
|
||||||
|
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
|
||||||
|
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
|
||||||
|
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
|
||||||
|
type l64Vec3Enc struct {
|
||||||
|
op uint32
|
||||||
|
lasx bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
|
||||||
|
// the opcode, the bank, the accepted immediate range, the bias the toolchain
|
||||||
|
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
|
||||||
|
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
|
||||||
|
type l64VecImmEnc struct {
|
||||||
|
op uint32
|
||||||
|
lasx bool
|
||||||
|
min, max int
|
||||||
|
bias int
|
||||||
|
mask int
|
||||||
|
}
|
||||||
|
|
||||||
|
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
|
||||||
|
// each accepts; presence in the map routes the mnemonic through the vector
|
||||||
|
// dispatcher rather than the integer/FP formats.
|
||||||
|
var l64VecBank = map[string]bool{}
|
||||||
|
|
||||||
|
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
|
||||||
|
var l64VecImmInfo = map[string]l64VecImmEnc{}
|
||||||
|
|
||||||
|
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
|
||||||
|
// vpcnt.v).
|
||||||
|
var l64Vec2R = map[string]bool{}
|
||||||
|
|
||||||
|
// l64Vec4R marks the four-operand vector mnemonics (INSTR va, vk, vj, vd,
|
||||||
|
// such as vshuf.b).
|
||||||
|
var l64Vec4R = map[string]bool{}
|
||||||
|
|
||||||
|
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
|
||||||
|
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
|
||||||
|
type l64VmovqEnc struct {
|
||||||
|
ld, st, ldx, stx uint32 // plain and indexed load/store
|
||||||
|
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
|
||||||
|
pickS, pickU uint32 // vpickve2gr.{,u} element extract
|
||||||
|
ins uint32 // vinsgr2vr element insert
|
||||||
|
dup uint32 // vreplgr2vr duplicate (width in [11:10])
|
||||||
|
move uint32 // vori.b/xvori.b $0 register move
|
||||||
|
}
|
||||||
|
|
||||||
|
var l64VmovqTable = map[bool]l64VmovqEnc{
|
||||||
|
false: { // VMOVQ, the LSX (V) bank
|
||||||
|
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
|
||||||
|
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
|
||||||
|
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
|
||||||
|
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
|
||||||
|
},
|
||||||
|
true: { // XVMOVQ, the LASX (X) bank
|
||||||
|
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
|
||||||
|
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
|
||||||
|
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
|
||||||
|
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
// 3R, integer.
|
// 3R, integer.
|
||||||
rrr := map[string]uint32{
|
rrr := map[string]uint32{
|
||||||
@@ -360,6 +446,18 @@ func init() {
|
|||||||
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
||||||
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
||||||
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
||||||
|
// LSX: convert a 64-bit integer lane to a double float. The operand
|
||||||
|
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
|
||||||
|
// so the entry stays on the 2R integer/FP format.
|
||||||
|
"FFINTDV": 0x474a << 10,
|
||||||
|
// The rest of the scalar conversions (all F-bank, 2R).
|
||||||
|
"FFINTFW": 0x4744 << 10, // ffint.s.w
|
||||||
|
"FFINTFV": 0x4746 << 10, // ffint.s.l
|
||||||
|
"FFINTDW": 0x4748 << 10, // ffint.d.w
|
||||||
|
"FTINTWF": 0x46c1 << 10, // ftint.w.s
|
||||||
|
"FTINTWD": 0x46c2 << 10, // ftint.w.d
|
||||||
|
"FTINTVF": 0x46c9 << 10, // ftint.l.s
|
||||||
|
"FTINTVD": 0x46ca << 10, // ftint.l.d
|
||||||
}
|
}
|
||||||
for m, op := range rr {
|
for m, op := range rr {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
||||||
@@ -416,12 +514,14 @@ func init() {
|
|||||||
// LUI is the Plan 9 spelling of lu12i.w.
|
// LUI is the Plan 9 spelling of lu12i.w.
|
||||||
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
||||||
|
|
||||||
// 4R, fused multiply-add.
|
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
|
||||||
|
// condition flag, the layout matches the 4R shape).
|
||||||
rrrr := map[string]uint32{
|
rrrr := map[string]uint32{
|
||||||
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
||||||
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
||||||
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
||||||
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
||||||
|
"FSEL": 0x340 << 18,
|
||||||
}
|
}
|
||||||
for m, op := range rrrr {
|
for m, op := range rrrr {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
||||||
@@ -455,6 +555,10 @@ func init() {
|
|||||||
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
||||||
|
|
||||||
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
||||||
|
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
|
||||||
|
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
|
||||||
|
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
|
||||||
|
// is rejected by the oracle.
|
||||||
am := map[string]uint32{
|
am := map[string]uint32{
|
||||||
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
||||||
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
||||||
@@ -472,10 +576,508 @@ func init() {
|
|||||||
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
||||||
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
||||||
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
||||||
|
// The _dbar (acquire/release) add, and, or variants: opcodes read off
|
||||||
|
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
|
||||||
|
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
|
||||||
|
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
|
||||||
|
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
|
||||||
|
// The remaining _dbar exchange variants (loong64enc1.s).
|
||||||
|
"AMXORDBW": 0x070DA << 15, "AMXORDBV": 0x070DB << 15,
|
||||||
|
"AMMAXDBW": 0x070DC << 15, "AMMAXDBV": 0x070DD << 15,
|
||||||
|
"AMMINDBW": 0x070DE << 15, "AMMINDBV": 0x070DF << 15,
|
||||||
|
"AMMAXDBWU": 0x070E0 << 15, "AMMAXDBVU": 0x070E1 << 15,
|
||||||
|
"AMMINDBWU": 0x070E2 << 15, "AMMINDBVU": 0x070E3 << 15,
|
||||||
}
|
}
|
||||||
for m, op := range am {
|
for m, op := range am {
|
||||||
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- LSX/LASX (V*/XV*) ----
|
||||||
|
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
|
||||||
|
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
|
||||||
|
// most of them), not assumed from the LoongArch manual.
|
||||||
|
|
||||||
|
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
|
||||||
|
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
|
||||||
|
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
|
||||||
|
vec3 := map[string]l64Vec3Enc{
|
||||||
|
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
|
||||||
|
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
|
||||||
|
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
|
||||||
|
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
|
||||||
|
"XVADDV": {0xE817 << 15, true},
|
||||||
|
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
|
||||||
|
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// The integer and FP add/subtract families: [X]VADD and [X]VSUB by lane
|
||||||
|
// width, plus the [X]VSADD/[X]VSSUB saturating pairs.
|
||||||
|
// Opcodes transcribed from the toolchain's loong64enc1.s.
|
||||||
|
addsub := map[string]l64Vec3Enc{
|
||||||
|
"VADDB": {0xE014 << 15, false}, "VADDH": {0xE015 << 15, false},
|
||||||
|
"VADDD": {0xE262 << 15, false}, "VADDF": {0xE261 << 15, false},
|
||||||
|
"VADDQ": {0xE25A << 15, false},
|
||||||
|
"VSUBB": {0xE018 << 15, false}, "VSUBH": {0xE019 << 15, false},
|
||||||
|
"VSUBW": {0xE01A << 15, false}, "VSUBV": {0xE01B << 15, false},
|
||||||
|
"VSUBQ": {0xE25B << 15, false},
|
||||||
|
"VSUBF": {0xE265 << 15, false}, "VSUBD": {0xE266 << 15, false},
|
||||||
|
"VSADDB": {0xE08C << 15, false}, "VSADDH": {0xE08D << 15, false},
|
||||||
|
"VSADDW": {0xE08E << 15, false}, "VSADDV": {0xE08F << 15, false},
|
||||||
|
"VSADDBU": {0xE094 << 15, false}, "VSADDHU": {0xE095 << 15, false},
|
||||||
|
"VSADDWU": {0xE096 << 15, false}, "VSADDVU": {0xE097 << 15, false},
|
||||||
|
"VSSUBB": {0xE090 << 15, false}, "VSSUBH": {0xE091 << 15, false},
|
||||||
|
"VSSUBW": {0xE092 << 15, false}, "VSSUBV": {0xE093 << 15, false},
|
||||||
|
"VSSUBBU": {0xE098 << 15, false}, "VSSUBHU": {0xE099 << 15, false},
|
||||||
|
"VSSUBWU": {0xE09A << 15, false}, "VSSUBVU": {0xE09B << 15, false},
|
||||||
|
"XVADDB": {0xE814 << 15, true}, "XVADDH": {0xE815 << 15, true},
|
||||||
|
"XVADDW": {0xE816 << 15, true},
|
||||||
|
"XVADDD": {0xEA62 << 15, true}, "XVADDF": {0xEA61 << 15, true},
|
||||||
|
"XVADDQ": {0xEA5A << 15, true},
|
||||||
|
"XVSUBB": {0xE818 << 15, true}, "XVSUBH": {0xE819 << 15, true},
|
||||||
|
"XVSUBW": {0xE81A << 15, true}, "XVSUBV": {0xE81B << 15, true},
|
||||||
|
"XVSUBQ": {0xEA5B << 15, true},
|
||||||
|
"XVSUBF": {0xEA65 << 15, true}, "XVSUBD": {0xEA66 << 15, true},
|
||||||
|
"XVSADDB": {0xE88C << 15, true}, "XVSADDH": {0xE88D << 15, true},
|
||||||
|
"XVSADDW": {0xE88E << 15, true}, "XVSADDV": {0xE88F << 15, true},
|
||||||
|
"XVSADDBU": {0xE894 << 15, true}, "XVSADDHU": {0xE895 << 15, true},
|
||||||
|
"XVSADDWU": {0xE896 << 15, true}, "XVSADDVU": {0xE897 << 15, true},
|
||||||
|
"XVSSUBB": {0xE890 << 15, true}, "XVSSUBH": {0xE891 << 15, true},
|
||||||
|
"XVSSUBW": {0xE892 << 15, true}, "XVSSUBV": {0xE893 << 15, true},
|
||||||
|
"XVSSUBBU": {0xE898 << 15, true}, "XVSSUBHU": {0xE899 << 15, true},
|
||||||
|
"XVSSUBWU": {0xE89A << 15, true}, "XVSSUBVU": {0xE89B << 15, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// The multiply families: plain and high-half [X]VMUL/[X]VMUH, the
|
||||||
|
// widening [X]VMULW{EV,OD} ladder and its accumulating [X]VMADDW twins,
|
||||||
|
// plus the [X]VMADD/[X]VMSUB fused multiply-add and the [X]VDIV/[X]VMOD
|
||||||
|
// divide and modulo pairs.
|
||||||
|
muldiv := map[string]l64Vec3Enc{
|
||||||
|
"VMULB": {0xE108 << 15, false}, "VMULH": {0xE109 << 15, false},
|
||||||
|
"VMULW": {0xE10A << 15, false}, "VMULV": {0xE10B << 15, false},
|
||||||
|
"VMUHB": {0xE10C << 15, false}, "VMUHH": {0xE10D << 15, false},
|
||||||
|
"VMUHW": {0xE10E << 15, false}, "VMUHV": {0xE10F << 15, false},
|
||||||
|
"VMUHBU": {0xE110 << 15, false}, "VMUHHU": {0xE111 << 15, false},
|
||||||
|
"VMUHWU": {0xE112 << 15, false}, "VMUHVU": {0xE113 << 15, false},
|
||||||
|
"VMULWEVHB": {0xE120 << 15, false}, "VMULWEVWH": {0xE121 << 15, false},
|
||||||
|
"VMULWEVVW": {0xE122 << 15, false}, "VMULWEVQV": {0xE123 << 15, false},
|
||||||
|
"VMULWODHB": {0xE124 << 15, false}, "VMULWODWH": {0xE125 << 15, false},
|
||||||
|
"VMULWODVW": {0xE126 << 15, false}, "VMULWODQV": {0xE127 << 15, false},
|
||||||
|
"VMULWEVHBU": {0xE130 << 15, false}, "VMULWEVWHU": {0xE131 << 15, false},
|
||||||
|
"VMULWEVVWU": {0xE132 << 15, false}, "VMULWEVQVU": {0xE133 << 15, false},
|
||||||
|
"VMULWODHBU": {0xE134 << 15, false}, "VMULWODWHU": {0xE135 << 15, false},
|
||||||
|
"VMULWODVWU": {0xE136 << 15, false}, "VMULWODQVU": {0xE137 << 15, false},
|
||||||
|
"VMULWEVHBUB": {0xE140 << 15, false}, "VMULWEVWHUH": {0xE141 << 15, false},
|
||||||
|
"VMULWEVVWUW": {0xE142 << 15, false}, "VMULWEVQVUV": {0xE143 << 15, false},
|
||||||
|
"VMULWODHBUB": {0xE144 << 15, false}, "VMULWODWHUH": {0xE145 << 15, false},
|
||||||
|
"VMULWODVWUW": {0xE146 << 15, false}, "VMULWODQVUV": {0xE147 << 15, false},
|
||||||
|
"VMADDB": {0xE150 << 15, false}, "VMADDH": {0xE151 << 15, false},
|
||||||
|
"VMADDW": {0xE152 << 15, false}, "VMADDV": {0xE153 << 15, false},
|
||||||
|
"VMSUBB": {0xE154 << 15, false}, "VMSUBH": {0xE155 << 15, false},
|
||||||
|
"VMSUBW": {0xE156 << 15, false}, "VMSUBV": {0xE157 << 15, false},
|
||||||
|
"VMADDWEVHB": {0xE158 << 15, false}, "VMADDWEVWH": {0xE159 << 15, false},
|
||||||
|
"VMADDWEVVW": {0xE15A << 15, false}, "VMADDWEVQV": {0xE15B << 15, false},
|
||||||
|
"VMADDWODHB": {0xE15C << 15, false}, "VMADDWODWH": {0xE15D << 15, false},
|
||||||
|
"VMADDWODVW": {0xE15E << 15, false}, "VMADDWODQV": {0xE15F << 15, false},
|
||||||
|
"VMADDWEVHBU": {0xE168 << 15, false}, "VMADDWEVWHU": {0xE169 << 15, false},
|
||||||
|
"VMADDWEVVWU": {0xE16A << 15, false}, "VMADDWEVQVU": {0xE16B << 15, false},
|
||||||
|
"VMADDWODHBU": {0xE16C << 15, false}, "VMADDWODWHU": {0xE16D << 15, false},
|
||||||
|
"VMADDWODVWU": {0xE16E << 15, false}, "VMADDWODQVU": {0xE16F << 15, false},
|
||||||
|
"VMADDWEVHBUB": {0xE178 << 15, false}, "VMADDWEVWHUH": {0xE179 << 15, false},
|
||||||
|
"VMADDWEVVWUW": {0xE17A << 15, false}, "VMADDWEVQVUV": {0xE17B << 15, false},
|
||||||
|
"VMADDWODHBUB": {0xE17C << 15, false}, "VMADDWODWHUH": {0xE17D << 15, false},
|
||||||
|
"VMADDWODVWUW": {0xE17E << 15, false}, "VMADDWODQVUV": {0xE17F << 15, false},
|
||||||
|
"VDIVB": {0xE1C0 << 15, false}, "VDIVH": {0xE1C1 << 15, false},
|
||||||
|
"VDIVW": {0xE1C2 << 15, false}, "VDIVV": {0xE1C3 << 15, false},
|
||||||
|
"VMODB": {0xE1C4 << 15, false}, "VMODH": {0xE1C5 << 15, false},
|
||||||
|
"VMODW": {0xE1C6 << 15, false}, "VMODV": {0xE1C7 << 15, false},
|
||||||
|
"VDIVBU": {0xE1C8 << 15, false}, "VDIVHU": {0xE1C9 << 15, false},
|
||||||
|
"VDIVWU": {0xE1CA << 15, false}, "VDIVVU": {0xE1CB << 15, false},
|
||||||
|
"VMODBU": {0xE1CC << 15, false}, "VMODHU": {0xE1CD << 15, false},
|
||||||
|
"VMODWU": {0xE1CE << 15, false}, "VMODVU": {0xE1CF << 15, false},
|
||||||
|
"VMULF": {0xE271 << 15, false}, "VMULD": {0xE272 << 15, false},
|
||||||
|
"VDIVF": {0xE275 << 15, false}, "VDIVD": {0xE276 << 15, false},
|
||||||
|
"XVMULB": {0xE908 << 15, true}, "XVMULH": {0xE909 << 15, true},
|
||||||
|
"XVMULW": {0xE90A << 15, true}, "XVMULV": {0xE90B << 15, true},
|
||||||
|
"XVMUHB": {0xE90C << 15, true}, "XVMUHH": {0xE90D << 15, true},
|
||||||
|
"XVMUHW": {0xE90E << 15, true}, "XVMUHV": {0xE90F << 15, true},
|
||||||
|
"XVMUHBU": {0xE910 << 15, true}, "XVMUHHU": {0xE911 << 15, true},
|
||||||
|
"XVMUHWU": {0xE912 << 15, true}, "XVMUHVU": {0xE913 << 15, true},
|
||||||
|
"XVMULWEVHB": {0xE920 << 15, true}, "XVMULWEVWH": {0xE921 << 15, true},
|
||||||
|
"XVMULWEVVW": {0xE922 << 15, true}, "XVMULWEVQV": {0xE923 << 15, true},
|
||||||
|
"XVMULWODHB": {0xE924 << 15, true}, "XVMULWODWH": {0xE925 << 15, true},
|
||||||
|
"XVMULWODVW": {0xE926 << 15, true}, "XVMULWODQV": {0xE927 << 15, true},
|
||||||
|
"XVMULWEVHBU": {0xE930 << 15, true}, "XVMULWEVWHU": {0xE931 << 15, true},
|
||||||
|
"XVMULWEVVWU": {0xE932 << 15, true}, "XVMULWEVQVU": {0xE933 << 15, true},
|
||||||
|
"XVMULWODHBU": {0xE934 << 15, true}, "XVMULWODWHU": {0xE935 << 15, true},
|
||||||
|
"XVMULWODVWU": {0xE936 << 15, true}, "XVMULWODQVU": {0xE937 << 15, true},
|
||||||
|
"XVMULWEVHBUB": {0xE940 << 15, true}, "XVMULWEVWHUH": {0xE941 << 15, true},
|
||||||
|
"XVMULWEVVWUW": {0xE942 << 15, true}, "XVMULWEVQVUV": {0xE943 << 15, true},
|
||||||
|
"XVMULWODHBUB": {0xE944 << 15, true}, "XVMULWODWHUH": {0xE945 << 15, true},
|
||||||
|
"XVMULWODVWUW": {0xE946 << 15, true}, "XVMULWODQVUV": {0xE947 << 15, true},
|
||||||
|
"XVMADDB": {0xE950 << 15, true}, "XVMADDH": {0xE951 << 15, true},
|
||||||
|
"XVMADDW": {0xE952 << 15, true}, "XVMADDV": {0xE953 << 15, true},
|
||||||
|
"XVMSUBB": {0xE954 << 15, true}, "XVMSUBH": {0xE955 << 15, true},
|
||||||
|
"XVMSUBW": {0xE956 << 15, true}, "XVMSUBV": {0xE957 << 15, true},
|
||||||
|
"XVMADDWEVHB": {0xE958 << 15, true}, "XVMADDWEVWH": {0xE959 << 15, true},
|
||||||
|
"XVMADDWEVVW": {0xE95A << 15, true}, "XVMADDWEVQV": {0xE95B << 15, true},
|
||||||
|
"XVMADDWODHB": {0xE95C << 15, true}, "XVMADDWODWH": {0xE95D << 15, true},
|
||||||
|
"XVMADDWODVW": {0xE95E << 15, true}, "XVMADDWODQV": {0xE95F << 15, true},
|
||||||
|
"XVMADDWEVHBU": {0xE968 << 15, true}, "XVMADDWEVWHU": {0xE969 << 15, true},
|
||||||
|
"XVMADDWEVVWU": {0xE96A << 15, true}, "XVMADDWEVQVU": {0xE96B << 15, true},
|
||||||
|
"XVMADDWODHBU": {0xE96C << 15, true}, "XVMADDWODWHU": {0xE96D << 15, true},
|
||||||
|
"XVMADDWODVWU": {0xE96E << 15, true}, "XVMADDWODQVU": {0xE96F << 15, true},
|
||||||
|
"XVMADDWEVHBUB": {0xE978 << 15, true}, "XVMADDWEVWHUH": {0xE979 << 15, true},
|
||||||
|
"XVMADDWEVVWUW": {0xE97A << 15, true}, "XVMADDWEVQVUV": {0xE97B << 15, true},
|
||||||
|
"XVMADDWODHBUB": {0xE97C << 15, true}, "XVMADDWODWHUH": {0xE97D << 15, true},
|
||||||
|
"XVMADDWODVWUW": {0xE97E << 15, true}, "XVMADDWODQVUV": {0xE97F << 15, true},
|
||||||
|
"XVDIVB": {0xE9C0 << 15, true}, "XVDIVH": {0xE9C1 << 15, true},
|
||||||
|
"XVDIVW": {0xE9C2 << 15, true}, "XVDIVV": {0xE9C3 << 15, true},
|
||||||
|
"XVMODB": {0xE9C4 << 15, true}, "XVMODH": {0xE9C5 << 15, true},
|
||||||
|
"XVMODW": {0xE9C6 << 15, true}, "XVMODV": {0xE9C7 << 15, true},
|
||||||
|
"XVDIVBU": {0xE9C8 << 15, true}, "XVDIVHU": {0xE9C9 << 15, true},
|
||||||
|
"XVDIVWU": {0xE9CA << 15, true}, "XVDIVVU": {0xE9CB << 15, true},
|
||||||
|
"XVMODBU": {0xE9CC << 15, true}, "XVMODHU": {0xE9CD << 15, true},
|
||||||
|
"XVMODWU": {0xE9CE << 15, true}, "XVMODVU": {0xE9CF << 15, true},
|
||||||
|
"XVMULF": {0xEA71 << 15, true}, "XVMULD": {0xEA72 << 15, true},
|
||||||
|
"XVDIVF": {0xEA75 << 15, true}, "XVDIVD": {0xEA76 << 15, true},
|
||||||
|
}
|
||||||
|
|
||||||
|
// The lane-wise shifts and rotates (three-register forms; the immediate
|
||||||
|
// forms live in l64VecImmInfo), the interleave families, the bit
|
||||||
|
// clear/set/rev register forms, the remaining logic and compare
|
||||||
|
// spellings, the widening add/subtract ladder and the vector FP
|
||||||
|
// arithmetic.
|
||||||
|
vecmisc := map[string]l64Vec3Enc{
|
||||||
|
"VSLLB": {0xE1D0 << 15, false}, "VSLLH": {0xE1D1 << 15, false},
|
||||||
|
"VSLLW": {0xE1D2 << 15, false}, "VSLLV": {0xE1D3 << 15, false},
|
||||||
|
"VSRLB": {0xE1D4 << 15, false}, "VSRLH": {0xE1D5 << 15, false},
|
||||||
|
"VSRLW": {0xE1D6 << 15, false}, "VSRLV": {0xE1D7 << 15, false},
|
||||||
|
"VSRAH": {0xE1D9 << 15, false}, "VSRAW": {0xE1DA << 15, false},
|
||||||
|
"VSRAV": {0xE1DB << 15, false},
|
||||||
|
"VROTRB": {0xE1DC << 15, false}, "VROTRH": {0xE1DD << 15, false},
|
||||||
|
"VROTRV": {0xE1DF << 15, false},
|
||||||
|
"VILVLB": {0xE234 << 15, false}, "VILVLH": {0xE235 << 15, false},
|
||||||
|
"VILVLW": {0xE236 << 15, false}, "VILVLV": {0xE237 << 15, false},
|
||||||
|
"VILVHB": {0xE238 << 15, false}, "VILVHH": {0xE239 << 15, false},
|
||||||
|
"VILVHW": {0xE23A << 15, false}, "VILVHV": {0xE23B << 15, false},
|
||||||
|
"VBITCLRB": {0xE218 << 15, false}, "VBITCLRH": {0xE219 << 15, false},
|
||||||
|
"VBITCLRW": {0xE21A << 15, false}, "VBITCLRV": {0xE21B << 15, false},
|
||||||
|
"VBITSETB": {0xE21C << 15, false}, "VBITSETH": {0xE21D << 15, false},
|
||||||
|
"VBITSETW": {0xE21E << 15, false}, "VBITSETV": {0xE21F << 15, false},
|
||||||
|
"VBITREVB": {0xE220 << 15, false}, "VBITREVH": {0xE221 << 15, false},
|
||||||
|
"VBITREVW": {0xE222 << 15, false}, "VBITREVV": {0xE223 << 15, false},
|
||||||
|
"VORV": {0xE24D << 15, false}, "VNORV": {0xE24F << 15, false},
|
||||||
|
"VANDNV": {0xE250 << 15, false}, "VORNV": {0xE251 << 15, false},
|
||||||
|
"VSEQH": {0xE001 << 15, false}, "VSEQW": {0xE002 << 15, false},
|
||||||
|
"VSLTB": {0xE00C << 15, false}, "VSLTH": {0xE00D << 15, false},
|
||||||
|
"VSLTW": {0xE00E << 15, false}, "VSLTV": {0xE00F << 15, false},
|
||||||
|
"VSLTBU": {0xE010 << 15, false}, "VSLTHU": {0xE011 << 15, false},
|
||||||
|
"VSLTWU": {0xE012 << 15, false}, "VSLTVU": {0xE013 << 15, false},
|
||||||
|
"VADDWEVHB": {0xE03C << 15, false}, "VADDWEVWH": {0xE03D << 15, false},
|
||||||
|
"VADDWEVVW": {0xE03E << 15, false}, "VADDWEVQV": {0xE03F << 15, false},
|
||||||
|
"VSUBWEVHB": {0xE040 << 15, false}, "VSUBWEVWH": {0xE041 << 15, false},
|
||||||
|
"VSUBWEVVW": {0xE042 << 15, false}, "VSUBWEVQV": {0xE043 << 15, false},
|
||||||
|
"VADDWODHB": {0xE044 << 15, false}, "VADDWODWH": {0xE045 << 15, false},
|
||||||
|
"VADDWODVW": {0xE046 << 15, false}, "VADDWODQV": {0xE047 << 15, false},
|
||||||
|
"VSUBWODHB": {0xE048 << 15, false}, "VSUBWODWH": {0xE049 << 15, false},
|
||||||
|
"VSUBWODVW": {0xE04A << 15, false}, "VSUBWODQV": {0xE04B << 15, false},
|
||||||
|
"VSUBWEVHBU": {0xE060 << 15, false}, "VSUBWEVWHU": {0xE061 << 15, false},
|
||||||
|
"VSUBWEVVWU": {0xE062 << 15, false}, "VSUBWEVQVU": {0xE063 << 15, false},
|
||||||
|
"VADDWEVHBU": {0xE05C << 15, false}, "VADDWEVWHU": {0xE05D << 15, false},
|
||||||
|
"VADDWEVVWU": {0xE05E << 15, false}, "VADDWEVQVU": {0xE05F << 15, false},
|
||||||
|
"VADDWODHBU": {0xE064 << 15, false}, "VADDWODWHU": {0xE065 << 15, false},
|
||||||
|
"VADDWODVWU": {0xE066 << 15, false}, "VADDWODQVU": {0xE067 << 15, false},
|
||||||
|
"VSUBWODHBU": {0xE068 << 15, false}, "VSUBWODWHU": {0xE069 << 15, false},
|
||||||
|
"VSUBWODVWU": {0xE06A << 15, false}, "VSUBWODQVU": {0xE06B << 15, false},
|
||||||
|
"VSHUFH": {0xE2F5 << 15, false}, "VSHUFW": {0xE2F6 << 15, false},
|
||||||
|
"VSHUFV": {0xE2F7 << 15, false},
|
||||||
|
"XVSLLB": {0xE9D0 << 15, true}, "XVSLLH": {0xE9D1 << 15, true},
|
||||||
|
"XVSLLW": {0xE9D2 << 15, true}, "XVSLLV": {0xE9D3 << 15, true},
|
||||||
|
"XVSRLB": {0xE9D4 << 15, true}, "XVSRLH": {0xE9D5 << 15, true},
|
||||||
|
"XVSRLW": {0xE9D6 << 15, true}, "XVSRLV": {0xE9D7 << 15, true},
|
||||||
|
"XVSRAB": {0xE9D8 << 15, true}, "XVSRAH": {0xE9D9 << 15, true},
|
||||||
|
"XVSRAW": {0xE9DA << 15, true}, "XVSRAV": {0xE9DB << 15, true},
|
||||||
|
"XVROTRB": {0xE9DC << 15, true}, "XVROTRH": {0xE9DD << 15, true},
|
||||||
|
"XVROTRW": {0xE9DE << 15, true}, "XVROTRV": {0xE9DF << 15, true},
|
||||||
|
"XVILVLB": {0xEA34 << 15, true}, "XVILVLH": {0xEA35 << 15, true},
|
||||||
|
"XVILVLW": {0xEA36 << 15, true}, "XVILVLV": {0xEA37 << 15, true},
|
||||||
|
"XVILVHB": {0xEA38 << 15, true}, "XVILVHH": {0xEA39 << 15, true},
|
||||||
|
"XVILVHW": {0xEA3A << 15, true}, "XVILVHV": {0xEA3B << 15, true},
|
||||||
|
"XVBITCLRB": {0xEA18 << 15, true}, "XVBITCLRH": {0xEA19 << 15, true},
|
||||||
|
"XVBITCLRW": {0xEA1A << 15, true}, "XVBITCLRV": {0xEA1B << 15, true},
|
||||||
|
"XVBITSETB": {0xEA1C << 15, true}, "XVBITSETH": {0xEA1D << 15, true},
|
||||||
|
"XVBITSETW": {0xEA1E << 15, true}, "XVBITSETV": {0xEA1F << 15, true},
|
||||||
|
"XVBITREVB": {0xEA20 << 15, true}, "XVBITREVH": {0xEA21 << 15, true},
|
||||||
|
"XVBITREVW": {0xEA22 << 15, true}, "XVBITREVV": {0xEA23 << 15, true},
|
||||||
|
"XVORV": {0xEA4D << 15, true}, "XVNORV": {0xEA4F << 15, true},
|
||||||
|
"XVANDNV": {0xEA50 << 15, true}, "XVORNV": {0xEA51 << 15, true},
|
||||||
|
"XVSEQH": {0xE801 << 15, true}, "XVSEQW": {0xE802 << 15, true},
|
||||||
|
"XVSLTB": {0xE80C << 15, true}, "XVSLTH": {0xE80D << 15, true},
|
||||||
|
"XVSLTW": {0xE80E << 15, true}, "XVSLTV": {0xE80F << 15, true},
|
||||||
|
"XVSLTBU": {0xE810 << 15, true}, "XVSLTHU": {0xE811 << 15, true},
|
||||||
|
"XVSLTWU": {0xE812 << 15, true}, "XVSLTVU": {0xE813 << 15, true},
|
||||||
|
"XVADDWEVHB": {0xE83C << 15, true}, "XVADDWEVWH": {0xE83D << 15, true},
|
||||||
|
"XVADDWEVVW": {0xE83E << 15, true}, "XVADDWEVQV": {0xE83F << 15, true},
|
||||||
|
"XVSUBWEVHB": {0xE840 << 15, true}, "XVSUBWEVWH": {0xE841 << 15, true},
|
||||||
|
"XVSUBWEVVW": {0xE842 << 15, true}, "XVSUBWEVQV": {0xE843 << 15, true},
|
||||||
|
"XVADDWODHB": {0xE844 << 15, true}, "XVADDWODWH": {0xE845 << 15, true},
|
||||||
|
"XVADDWODVW": {0xE846 << 15, true}, "XVADDWODQV": {0xE847 << 15, true},
|
||||||
|
"XVSUBWODHB": {0xE848 << 15, true}, "XVSUBWODWH": {0xE849 << 15, true},
|
||||||
|
"XVSUBWODVW": {0xE84A << 15, true}, "XVSUBWODQV": {0xE84B << 15, true},
|
||||||
|
"XVADDWEVHBU": {0xE85C << 15, true}, "XVADDWEVWHU": {0xE85D << 15, true},
|
||||||
|
"XVADDWEVVWU": {0xE85E << 15, true}, "XVADDWEVQVU": {0xE85F << 15, true},
|
||||||
|
"XVSUBWEVHBU": {0xE860 << 15, true}, "XVSUBWEVWHU": {0xE861 << 15, true},
|
||||||
|
"XVSUBWEVVWU": {0xE862 << 15, true}, "XVSUBWEVQVU": {0xE863 << 15, true},
|
||||||
|
"XVADDWODHBU": {0xE864 << 15, true}, "XVADDWODWHU": {0xE865 << 15, true},
|
||||||
|
"XVADDWODVWU": {0xE866 << 15, true}, "XVADDWODQVU": {0xE867 << 15, true},
|
||||||
|
"XVSUBWODHBU": {0xE868 << 15, true}, "XVSUBWODWHU": {0xE869 << 15, true},
|
||||||
|
"XVSUBWODVWU": {0xE86A << 15, true}, "XVSUBWODQVU": {0xE86B << 15, true},
|
||||||
|
"XVSHUFH": {0xEAF5 << 15, true}, "XVSHUFW": {0xEAF6 << 15, true},
|
||||||
|
"XVSHUFV": {0xEAF7 << 15, true},
|
||||||
|
}
|
||||||
|
for _, tab := range []map[string]l64Vec3Enc{addsub, muldiv, vecmisc} {
|
||||||
|
for m, e := range tab {
|
||||||
|
if _, dup := vec3[m]; dup {
|
||||||
|
panic("loong64: duplicate vector mnemonic " + m)
|
||||||
|
}
|
||||||
|
vec3[m] = e
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for m, e := range vec3 {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
}
|
||||||
|
|
||||||
|
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
|
||||||
|
// range, bias and field mask are the ones the toolchain encodes: vandi.b
|
||||||
|
// stores the raw 8-bit constant, vsrari.b stores imm+8 (lane-width
|
||||||
|
// bias), the si5 compares store 5-bit two's-complement values and vseqi.d
|
||||||
|
// a 7-bit field the toolchain range-checks down to si5.
|
||||||
|
// The mnemonics that also have a register form (the shifts, the bit
|
||||||
|
// clear/set/rev families, VSEQ and the logic immediates) keep their
|
||||||
|
// three-register entry in l64InstrTable; the dispatcher picks the
|
||||||
|
// immediate opcode from l64VecImmInfo by operand kind, so the immediate
|
||||||
|
// entries must not overwrite the table.
|
||||||
|
vecImm := map[string]l64VecImmEnc{
|
||||||
|
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VORB": {0xE7A8 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVORB": {0xEFA8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VXORB": {0xE7B0 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVXORB": {0xEFB0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
|
||||||
|
// 7-bit field, but the toolchain range-checks it down to si5 as well
|
||||||
|
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
|
||||||
|
"VSEQH": {0xE501 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQH": {0xED01 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
|
||||||
|
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
|
||||||
|
// vslti compares against a signed (or, in the U spellings, unsigned)
|
||||||
|
// si5/ui5 constant.
|
||||||
|
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTB": {0xED0C << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTH": {0xE50D << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTH": {0xED0D << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTW": {0xE50E << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTW": {0xED0E << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTV": {0xE50F << 15, false, -16, 15, 0, 0x1F},
|
||||||
|
"XVSLTV": {0xED0F << 15, true, -16, 15, 0, 0x1F},
|
||||||
|
"VSLTBU": {0xE510 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTBU": {0xED10 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLTHU": {0xE511 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTHU": {0xED11 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLTWU": {0xE512 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTWU": {0xED12 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLTVU": {0xE513 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLTVU": {0xED13 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
// vaddi/vsubi take ui5 constants for every width on this toolchain
|
||||||
|
// (VADDVU $32 is rejected by the oracle although the field is ui8).
|
||||||
|
"VADDBU": {0xE514 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDBU": {0xED14 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VADDHU": {0xE515 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDHU": {0xED15 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VADDWU": {0xE516 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDWU": {0xED16 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VADDVU": {0xE517 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVADDVU": {0xED17 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBBU": {0xE518 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBBU": {0xED18 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBHU": {0xE519 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBHU": {0xED19 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBWU": {0xE51A << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBWU": {0xED1A << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSUBVU": {0xE51B << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSUBVU": {0xED1B << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
// The shift/rotate immediates ride in a width-sized field whose upper
|
||||||
|
// bits carry the lane-width code: vslli.b stores ui3 at [12:0] with
|
||||||
|
// bits [14:13] inside the opcode, vslli.h ui4 under a 4 bit mask, and
|
||||||
|
// the .w/.d spellings a raw ui5/ui6.
|
||||||
|
"VSLLB": {0x732C2000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVSLLB": {0x772C2000, true, 0, 7, 0, 0x7},
|
||||||
|
"VSLLH": {0x732C4000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVSLLH": {0x772C4000, true, 0, 15, 0, 0xF},
|
||||||
|
"VSLLW": {0xE659 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSLLW": {0xEE59 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSLLV": {0xE65A << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVSLLV": {0xEE5A << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VSRLB": {0x73302000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVSRLB": {0x77302000, true, 0, 7, 0, 0x7},
|
||||||
|
"VSRLH": {0x73304000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVSRLH": {0x77304000, true, 0, 15, 0, 0xF},
|
||||||
|
"VSRLW": {0xE661 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSRLW": {0xEE61 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSRLV": {0xE662 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVSRLV": {0xEE62 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
// vsrari/vrotri bias the field so the lane-width code rides above the
|
||||||
|
// shift amount (.b adds 8, .h 16, .w 32; .d is a raw ui6).
|
||||||
|
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
|
||||||
|
"XVSRAB": {0xEE68 << 15, true, 0, 7, 8, 0x1F},
|
||||||
|
"VSRAH": {0x73344000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVSRAH": {0x77344000, true, 0, 15, 0, 0xF},
|
||||||
|
"VSRAW": {0xE669 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVSRAW": {0xEE69 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VSRAV": {0xE66A << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVSRAV": {0xEE6A << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VROTRB": {0x72A02000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVROTRB": {0x76A02000, true, 0, 7, 0, 0x7},
|
||||||
|
"VROTRH": {0x72A04000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVROTRH": {0x76A04000, true, 0, 15, 0, 0xF},
|
||||||
|
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVROTRW": {0xED41 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VROTRV": {0xE542 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVROTRV": {0xED42 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
// vbitclri/vbitseti/vbitrevi follow the same width-coded layout.
|
||||||
|
"VBITCLRB": {0x73102000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVBITCLRB": {0x77102000, true, 0, 7, 0, 0x7},
|
||||||
|
"VBITCLRH": {0x73104000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVBITCLRH": {0x77104000, true, 0, 15, 0, 0xF},
|
||||||
|
"VBITCLRW": {0xE621 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVBITCLRW": {0xEE21 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VBITCLRV": {0xE622 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVBITCLRV": {0xEE22 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VBITSETB": {0x73142000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVBITSETB": {0x77142000, true, 0, 7, 0, 0x7},
|
||||||
|
"VBITSETH": {0x73144000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVBITSETH": {0x77144000, true, 0, 15, 0, 0xF},
|
||||||
|
"VBITSETW": {0xE629 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVBITSETW": {0xEE29 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VBITSETV": {0xE62A << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVBITSETV": {0xEE2A << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
"VBITREVB": {0x73182000, false, 0, 7, 0, 0x7},
|
||||||
|
"XVBITREVB": {0x77182000, true, 0, 7, 0, 0x7},
|
||||||
|
"VBITREVH": {0x73184000, false, 0, 15, 0, 0xF},
|
||||||
|
"XVBITREVH": {0x77184000, true, 0, 15, 0, 0xF},
|
||||||
|
"VBITREVW": {0xE631 << 15, false, 0, 31, 0, 0x1F},
|
||||||
|
"XVBITREVW": {0xEE31 << 15, true, 0, 31, 0, 0x1F},
|
||||||
|
"VBITREVV": {0xE632 << 15, false, 0, 63, 0, 0x3F},
|
||||||
|
"XVBITREVV": {0xEE32 << 15, true, 0, 63, 0, 0x3F},
|
||||||
|
// The 4-bit-select shuffles and the byte-extract/insert permutations
|
||||||
|
// take ui8 (the .d shuffle ui4 range-checked to 0..15 by the
|
||||||
|
// toolchain) packing both position nibbles.
|
||||||
|
"VSHUF4IB": {0xE720 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVSHUF4IB": {0xEF20 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSHUF4IH": {0xE728 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVSHUF4IH": {0xEF28 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSHUF4IW": {0xE730 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVSHUF4IW": {0xEF30 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VSHUF4IV": {0xE738 << 15, false, 0, 15, 0, 0xFF},
|
||||||
|
"XVSHUF4IV": {0xEF38 << 15, true, 0, 15, 0, 0xFF},
|
||||||
|
"VPERMIW": {0xE7C8 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVPERMIW": {0xEFC8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"XVPERMIV": {0xEFD0 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"XVPERMIQ": {0xEFD8 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSB": {0xE718 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSB": {0xEF18 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSH": {0xE710 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSH": {0xEF10 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSW": {0xE708 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSW": {0xEF08 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
"VEXTRINSV": {0xE700 << 15, false, 0, 255, 0, 0xFF},
|
||||||
|
"XVEXTRINSV": {0xEF00 << 15, true, 0, 255, 0, 0xFF},
|
||||||
|
}
|
||||||
|
for m, e := range vecImm {
|
||||||
|
l64VecImmInfo[m] = e
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
}
|
||||||
|
|
||||||
|
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
|
||||||
|
// vsetallnez.*): the sub-op rides in the rk field.
|
||||||
|
vecCf := map[string]uint32{
|
||||||
|
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
|
||||||
|
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
|
||||||
|
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
|
||||||
|
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
|
||||||
|
"VSETEQV": 0xE539<<15 | 6<<10, "XVSETEQV": 0xED39<<15 | 6<<10,
|
||||||
|
"VSETANYEQH": 0xE539<<15 | 9<<10, "XVSETANYEQH": 0xED39<<15 | 9<<10,
|
||||||
|
"VSETANYEQW": 0xE539<<15 | 10<<10, "XVSETANYEQW": 0xED39<<15 | 10<<10,
|
||||||
|
"VSETALLNEB": 0xE539<<15 | 12<<10, "XVSETALLNEB": 0xED39<<15 | 12<<10,
|
||||||
|
"VSETALLNEH": 0xE539<<15 | 13<<10, "XVSETALLNEH": 0xED39<<15 | 13<<10,
|
||||||
|
"VSETALLNEW": 0xE539<<15 | 14<<10, "XVSETALLNEW": 0xED39<<15 | 14<<10,
|
||||||
|
}
|
||||||
|
for m, op := range vecCf {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
|
||||||
|
l64VecBank[m] = strings.HasPrefix(m, "XV")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lane popcount and the two-operand vector FP/unary spellings: INSTR vj,
|
||||||
|
// vd (the 2R layout with the opcode extending over the unused vk field;
|
||||||
|
// the low byte of each constant is the instruction's own sub-op).
|
||||||
|
vec2r := map[string]l64Vec3Enc{
|
||||||
|
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
|
||||||
|
}
|
||||||
|
// The rest of the lane popcounts, the vector negations and the vector FP
|
||||||
|
// unary conversions (loong64enc1.s).
|
||||||
|
vec2rMore := map[string]l64Vec3Enc{
|
||||||
|
"VPCNTB": {0x1CA708 << 10, false}, "VPCNTH": {0x1CA709 << 10, false},
|
||||||
|
"VPCNTW": {0x1CA70A << 10, false},
|
||||||
|
"VNEGB": {0x1CA70C << 10, false}, "VNEGH": {0x1CA70D << 10, false},
|
||||||
|
"VNEGW": {0x1CA70E << 10, false}, "VNEGV": {0x1CA70F << 10, false},
|
||||||
|
"VFCLASSF": {0x1CA735 << 10, false}, "VFCLASSD": {0x1CA736 << 10, false},
|
||||||
|
"VFSQRTF": {0x1CA739 << 10, false}, "VFSQRTD": {0x1CA73A << 10, false},
|
||||||
|
"VFRECIPF": {0x1CA73D << 10, false}, "VFRECIPD": {0x1CA73E << 10, false},
|
||||||
|
"VFRSQRTF": {0x1CA741 << 10, false}, "VFRSQRTD": {0x1CA742 << 10, false},
|
||||||
|
"VFRINTF": {0x1CA74D << 10, false}, "VFRINTD": {0x1CA74E << 10, false},
|
||||||
|
"VFRINTRMF": {0x1CA751 << 10, false}, "VFRINTRMD": {0x1CA752 << 10, false},
|
||||||
|
"VFRINTRPF": {0x1CA755 << 10, false}, "VFRINTRPD": {0x1CA756 << 10, false},
|
||||||
|
"VFRINTRZF": {0x1CA759 << 10, false}, "VFRINTRZD": {0x1CA75A << 10, false},
|
||||||
|
"VFRINTRNEF": {0x1CA75D << 10, false}, "VFRINTRNED": {0x1CA75E << 10, false},
|
||||||
|
"XVPCNTB": {0x1DA708 << 10, true}, "XVPCNTH": {0x1DA709 << 10, true},
|
||||||
|
"XVPCNTW": {0x1DA70A << 10, true},
|
||||||
|
"XVNEGB": {0x1DA70C << 10, true}, "XVNEGH": {0x1DA70D << 10, true},
|
||||||
|
"XVNEGW": {0x1DA70E << 10, true}, "XVNEGV": {0x1DA70F << 10, true},
|
||||||
|
"XVFCLASSF": {0x1DA735 << 10, true}, "XVFCLASSD": {0x1DA736 << 10, true},
|
||||||
|
"XVFSQRTF": {0x1DA739 << 10, true}, "XVFSQRTD": {0x1DA73A << 10, true},
|
||||||
|
"XVFRECIPF": {0x1DA73D << 10, true}, "XVFRECIPD": {0x1DA73E << 10, true},
|
||||||
|
"XVFRSQRTF": {0x1DA741 << 10, true}, "XVFRSQRTD": {0x1DA742 << 10, true},
|
||||||
|
"XVFRINTF": {0x1DA74D << 10, true}, "XVFRINTD": {0x1DA74E << 10, true},
|
||||||
|
"XVFRINTRMF": {0x1DA751 << 10, true}, "XVFRINTRMD": {0x1DA752 << 10, true},
|
||||||
|
"XVFRINTRPF": {0x1DA755 << 10, true}, "XVFRINTRPD": {0x1DA756 << 10, true},
|
||||||
|
"XVFRINTRZF": {0x1DA759 << 10, true}, "XVFRINTRZD": {0x1DA75A << 10, true},
|
||||||
|
"XVFRINTRNEF": {0x1DA75D << 10, true}, "XVFRINTRNED": {0x1DA75E << 10, true},
|
||||||
|
}
|
||||||
|
maps.Copy(vec2r, vec2rMore)
|
||||||
|
for m, e := range vec2r {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
l64Vec2R[m] = true
|
||||||
|
}
|
||||||
|
|
||||||
|
// The four-register byte shuffle: INSTR va, vk, vj, vd (the operand the
|
||||||
|
// table reads in each field position, va at bits [19:15]).
|
||||||
|
vec4r := map[string]l64Vec3Enc{
|
||||||
|
"VSHUFB": {0x0D50 << 16, false}, "XVSHUFB": {0x0D60 << 16, true},
|
||||||
|
}
|
||||||
|
for m, e := range vec4r {
|
||||||
|
l64InstrTable[m] = l64Enc{format: l64Fvvvv, op: e.op}
|
||||||
|
l64VecBank[m] = e.lasx
|
||||||
|
l64Vec4R[m] = true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
||||||
|
|||||||
+505
-2
@@ -8,8 +8,8 @@ import (
|
|||||||
"encoding/binary"
|
"encoding/binary"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
|
||||||
@@ -263,6 +263,20 @@ func TestLOONG64_regNames(t *testing.T) {
|
|||||||
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// The X/V spellings name the LSX/LASX vector banks, a register class of
|
||||||
|
// their own: the oracle (GOARCH=loong64 go tool asm) rejects `BEQZ X0`
|
||||||
|
// with "unrecognized instruction" while assembling `VADDV V0, V1, V2`
|
||||||
|
// and `XVADDV X0, X1, X2`, so loong64RegNum stays strict and the vector
|
||||||
|
// operands resolve through loong64VecRegNum only.
|
||||||
|
vecCases := map[string]int{
|
||||||
|
"V0": 0, "V31": 31, "X0": 0, "X31": 31,
|
||||||
|
"R4": -1, "F0": -1, "FCC0": -1, "V32": -1, "X32": -1, "V": -1, "X": -1,
|
||||||
|
}
|
||||||
|
for name, want := range vecCases {
|
||||||
|
if got := loong64VecRegNum(name); got != want {
|
||||||
|
t.Errorf("loong64VecRegNum(%q) = %d, want %d", name, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
func TestLOONG64_bytesEqualGroundTruth(t *testing.T) {
|
||||||
@@ -328,3 +342,492 @@ TEXT ·f(SB), NOSPLIT, $0-0
|
|||||||
0x4C000020, // jirl r0, r1, 0 (RET)
|
0x4C000020, // jirl r0, r1, 0 (RET)
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vector pins the LSX/LASX slice against words read off
|
||||||
|
// GOARCH=loong64 go tool asm (cross-checked against the toolchain's own
|
||||||
|
// loong64enc1.s): the three-register forms, the immediate forms with their
|
||||||
|
// biases, the vector-to-condition forms, lane popcount, the FP conversion,
|
||||||
|
// FSEL and the VMOVQ move family.
|
||||||
|
func TestLOONG64_vector(t *testing.T) {
|
||||||
|
t.Run("three-register and immediate forms", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDV V1, V2, V3
|
||||||
|
VADDW V1, V2, V3
|
||||||
|
VADDV V2, V1
|
||||||
|
VANDV V1, V2
|
||||||
|
VXORV V1, V2, V3
|
||||||
|
VSEQB V1, V2, V3
|
||||||
|
VSEQV V1, V2, V3
|
||||||
|
VSRAB V1, V2, V3
|
||||||
|
VROTRW V1, V2, V3
|
||||||
|
VANDB $0, V2, V3
|
||||||
|
VANDB $255, V2
|
||||||
|
VSEQB $3, V2, V3
|
||||||
|
VSEQV $15, V2, V3
|
||||||
|
VSEQV $-15, V2, V3
|
||||||
|
VSRAB $7, V1, V2
|
||||||
|
VROTRW $16, V1, V2
|
||||||
|
VPCNTV V1, V2
|
||||||
|
XVADDV X1, X2, X3
|
||||||
|
XVXORV X1, X2, X3
|
||||||
|
XVSEQB X1, X2, X3
|
||||||
|
XVPCNTV X1, X2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x700B8443, // vadd.v v3, v2, v1
|
||||||
|
0x700B0443, // vadd.w
|
||||||
|
0x700B8821, // vadd.v v1, v1, v2 (two-operand form)
|
||||||
|
0x71260442, // vand.v v2, v2, v1
|
||||||
|
0x71270443, // vxor.v
|
||||||
|
0x70000443, // vseq.b
|
||||||
|
0x70018443, // vseq.d
|
||||||
|
0x70EC0443, // vsra.b
|
||||||
|
0x70EF0443, // vrotr.w
|
||||||
|
0x73D00043, // vandi.b v3, v2, 0
|
||||||
|
0x73D3FC42, // vandi.b v2, v2, 255 (two-operand form)
|
||||||
|
0x72800C43, // vseqi.b v3, v2, 3
|
||||||
|
0x7281BC43, // vseqi.d v3, v2, 15
|
||||||
|
0x7281C443, // vseqi.d v3, v2, -15 (7-bit two's complement)
|
||||||
|
0x73343C22, // vsrai.b v2, v1, 7 (encoded as 7+8)
|
||||||
|
0x72A0C022, // vrotri.w v2, v1, 16
|
||||||
|
0x729C2C22, // vpcnt.d v2, v1
|
||||||
|
0x740B8443, // xvadd.d x3, x2, x1
|
||||||
|
0x75270443, // xvxor.d
|
||||||
|
0x74000443, // xvseq.b
|
||||||
|
0x769C2C22, // xvpcnt.d x2, x1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("vector-to-condition", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSETNEV V1, FCC0
|
||||||
|
VSETANYEQB V1, FCC0
|
||||||
|
VSETANYEQV V2, FCC0
|
||||||
|
VSETALLNEV V0, FCC0
|
||||||
|
XVSETNEV X1, FCC0
|
||||||
|
XVSETALLNEV X1, FCC0
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x729C9C20, // vsetnez.d fcc0, v1
|
||||||
|
0x729CA020, // vsetanyeqz.b
|
||||||
|
0x729CAC40, // vsetanyeqz.d
|
||||||
|
0x729CBC00, // vsetallnez.d
|
||||||
|
0x769C9C20, // xvsetnez.d
|
||||||
|
0x769CBC20, // xvsetallnez.d
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("FP convert and FSEL", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
FFINTDV F0, F1
|
||||||
|
FSEL FCC0, F3, F4, F3
|
||||||
|
FSEL FCC1, F1, F2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x011D2801, // ffint.d.v f1, f0
|
||||||
|
0x0D000C83, // fsel f3, f4, f3, fcc0
|
||||||
|
0x0D008442, // fsel f2, f2, f1, fcc1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("VMOVQ move family", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V1, V9
|
||||||
|
VMOVQ (R4), V2
|
||||||
|
VMOVQ 16(R4), V2
|
||||||
|
VMOVQ V0, (R4)
|
||||||
|
VMOVQ V0, 32(R4)
|
||||||
|
VMOVQ (R4)(R7), V3
|
||||||
|
VMOVQ V3, (R4)(R7)
|
||||||
|
VMOVQ R6, V0.B16
|
||||||
|
VMOVQ R6, V12.W4
|
||||||
|
VMOVQ (R4), V4.W4
|
||||||
|
XVMOVQ X3, X7
|
||||||
|
XVMOVQ (R4), X2
|
||||||
|
XVMOVQ X0, (R4)
|
||||||
|
XVMOVQ (R4)(R7), X4
|
||||||
|
XVMOVQ X0, (R4)(R7)
|
||||||
|
XVMOVQ R6, X0.B32
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x732D0029, // vori.b v9, v1, 0 (register move)
|
||||||
|
0x2C000082, // vld v2, r4, 0
|
||||||
|
0x2C004082, // vld v2, r4, 16
|
||||||
|
0x2C400080, // vst v0, r4, 0
|
||||||
|
0x2C408080, // vst v0, r4, 32
|
||||||
|
0x38401C83, // vldx v3, r4, r7
|
||||||
|
0x38441C83, // vstx v3, r4, r7
|
||||||
|
0x729F00C0, // vreplgr2vr.b v0, r6
|
||||||
|
0x729F08CC, // vreplgr2vr.w v12, r6
|
||||||
|
0x30200084, // vldrepl.w v4, r4, 0
|
||||||
|
0x772D0067, // xvori.b x7, x3, 0
|
||||||
|
0x2C800082, // xvld x2, r4, 0
|
||||||
|
0x2CC00080, // xvst x0, r4, 0
|
||||||
|
0x38481C84, // xvldx x4, r4, r7
|
||||||
|
0x384C1C80, // xvstx x0, r4, r7
|
||||||
|
0x769F00C0, // xvreplgr2vr.b x0, r6
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("element extract and insert", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMOVQ V0.V[0], R10
|
||||||
|
VMOVQ V6.V[1], R8
|
||||||
|
VMOVQ R9, V1.V[0]
|
||||||
|
XVMOVQ X0.V[0], R10
|
||||||
|
XVMOVQ X5.W[7], R7
|
||||||
|
XVMOVQ R4, X7.V[3]
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x72EFF00A, // vpickve2gr.d r10, v0, 0
|
||||||
|
0x72EFF4C8, // vpickve2gr.d r8, v6, 1
|
||||||
|
0x72EBF121, // vinsgr2vr.d v1, r9, 0
|
||||||
|
0x76EFE00A, // xvpickve2gr.d r10, x0, 0
|
||||||
|
0x76EFDCA7, // xvpickve2gr.w r7, x5, 7
|
||||||
|
0x76EBEC87, // xvinsgr2vr.d x7, r4, 3
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The integer and FP add/subtract families with their saturating pairs
|
||||||
|
// and immediate spellings (loong64enc1.s words).
|
||||||
|
t.Run("add and subtract families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDB V1, V2, V3
|
||||||
|
VADDF V1, V2, V3
|
||||||
|
VADDD V1, V2, V3
|
||||||
|
VSUBD V1, V2, V3
|
||||||
|
VSADDV V1, V2, V3
|
||||||
|
VSSUBVU V1, V2, V3
|
||||||
|
VADDBU $1, V2, V1
|
||||||
|
VADDBU $1, V2
|
||||||
|
VSUBVU $31, V2
|
||||||
|
XVSADDV X3, X2, X1
|
||||||
|
XVSUBD X1, X2, X3
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x700A0443, // vadd.b
|
||||||
|
0x71308443, // vadd.f
|
||||||
|
0x71310443, // vadd.d
|
||||||
|
0x71330443, // vsub.d
|
||||||
|
0x70478443, // vsadd.v
|
||||||
|
0x704D8443, // vssub.u.d
|
||||||
|
0x728A0441, // vaddi.bu v1, v2, 1
|
||||||
|
0x728A0442, // vaddi.bu v2, v2, 1 (two-operand form)
|
||||||
|
0x728DFC42, // vsubi.du v2, v2, 31 (two-operand form)
|
||||||
|
0x74478C41, // xvsadd.d x1, x2, x3
|
||||||
|
0x75330443, // xvsub.d x3, x2, x1
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The multiply, divide and accumulate families.
|
||||||
|
t.Run("multiply and divide families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VMULV V1, V2, V3
|
||||||
|
VMUHHU V1, V2, V3
|
||||||
|
VDIVBU V1, V2, V3
|
||||||
|
VMODV V1, V2, V3
|
||||||
|
VMADDB V1, V2, V3
|
||||||
|
VMSUBV V1, V2, V3
|
||||||
|
VMULWEVHB V1, V2, V3
|
||||||
|
VMULWODQV V1, V2, V3
|
||||||
|
VMADDWEVHBUB V1, V2, V3
|
||||||
|
XVDIVD X1, X2, X3
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x70858443, // vmul.v
|
||||||
|
0x70888443, // vmuh.u.d
|
||||||
|
0x70E40443, // vdiv.u.b
|
||||||
|
0x70E38443, // vmod.d
|
||||||
|
0x70A80443, // vmadd.b
|
||||||
|
0x70AB8443, // vmsub.d
|
||||||
|
0x70900443, // vmulwev.h.b
|
||||||
|
0x70938443, // vmulwod.q.d
|
||||||
|
0x70BC0443, // vmaddwev.h.bu.b
|
||||||
|
0x753B0443, // xvdiv.d
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The shift, bit and interleave families in register and immediate
|
||||||
|
// spellings, with the width-coded shift immediates.
|
||||||
|
t.Run("shift, bit and interleave families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSLLV V1, V2, V3
|
||||||
|
VROTRB V1, V2, V3
|
||||||
|
VBITCLRV V1, V2, V3
|
||||||
|
VBITSETW V1, V2, V3
|
||||||
|
VBITREVV V1, V2, V3
|
||||||
|
VILVLB V1, V2, V3
|
||||||
|
VILVHV V1, V2, V3
|
||||||
|
VSLLB $7, V1, V2
|
||||||
|
VSLLB $5, V1
|
||||||
|
VSRLH $15, V1, V2
|
||||||
|
VSRAW $31, V1, V2
|
||||||
|
VSRAV $63, V1, V2
|
||||||
|
VROTRV $63, V1, V2
|
||||||
|
VBITCLRB $7, V2, V3
|
||||||
|
VBITREVV $63, V2, V3
|
||||||
|
VSEQH $-16, V2, V3
|
||||||
|
VSLTB $1, V2, V3
|
||||||
|
VSLTHU $31, V2, V3
|
||||||
|
XVILVLV X3, X2, X1
|
||||||
|
XVSLLB $7, X2, X1
|
||||||
|
XVSRAV $63, X2, X1
|
||||||
|
XVBITREVV $63, X2, X1
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x70E98443, // vsll.d
|
||||||
|
0x70EE0443, // vrotr.b
|
||||||
|
0x710D8443, // vbitclr.d
|
||||||
|
0x710F0443, // vbitset.w
|
||||||
|
0x71118443, // vbitrev.d
|
||||||
|
0x711A0443, // vilvl.b
|
||||||
|
0x711D8443, // vilvh.d
|
||||||
|
0x732C3C22, // vslli.b v2, v1, 7
|
||||||
|
0x732C3421, // vslli.b v1, v1, 5 (two-operand form)
|
||||||
|
0x73307C22, // vsrli.h v2, v1, 15
|
||||||
|
0x7334FC22, // vsrai.w v2, v1, 31
|
||||||
|
0x7335FC22, // vsrai.d v2, v1, 63
|
||||||
|
0x72A1FC22, // vrotri.d v2, v1, 63
|
||||||
|
0x73103C43, // vbitclri.b v3, v2, 7
|
||||||
|
0x7319FC43, // vbitrevi.d v3, v2, 63
|
||||||
|
0x7280C043, // vseqi.h v3, v2, -16
|
||||||
|
0x72860443, // vslti.b v3, v2, 1
|
||||||
|
0x7288FC43, // vslti.hu v3, v2, 31
|
||||||
|
0x751B8C41, // xvilvl.d x1, x2, x3
|
||||||
|
0x772C3C41, // xvslli.b x1, x2, 7
|
||||||
|
0x7735FC41, // xvsrai.d x1, x2, 63
|
||||||
|
0x7719FC41, // xvbitrevi.d x1, x2, 63
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The shuffle, select and permutation families, including the
|
||||||
|
// four-register byte shuffle.
|
||||||
|
t.Run("shuffle and permutation families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSHUFH V1, V2, V3
|
||||||
|
VSHUFW V1, V2, V3
|
||||||
|
VSHUFV V1, V2, V3
|
||||||
|
VSHUFB V1, V2, V3, V4
|
||||||
|
XVSHUFB X1, X2, X3, X4
|
||||||
|
VSHUF4IB $255, V2, V1
|
||||||
|
VSHUF4IV $15, V2, V1
|
||||||
|
XVSHUF4IV $15, X1, X2
|
||||||
|
VEXTRINSB $0x18, V1, V2
|
||||||
|
XVEXTRINSV $0x81, X1, X2
|
||||||
|
VPERMIW $0x1B, V1, V2
|
||||||
|
XVPERMIQ $0x4B, X1, X2
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x717A8443, // vshuf.h
|
||||||
|
0x717B0443, // vshuf.w
|
||||||
|
0x717B8443, // vshuf.d
|
||||||
|
0x0D508864, // vshuf.b v4, v3, v2, v1
|
||||||
|
0x0D608864, // xvshuf.b
|
||||||
|
0x7393FC41, // vshuf4i.b v1, v2, 255
|
||||||
|
0x739C3C41, // vshuf4i.d v1, v2, 15
|
||||||
|
0x779C3C22, // xvshuf4i.d x2, x1, 15
|
||||||
|
0x738C6022, // vextrins.b v2, v1, 0x18
|
||||||
|
0x77820422, // xvextrins.d x2, x1, 0x81
|
||||||
|
0x73E46C22, // vpermi.w v2, v1, 0x1b
|
||||||
|
0x77ED2C22, // xvpermi.q x2, x1, 0x4b
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
// The vector FP families, the unary spellings, the compare-to-flag
|
||||||
|
// additions and the scalar int/float conversions.
|
||||||
|
t.Run("FP and conversion families", func(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VADDF V1, V2, V3
|
||||||
|
VMULF V1, V2, V3
|
||||||
|
VFCLASSD V1, V2
|
||||||
|
VFSQRTF V1, V2
|
||||||
|
VFRECIPD V1, V2
|
||||||
|
VFRSQRTF V1, V2
|
||||||
|
VFRINTF V1, V2
|
||||||
|
VFRINTRNED V1, V2
|
||||||
|
VNEGB V1, V2
|
||||||
|
VPCNTB V1, V2
|
||||||
|
XVNEGV X2, X1
|
||||||
|
XVPCNTW X3, X2
|
||||||
|
XVFRINTRNEF X1, X2
|
||||||
|
VSETEQV V1, FCC0
|
||||||
|
VSETANYEQH V1, FCC0
|
||||||
|
VSETALLNEB V1, FCC0
|
||||||
|
XVSETALLNEW X1, FCC0
|
||||||
|
FFINTFW F0, F1
|
||||||
|
FTINTVD F0, F1
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x71308443, // vfadd.s
|
||||||
|
0x71388443, // vfmul.s
|
||||||
|
0x729CD822, // vfclass.d
|
||||||
|
0x729CE422, // vfsqrt.s
|
||||||
|
0x729CF822, // vfrecip.d
|
||||||
|
0x729D0422, // vfrsqrt.s
|
||||||
|
0x729D3422, // vfrint.s
|
||||||
|
0x729D7822, // vfrintne.s
|
||||||
|
0x729C3022, // vneg.b
|
||||||
|
0x729C2022, // vpcnt.b
|
||||||
|
0x769C3C41, // xvneg.d x1, x2
|
||||||
|
0x769C2862, // xvpcnt.w x2, x3
|
||||||
|
0x769D7422, // xvfrintne.s x2, x1
|
||||||
|
0x729C9820, // vseteqz.d fcc0, v1
|
||||||
|
0x729CA420, // vsetanyeqz.h
|
||||||
|
0x729CB020, // vsetallnez.b
|
||||||
|
0x769CB820, // xvsetallnez.w
|
||||||
|
0x011D1001, // ffint.s.w f1, f0
|
||||||
|
0x011B2801, // ftint.l.d f1, f0
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_vectorErrors pins the register-class and range diagnostics of
|
||||||
|
// the vector slice; each shape is rejected by the oracle as well
|
||||||
|
// (GOARCH=loong64 go tool asm).
|
||||||
|
func TestLOONG64_vectorErrors(t *testing.T) {
|
||||||
|
cases := []string{
|
||||||
|
// Integer registers in vector positions.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDV R4, R5, R6
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// Crossed banks: LSX spellings take V, LASX spellings X.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDV X1, X2, X3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVADDV V1, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// The LASX bank has no .b/.h element forms.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
XVMOVQ R4, X2.B[0]
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// Immediate ranges.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VANDB $256, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSEQB $16, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VROTRW $32, V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VADDVU $32, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSEQV $32, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSHUF4IV $16, V2, V1
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VEXTRINSB $256, V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSLTV $-17, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// VSHUFB wants four vector registers.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSHUFB V1, V2, V3
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// The FCC forms still refuse vector registers.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSETEQV V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
// VSET* wants an FCC flag, not a vector register.
|
||||||
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
VSETNEV V1, V2
|
||||||
|
RET
|
||||||
|
`,
|
||||||
|
}
|
||||||
|
for i, src := range cases {
|
||||||
|
fn := firstTextLOONG64(t, src)
|
||||||
|
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
|
||||||
|
t.Errorf("case %d: expected an error, got none", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLOONG64_dbarAtomics pins the _dbar (acquire/release) AMO variants.
|
||||||
|
// The oracle words come from GOARCH=loong64 go tool objdump of kernels
|
||||||
|
// assembled with go tool asm, and match the toolchain's loong64enc1.s.
|
||||||
|
func TestLOONG64_dbarAtomics(t *testing.T) {
|
||||||
|
fn := firstTextLOONG64(t, `#include "textflag.h"
|
||||||
|
TEXT ·atoms(SB), NOSPLIT, $0
|
||||||
|
AMADDDBW R14, (R13), R12
|
||||||
|
AMADDDBV R14, (R13), R12
|
||||||
|
AMANDDBW R5, (R4), R6
|
||||||
|
AMANDDBV R5, (R4), R6
|
||||||
|
AMORDBW R5, (R4), R0
|
||||||
|
AMORDBV R5, (R4), R6
|
||||||
|
AMSWAPDBW R5, (R4), R6
|
||||||
|
AMCASDBV R6, (R4), R5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleLOONG64Helper(t, fn)
|
||||||
|
wantWords(t, code,
|
||||||
|
0x386A39AC, // amadd_db.w r12, r13, r14
|
||||||
|
0x386AB9AC, // amadd_db.d
|
||||||
|
0x386B1486, // amand_db.w r6, r4, r5
|
||||||
|
0x386B9486, // amand_db.d
|
||||||
|
0x386C1480, // amor_db.w r0, r4, r5
|
||||||
|
0x386C9486, // amor_db.d
|
||||||
|
0x38691486, // amswap_db.w
|
||||||
|
0x385B9885, // amcas_db.w
|
||||||
|
0x4C000020,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ import (
|
|||||||
"bytes"
|
"bytes"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestLOONG64_sys exercises the no-operand system instructions and the
|
// TestLOONG64_sys exercises the no-operand system instructions and the
|
||||||
@@ -232,7 +232,10 @@ DATA ·table+0(SB)/8, $42
|
|||||||
}
|
}
|
||||||
|
|
||||||
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
// TestLOONG64_errors checks the encoder's error paths: undefined labels,
|
||||||
// invalid register operands and operand-count mismatches.
|
// invalid register operands and operand-count mismatches. The X0 and
|
||||||
|
// AMADDW cases follow the oracle: GOARCH=loong64 go tool asm rejects
|
||||||
|
// `BEQZ X0` (the X bank is not an integer register) and the two-register
|
||||||
|
// `AMADDW R4, R5` (the AM* family is strictly `val, (addr), result`).
|
||||||
func TestLOONG64_errors(t *testing.T) {
|
func TestLOONG64_errors(t *testing.T) {
|
||||||
cases := []string{
|
cases := []string{
|
||||||
`TEXT ·e(SB), NOSPLIT, $0
|
`TEXT ·e(SB), NOSPLIT, $0
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
|
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
|
||||||
|
|||||||
@@ -14,6 +14,51 @@ type Imm int64
|
|||||||
|
|
||||||
func (Imm) isOperand() {}
|
func (Imm) isOperand() {}
|
||||||
|
|
||||||
|
// RegList is a bracketed register range, [Z0-Z3]: the four-register source
|
||||||
|
// of the 4FMAPS and 4VNNIW families. The EVEX emit path carries the list's
|
||||||
|
// low register through the inverted 5-bit V'VVVV field; the three higher
|
||||||
|
// registers are implied by the instruction, so only the pair travels here.
|
||||||
|
type RegList struct {
|
||||||
|
Lo Reg
|
||||||
|
Hi Reg // implied by the encoding; Lo.idx+3 by construction
|
||||||
|
}
|
||||||
|
|
||||||
|
func (RegList) isOperand() {}
|
||||||
|
|
||||||
|
// FloatImm is a floating-point immediate ($-1.0). The SSE mnemonics whose
|
||||||
|
// encoding takes an XMM/memory source at that position rewrite it as a read
|
||||||
|
// from a read-only pool constant ($f64.<hex> or $f32.<hex>), the toolchain's
|
||||||
|
// own behaviour; every other instruction rejects it.
|
||||||
|
type FloatImm struct {
|
||||||
|
Text string // the numeric text as written, sign excluded
|
||||||
|
Neg bool // a leading minus
|
||||||
|
}
|
||||||
|
|
||||||
|
func (FloatImm) isOperand() {}
|
||||||
|
|
||||||
|
// TLSMem is a thread-local access, the source form off(base)(TLS*1) with the
|
||||||
|
// base dropped: the toolchain's one-instruction TLS rewrite assembles it as
|
||||||
|
// the segment-prefixed absolute whose disp32 carries an R_TLS_LE patch site
|
||||||
|
// (the linker fills the TLS slot offset).
|
||||||
|
type TLSMem struct {
|
||||||
|
Disp int64
|
||||||
|
Size int
|
||||||
|
Seg byte // the segment override: FS (0x64) or GS (0x65) on windows
|
||||||
|
}
|
||||||
|
|
||||||
|
func (TLSMem) isOperand() {}
|
||||||
|
|
||||||
|
// SegAbs is a segment-absolute access, 0x30(GS): the segment override
|
||||||
|
// prefixes a disp32 absolute reference with no relocation. The base
|
||||||
|
// register spellings GS and FS produce it.
|
||||||
|
type SegAbs struct {
|
||||||
|
Disp int64
|
||||||
|
Size int
|
||||||
|
Seg byte // 0x64 FS, 0x65 GS
|
||||||
|
}
|
||||||
|
|
||||||
|
func (SegAbs) isOperand() {}
|
||||||
|
|
||||||
// Mem is a memory operand of the form disp(base)(index*scale).
|
// Mem is a memory operand of the form disp(base)(index*scale).
|
||||||
type Mem struct {
|
type Mem struct {
|
||||||
Base Reg
|
Base Reg
|
||||||
@@ -23,6 +68,7 @@ type Mem struct {
|
|||||||
Size int // operand width in bytes
|
Size int // operand width in bytes
|
||||||
HasBase bool
|
HasBase bool
|
||||||
HasIndex bool
|
HasIndex bool
|
||||||
|
Seg byte // segment override prefix (0x64 FS, 0x65 GS); 0 = none
|
||||||
}
|
}
|
||||||
|
|
||||||
func (Mem) isOperand() {}
|
func (Mem) isOperand() {}
|
||||||
@@ -48,3 +94,16 @@ type sbMem struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (sbMem) isOperand() {}
|
func (sbMem) isOperand() {}
|
||||||
|
|
||||||
|
// isX86Mem reports whether the operand is an amd64 memory reference: a base
|
||||||
|
// or indexed Mem, or an SB-relative sbMem. Encoders that gate on "memory in
|
||||||
|
// this position" must accept both; the r/m emitters distinguish the two
|
||||||
|
// themselves.
|
||||||
|
func isX86Mem(o Operand) bool {
|
||||||
|
switch o.(type) {
|
||||||
|
case Mem, sbMem:
|
||||||
|
return true
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+6
-1
@@ -17,12 +17,13 @@ import "strings"
|
|||||||
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
|
||||||
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
|
||||||
// those indices but require one. The mask flag marks the AVX-512 opmask
|
// those indices but require one. The mask flag marks the AVX-512 opmask
|
||||||
// registers K0-K7.
|
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
|
||||||
type Reg struct {
|
type Reg struct {
|
||||||
idx int
|
idx int
|
||||||
size int // informational width implied by the name; the mnemonic decides
|
size int // informational width implied by the name; the mnemonic decides
|
||||||
high bool // AH/CH/DH/BH
|
high bool // AH/CH/DH/BH
|
||||||
mask bool // K0-K7 opmask register
|
mask bool // K0-K7 opmask register
|
||||||
|
fp bool // F0-F7 x87 stack register
|
||||||
}
|
}
|
||||||
|
|
||||||
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
|
||||||
@@ -144,6 +145,10 @@ func buildRegByName() map[string]Reg {
|
|||||||
for i := 0; i <= 7; i++ {
|
for i := 0; i <= 7; i++ {
|
||||||
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
m["K"+itoa(i)] = Reg{idx: i, size: 8, mask: true}
|
||||||
}
|
}
|
||||||
|
// x87 stack: F0..F7.
|
||||||
|
for i := 0; i <= 7; i++ {
|
||||||
|
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
|
||||||
|
}
|
||||||
return m
|
return m
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+1395
-61
File diff suppressed because it is too large
Load Diff
+151
-31
@@ -63,7 +63,7 @@ func riscvRegNum(name string) int {
|
|||||||
return 24
|
return 24
|
||||||
case "X25", "S9":
|
case "X25", "S9":
|
||||||
return 25
|
return 25
|
||||||
case "X26", "S10":
|
case "X26", "S10", "CTXT":
|
||||||
return 26
|
return 26
|
||||||
case "X27", "S11", "g":
|
case "X27", "S11", "g":
|
||||||
return 27
|
return 27
|
||||||
@@ -141,10 +141,36 @@ func riscvRegNum(name string) int {
|
|||||||
case "F31", "FT11":
|
case "F31", "FT11":
|
||||||
return 31
|
return 31
|
||||||
default:
|
default:
|
||||||
|
// Vector registers V0-V31 (the "V" extension). They share the
|
||||||
|
// register numbering with the integer file: a bare number 0-31.
|
||||||
|
if len(name) >= 2 && name[0] == 'V' {
|
||||||
|
if n, ok := parseRegDigits(name[1:], 31); ok {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
}
|
||||||
return -1
|
return -1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// parseRegDigits parses a decimal register suffix and reports whether it is
|
||||||
|
// within [0, max].
|
||||||
|
func parseRegDigits(digits string, max int) (int, bool) {
|
||||||
|
if digits == "" {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for i := 0; i < len(digits); i++ {
|
||||||
|
if digits[i] < '0' || digits[i] > '9' {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
n = n*10 + int(digits[i]-'0')
|
||||||
|
if n > max {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n, true
|
||||||
|
}
|
||||||
|
|
||||||
// RISC-V instruction encoding parameters.
|
// RISC-V instruction encoding parameters.
|
||||||
type riscvEnc struct {
|
type riscvEnc struct {
|
||||||
opcode uint32 // bits [6:0]
|
opcode uint32 // bits [6:0]
|
||||||
@@ -193,6 +219,9 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"DIVUW": {0x3B, 0x5, 0x01},
|
"DIVUW": {0x3B, 0x5, 0x01},
|
||||||
"REMW": {0x3B, 0x6, 0x01},
|
"REMW": {0x3B, 0x6, 0x01},
|
||||||
"REMUW": {0x3B, 0x7, 0x01},
|
"REMUW": {0x3B, 0x7, 0x01},
|
||||||
|
// Zicond conditional zeroing.
|
||||||
|
"CZEROEQZ": {0x33, 0x5, 0x07},
|
||||||
|
"CZERONEZ": {0x33, 0x7, 0x07},
|
||||||
// RV64I, I-type arithmetic.
|
// RV64I, I-type arithmetic.
|
||||||
"ADDI": {0x13, 0x0, 0x00},
|
"ADDI": {0x13, 0x0, 0x00},
|
||||||
"ADDIW": {0x1B, 0x0, 0x00},
|
"ADDIW": {0x1B, 0x0, 0x00},
|
||||||
@@ -221,36 +250,47 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"BGE": {0x63, 0x5, 0x00},
|
"BGE": {0x63, 0x5, 0x00},
|
||||||
"BLTU": {0x63, 0x6, 0x00},
|
"BLTU": {0x63, 0x6, 0x00},
|
||||||
"BGEU": {0x63, 0x7, 0x00},
|
"BGEU": {0x63, 0x7, 0x00},
|
||||||
|
// The swapped-spelling comparison forms: encoded as BLT/BGE/BLTU/BGEU
|
||||||
|
// with the register operands swapped.
|
||||||
|
"BGT": {0x63, 0x4, 0x00},
|
||||||
|
"BLE": {0x63, 0x5, 0x00},
|
||||||
|
"BGTU": {0x63, 0x6, 0x00},
|
||||||
|
"BLEU": {0x63, 0x7, 0x00},
|
||||||
// U-type.
|
// U-type.
|
||||||
"LUI": {0x37, 0x0, 0x00},
|
"LUI": {0x37, 0x0, 0x00},
|
||||||
"AUIPC": {0x17, 0x0, 0x00},
|
"AUIPC": {0x17, 0x0, 0x00},
|
||||||
// System.
|
// System.
|
||||||
"ECALL": {0x73, 0x0, 0x00},
|
"ECALL": {0x73, 0x0, 0x00},
|
||||||
"EBREAK": {0x73, 0x0, 0x00},
|
"EBREAK": {0x73, 0x0, 0x00},
|
||||||
"FENCE": {0x0F, 0x0, 0x00},
|
"FENCE": {0x0F, 0x0, 0x00},
|
||||||
|
"FENCE.TSO": {0x0F, 0x0, 0x00},
|
||||||
|
"PAUSE": {0x0F, 0x0, 0x00},
|
||||||
// JALR, indirect jump/call (I-type).
|
// JALR, indirect jump/call (I-type).
|
||||||
"JALR": {0x67, 0x0, 0x00},
|
"JALR": {0x67, 0x0, 0x00},
|
||||||
|
|
||||||
// RV64A, atomics (AMO opcode 0x2F).
|
// RV64A, atomics (AMO opcode 0x2F).
|
||||||
// funct3: 0x2 = word, 0x3 = doubleword. funct5 in bits [31:27].
|
// funct3: 0x2 = word, 0x3 = doubleword. The stored funct7 is the full
|
||||||
"AMOSWAPW": {0x2F, 0x2, 0x01 << 2},
|
// 7-bit field: funct5 in the upper five bits and the aq/rl ordering bits in
|
||||||
"AMOSWAPD": {0x2F, 0x3, 0x01 << 2},
|
// the lower two, exactly as the toolchain writes them: every AMO sets both
|
||||||
"AMOADDW": {0x2F, 0x2, 0x00 << 2},
|
// aq and rl (funct7 |= 3).
|
||||||
"AMOADDD": {0x2F, 0x3, 0x00 << 2},
|
"AMOSWAPW": {0x2F, 0x2, 0x01<<2 | 0x3},
|
||||||
"AMOANDW": {0x2F, 0x2, 0x0C << 2},
|
"AMOSWAPD": {0x2F, 0x3, 0x01<<2 | 0x3},
|
||||||
"AMOANDD": {0x2F, 0x3, 0x0C << 2},
|
"AMOADDW": {0x2F, 0x2, 0x00<<2 | 0x3},
|
||||||
"AMOORW": {0x2F, 0x2, 0x06 << 2},
|
"AMOADDD": {0x2F, 0x3, 0x00<<2 | 0x3},
|
||||||
"AMOORD": {0x2F, 0x3, 0x06 << 2},
|
"AMOANDW": {0x2F, 0x2, 0x0C<<2 | 0x3},
|
||||||
"AMOXORW": {0x2F, 0x2, 0x04 << 2},
|
"AMOANDD": {0x2F, 0x3, 0x0C<<2 | 0x3},
|
||||||
"AMOXORD": {0x2F, 0x3, 0x04 << 2},
|
"AMOORW": {0x2F, 0x2, 0x08<<2 | 0x3},
|
||||||
"AMOMAXW": {0x2F, 0x2, 0x14 << 2},
|
"AMOORD": {0x2F, 0x3, 0x08<<2 | 0x3},
|
||||||
"AMOMAXD": {0x2F, 0x3, 0x14 << 2},
|
"AMOXORW": {0x2F, 0x2, 0x04<<2 | 0x3},
|
||||||
"AMOMINW": {0x2F, 0x2, 0x10 << 2},
|
"AMOXORD": {0x2F, 0x3, 0x04<<2 | 0x3},
|
||||||
"AMOMIND": {0x2F, 0x3, 0x10 << 2},
|
"AMOMAXW": {0x2F, 0x2, 0x14<<2 | 0x3},
|
||||||
"AMOMAXUW": {0x2F, 0x2, 0x1C << 2},
|
"AMOMAXD": {0x2F, 0x3, 0x14<<2 | 0x3},
|
||||||
"AMOMAXUD": {0x2F, 0x3, 0x1C << 2},
|
"AMOMINW": {0x2F, 0x2, 0x10<<2 | 0x3},
|
||||||
"AMOMINUW": {0x2F, 0x2, 0x18 << 2},
|
"AMOMIND": {0x2F, 0x3, 0x10<<2 | 0x3},
|
||||||
"AMOMINUD": {0x2F, 0x3, 0x18 << 2},
|
"AMOMAXUW": {0x2F, 0x2, 0x1C<<2 | 0x3},
|
||||||
|
"AMOMAXUD": {0x2F, 0x3, 0x1C<<2 | 0x3},
|
||||||
|
"AMOMINUW": {0x2F, 0x2, 0x18<<2 | 0x3},
|
||||||
|
"AMOMINUD": {0x2F, 0x3, 0x18<<2 | 0x3},
|
||||||
|
|
||||||
// RV64F/D, floating-point arithmetic.
|
// RV64F/D, floating-point arithmetic.
|
||||||
"FADDS": {0x53, 0x0, 0x00},
|
"FADDS": {0x53, 0x0, 0x00},
|
||||||
@@ -273,12 +313,23 @@ var riscvInstrTable = map[string]riscvEnc{
|
|||||||
"FMAXS": {0x53, 0x1, 0x14},
|
"FMAXS": {0x53, 0x1, 0x14},
|
||||||
"FMIND": {0x53, 0x0, 0x15},
|
"FMIND": {0x53, 0x0, 0x15},
|
||||||
"FMAXD": {0x53, 0x1, 0x15},
|
"FMAXD": {0x53, 0x1, 0x15},
|
||||||
|
// FP sign injection (double): rs2 carries the sign source.
|
||||||
|
"FSGNJD": {0x53, 0x0, 0x11},
|
||||||
|
"FSGNJS": {0x53, 0x0, 0x10},
|
||||||
|
"FSGNJX": {0x53, 0x0, 0x14},
|
||||||
|
"FSGNJXD": {0x53, 0x0, 0x15},
|
||||||
|
"FSGNJXS": {0x53, 0x0, 0x14},
|
||||||
|
"FSGNJND": {0x53, 0x1, 0x11},
|
||||||
|
"FSGNJNS": {0x53, 0x1, 0x10},
|
||||||
|
"FSGNJNX": {0x53, 0x1, 0x14},
|
||||||
|
|
||||||
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
// RV64A, load-reserved / store-conditional (funct5 0x02 / 0x03).
|
||||||
"LRW": {0x2F, 0x2, 0x02 << 2},
|
// The toolchain gives LR acquire ordering (aq = 1) and SC release
|
||||||
"LRD": {0x2F, 0x3, 0x02 << 2},
|
// ordering (rl = 1).
|
||||||
"SCW": {0x2F, 0x2, 0x03 << 2},
|
"LRW": {0x2F, 0x2, 0x02<<2 | 0x2},
|
||||||
"SCD": {0x2F, 0x3, 0x03 << 2},
|
"LRD": {0x2F, 0x3, 0x02<<2 | 0x2},
|
||||||
|
"SCW": {0x2F, 0x2, 0x03<<2 | 0x1},
|
||||||
|
"SCD": {0x2F, 0x3, 0x03<<2 | 0x1},
|
||||||
|
|
||||||
// FP compare, result in integer register (funct7 0x50/0x51).
|
// FP compare, result in integer register (funct7 0x50/0x51).
|
||||||
"FEQS": {0x53, 0x2, 0x50},
|
"FEQS": {0x53, 0x2, 0x50},
|
||||||
@@ -296,11 +347,11 @@ func riscvRType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// riscvAMOType encodes an atomic (AMO) instruction.
|
// riscvAMOType encodes an atomic (AMO) instruction.
|
||||||
// Layout: funct5 | aq | rl | rs2 | rs1 | funct3 | rd | opcode.
|
// Layout: funct7 | rs2 | rs1 | funct3 | rd | opcode, where funct7 carries the
|
||||||
// The funct5 is stored in the upper bits of enc.funct7 (shifted left by 2).
|
// funct5 in its upper five bits and the aq/rl ordering bits in the lower two
|
||||||
|
// (the table stores the full field, so the word needs no reassembly).
|
||||||
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
func riscvAMOType(enc riscvEnc, rd, rs1, rs2 int) uint32 {
|
||||||
funct5 := enc.funct7 >> 2 // extract funct5 from the stored value
|
return (enc.funct7 << 25) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
||||||
return (funct5 << 27) | (uint32(rs2) << 20) | (uint32(rs1) << 15) |
|
|
||||||
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
(enc.funct3 << 12) | (uint32(rd) << 7) | enc.opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -342,6 +393,10 @@ var riscvCvtTable = map[string]riscvCvtEnc{
|
|||||||
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
"FMVDX": {0x79, 0x0, 0x53}, // int64 → float64 (bit move)
|
||||||
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
"FMVXW": {0x70, 0x0, 0x53}, // float32 → int32 (bit move)
|
||||||
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
"FMVWX": {0x78, 0x0, 0x53}, // int32 → float32 (bit move)
|
||||||
|
// The toolchain's W/D suffix spellings of the same moves.
|
||||||
|
"FMVXS": {0x70, 0x0, 0x53},
|
||||||
|
"FMVFS": {0x78, 0x0, 0x53},
|
||||||
|
"FMVSX": {0x79, 0x0, 0x53},
|
||||||
}
|
}
|
||||||
|
|
||||||
// riscvCvtType encodes an FP conversion instruction.
|
// riscvCvtType encodes an FP conversion instruction.
|
||||||
@@ -441,6 +496,71 @@ func riscvJType(rd int, offset int32) uint32 {
|
|||||||
0x6F // JAL opcode
|
0x6F // JAL opcode
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- RVV ("V" extension) encoding helpers ----
|
||||||
|
|
||||||
|
// The OP-V major opcode and its funct3 subclasses.
|
||||||
|
const (
|
||||||
|
riscvOpV = 0x57 // the vector operation opcode (also OPcfg for vset*)
|
||||||
|
// funct3 values: 0 OPIVV, 1 OPFVV, 2 OPMVV, 3 OPIVI, 4 OPIVX,
|
||||||
|
// 5 OPFVF, 6 OPMVX, 7 vsetvli.
|
||||||
|
riscvVf3VV = 0x0 // vector-vector
|
||||||
|
riscvVf3MV = 0x2 // vector mask
|
||||||
|
riscvVf3VI = 0x3 // vector-immediate
|
||||||
|
riscvVf3VX = 0x4 // vector-scalar
|
||||||
|
riscvVf3Cfg = 0x7 // vsetvli
|
||||||
|
)
|
||||||
|
|
||||||
|
// riscvVType composes the vsetvli/vsetivli vtype immediate: the register
|
||||||
|
// group multiplier in [2:0], the selected element width in [5:3] and the
|
||||||
|
// tail-agnostic and mask-agnostic policies in bits 6 and 7.
|
||||||
|
func riscvVType(vsew, vlmul, vta, vma int) int {
|
||||||
|
return vlmul | vsew<<3 | vta<<6 | vma<<7
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVSetEnc encodes VSETVLI and VSETIVLI: imm[31:20] = vtype, rs1 = the
|
||||||
|
// avl register or 5-bit uimm, rd = the destination. Both carry funct3 7; a
|
||||||
|
// vsetivli is distinguished by bits [31:30] set in the immediate (the 0xC00
|
||||||
|
// the toolchain writes above its 10-bit vtype).
|
||||||
|
func riscvVSetEnc(vsetivli bool, avl, vtype, rd int) uint32 {
|
||||||
|
imm := vtype & 0x3FF
|
||||||
|
if vsetivli {
|
||||||
|
imm |= 0xC00
|
||||||
|
}
|
||||||
|
return uint32(imm)<<20 | uint32(avl&0x1F)<<15 | uint32(riscvVf3Cfg)<<12 |
|
||||||
|
uint32(rd)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVLSType encodes a vector load or store: the full 32-bit word with the
|
||||||
|
// segment count in bits [31:29], the addressing mode in bits [28:26], the
|
||||||
|
// unmasked bit at 25 and the width in funct3. width follows the load
|
||||||
|
// convention (0 = 8-bit, 5 = 16-bit, 6 = 32-bit, 7 = 64-bit).
|
||||||
|
func riscvVLSType(op uint32, nf, mop, width int, rs2 int32, rs1, rd int) uint32 {
|
||||||
|
return uint32(nf&0x7)<<29 | uint32(mop&0x7)<<26 | 1<<25 |
|
||||||
|
uint32(rs2)<<20 | uint32(rs1)<<15 | uint32(width&0x7)<<12 |
|
||||||
|
uint32(rd)<<7 | op
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVVInstr encodes an OP-V instruction with the six-bit operation code in
|
||||||
|
// funct7's upper bits, bit 25 as the unmasked flag and the three registers in
|
||||||
|
// the standard positions. vs1 may name an integer register for the *VX forms
|
||||||
|
// (the scalar sits in the rs1 field) or an immediate for the *VI forms.
|
||||||
|
func riscvVVInstr(funct6, funct3 int, vs1 int32, vs2, vd int) uint32 {
|
||||||
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs1)<<15 |
|
||||||
|
uint32(funct3)<<12 | uint32(vs2)<<20 | uint32(vd)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvVUnaryInstr encodes a one-vector-operand OP-V instruction whose fixed
|
||||||
|
// fields live where the second source register would be: rs1Field and vs2 are
|
||||||
|
// written verbatim (the oracle writes fixed non-zero constants there for some
|
||||||
|
// instructions, such as 0x11 in the rs1 field of vmfirst.m and vid.v).
|
||||||
|
func riscvVUnaryInstr(funct6, funct3 int, rs1Field int32, vs2, vd int) uint32 {
|
||||||
|
return uint32(funct6&0x3F)<<26 | 1<<25 | uint32(vs2&0x1F)<<20 |
|
||||||
|
uint32(rs1Field&0x1F)<<15 | uint32(funct3&0x7)<<12 | uint32(vd&0x1F)<<7 | riscvOpV
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvSegNF maps a segment count to the 3-bit nf field (count - 1).
|
||||||
|
func riscvSegNF(n int) int32 { return int32(n - 1) }
|
||||||
|
|
||||||
// ---- RVC (compressed) encoding helpers ----
|
// ---- RVC (compressed) encoding helpers ----
|
||||||
|
|
||||||
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
// isRVCIntReg reports whether a register number can be encoded in the 3-bit
|
||||||
|
|||||||
+394
-25
@@ -5,11 +5,13 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"bytes"
|
"bytes"
|
||||||
|
"encoding/binary"
|
||||||
|
"encoding/hex"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
// firstTextRISCV parses assembly source and returns the first TEXT function body.
|
||||||
@@ -31,7 +33,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
|
|||||||
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
|
||||||
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
code, _, _, _, _, err := assembleRISCV(fn)
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("assemble: %v", err)
|
t.Fatalf("assemble: %v", err)
|
||||||
}
|
}
|
||||||
@@ -783,16 +785,140 @@ TEXT ·sys(SB), NOSPLIT, $0
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestRISCV_MOV_sym_FP_error(t *testing.T) {
|
func TestRISCV_MOV_sym_FP(t *testing.T) {
|
||||||
// MOV $sym(FP), rd should return an error (unsupported).
|
// MOV $sym(FP), rd lowers to the frame-adjusted ADDI against SP: the
|
||||||
|
// toolchain's argframe spelling. A zero frame leaves the offset at the
|
||||||
|
// 8-byte link slot, compressed to C.ADDI4SPN.
|
||||||
fn := firstTextRISCV(t, `#include "textflag.h"
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
TEXT ·badfp(SB), NOSPLIT, $0
|
TEXT ·argfp(SB), NOSPLIT, $0
|
||||||
MOV $arg(FP), X10
|
MOV $arg(FP), X10
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
_, _, _, _, _, err := assembleRISCV(fn)
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
if err == nil {
|
if err != nil {
|
||||||
t.Error("expected error for MOV $arg(FP), got nil")
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// prologue (0: leaf, zero frame) + C.ADDI4SPN (2) + RET (4) = 6
|
||||||
|
want := []byte{0x28, 0x00, 0x67, 0x80, 0x00, 0x00}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_Bookkeeping(t *testing.T) {
|
||||||
|
// FUNCDATA and PCDATA contribute no bytes; UNDEF is the toolchain's
|
||||||
|
// ebreak, compressed to C.EBREAK under RVC.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·book(SB), NOSPLIT, $0-8
|
||||||
|
FUNCDATA $0, marks<>(SB)
|
||||||
|
PCDATA $1, $1
|
||||||
|
UNDEF
|
||||||
|
MOV $1, X10
|
||||||
|
MOV X10, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// C.EBREAK (2) + C.LI X10, 1 (2) + C.SWSP (2) + RET (4) = 10: the
|
||||||
|
// FUNCDATA and PCDATA statements contribute nothing.
|
||||||
|
want := []byte{0x02, 0x90, 0x05, 0x45, 0x2a, 0xe4, 0x67, 0x80, 0x00, 0x00}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_JMPPCRel(t *testing.T) {
|
||||||
|
// JMP N(PC): the displacement tracks the instruction N source slots
|
||||||
|
// away in the final layout (0 the jump itself, negative backwards).
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·slots(SB), NOSPLIT, $0-0
|
||||||
|
JMP 2(PC)
|
||||||
|
MOV $1, X11
|
||||||
|
MOV $2, X12
|
||||||
|
MOV X12, X11
|
||||||
|
JMP -3(PC)
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// JMP 2(PC) lands on the C.MV six bytes ahead; JMP -3(PC) lands back on
|
||||||
|
// the first C.LI, six bytes behind.
|
||||||
|
want := []byte{
|
||||||
|
0x6f, 0x00, 0x60, 0x00, // JAL X0, 6
|
||||||
|
0x85, 0x45, // C.LI X11, 1
|
||||||
|
0x09, 0x46, // C.LI X12, 2
|
||||||
|
0xb2, 0x85, // C.MV X11, X12
|
||||||
|
0x6f, 0xf0, 0xbf, 0xff, // JAL X0, -6
|
||||||
|
0x67, 0x80, 0x00, 0x00, // RET
|
||||||
|
}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOVWideImm(t *testing.T) {
|
||||||
|
// Shift-sequence constants compress like the toolchain's expansion.
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·wide(SB), NOSPLIT, $0-0
|
||||||
|
MOV $0x8000000000000000, X5
|
||||||
|
MOV $0x100000000, X5
|
||||||
|
MOV $0x000fffffffffffda, X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("assemble: %v", err)
|
||||||
|
}
|
||||||
|
// C.LI -1, C.SLLI 63; C.LI 1, C.SLLI 32; C.LI -19, C.SLLI 13, SRLI 12.
|
||||||
|
want := []byte{
|
||||||
|
0xfd, 0x52, 0xfe, 0x12,
|
||||||
|
0x85, 0x42, 0x82, 0x12,
|
||||||
|
0xb5, 0x52, 0xb6, 0x02, 0x93, 0xd2, 0xc2, 0x00,
|
||||||
|
0x67, 0x80, 0x00, 0x00,
|
||||||
|
}
|
||||||
|
if string(code) != string(want) {
|
||||||
|
t.Errorf("got % x, want % x", code, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRISCV_MOVImmPool(t *testing.T) {
|
||||||
|
// A constant outside the shift shapes loads from the pooled $i64 data
|
||||||
|
// symbol via AUIPC+LD, named like the toolchain's pool.
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
TEXT ·pool(SB), NOSPLIT, $0-8
|
||||||
|
MOV $0x0101010101010101, X16
|
||||||
|
MOV X16, ret+0(FP)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("pool_riscv64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileRISCV(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileRISCV: %v", err)
|
||||||
|
}
|
||||||
|
// AUIPC X16, 0 + LD X16, 0(X16): the relocation pair carries the symbol.
|
||||||
|
wantCode := []byte{0x17, 0x08, 0x00, 0x00, 0x03, 0x38, 0x08, 0x00}
|
||||||
|
if string(img.Code[0:8]) != string(wantCode) {
|
||||||
|
t.Errorf("pool load: got % x", img.Code[0:8])
|
||||||
|
}
|
||||||
|
var lit *DataSymbol
|
||||||
|
for i := range img.DataSyms {
|
||||||
|
if img.DataSyms[i].Name == "$i64.0101010101010101" {
|
||||||
|
lit = &img.DataSyms[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if lit == nil {
|
||||||
|
t.Fatalf("pool symbol missing: %v", img.DataSyms)
|
||||||
|
}
|
||||||
|
wantData := []byte{0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01, 0x01}
|
||||||
|
if string(img.Data[lit.Offset:lit.Offset+8]) != string(wantData) {
|
||||||
|
t.Errorf("pool bytes: got % x", img.Data[lit.Offset:lit.Offset+8])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -803,7 +929,7 @@ TEXT ·calltest(SB), NOSPLIT, $0
|
|||||||
CALL ext(SB)
|
CALL ext(SB)
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
code, _, relocs, _, _, err := assembleRISCV(fn)
|
code, _, relocs, _, _, _, err := assembleRISCV(fn)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("assemble: %v", err)
|
t.Fatalf("assemble: %v", err)
|
||||||
}
|
}
|
||||||
@@ -832,7 +958,7 @@ TEXT ·calllocal(SB), NOSPLIT, $0
|
|||||||
sub:
|
sub:
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
_, _, _, _, _, err := assembleRISCV(fn)
|
_, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
if err == nil {
|
if err == nil {
|
||||||
t.Error("expected error for CALL to local label, got nil")
|
t.Error("expected error for CALL to local label, got nil")
|
||||||
}
|
}
|
||||||
@@ -866,7 +992,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
|
|||||||
t.Helper()
|
t.Helper()
|
||||||
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
|
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
|
||||||
instr := fn.Body[0].(*ast.Instr)
|
instr := fn.Body[0].(*ast.Instr)
|
||||||
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil)
|
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil)
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
|
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
|
||||||
@@ -903,9 +1029,10 @@ func TestRISCVBranchJumpRange(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestRISCVBranchFarBody drives the range check through the full two-pass
|
// TestRISCVBranchFarBody drives the relaxation pass through the full
|
||||||
// assembler: a forward branch over a body larger than the B-type span must
|
// assembler: a forward branch over a body larger than the B-type span is
|
||||||
// error rather than wrap.
|
// rewritten as an inverted branch over an inserted JMP, the same layout the
|
||||||
|
// toolchain produces, instead of wrapping to a wrong target.
|
||||||
func TestRISCVBranchFarBody(t *testing.T) {
|
func TestRISCVBranchFarBody(t *testing.T) {
|
||||||
var sb strings.Builder
|
var sb strings.Builder
|
||||||
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
|
sb.WriteString("#include \"textflag.h\"\nTEXT ·far(SB), NOSPLIT, $0\n\tBEQ X10, X11, done\n")
|
||||||
@@ -914,8 +1041,20 @@ func TestRISCVBranchFarBody(t *testing.T) {
|
|||||||
}
|
}
|
||||||
sb.WriteString("done:\n\tRET\n")
|
sb.WriteString("done:\n\tRET\n")
|
||||||
fn := firstTextRISCV(t, sb.String())
|
fn := firstTextRISCV(t, sb.String())
|
||||||
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
out, _, _, _, _, _, err := assembleRISCV(fn)
|
||||||
t.Error("expected a branch-out-of-range error, got none")
|
if err != nil {
|
||||||
|
t.Fatalf("unexpected error: %v", err)
|
||||||
|
}
|
||||||
|
// The relaxed branch at offset 0 targets the inserted JMP at 4 (bne
|
||||||
|
// x10, x11, +4); the JMP at 4 carries the far forward displacement.
|
||||||
|
wantBranch := wordLE(riscvBType(riscvEnc{0x63, 0x1, 0x00}, 10, 11, 4))
|
||||||
|
if !bytes.Equal(out[0:4], wantBranch) {
|
||||||
|
t.Errorf("relaxed branch = %x, want %x", out[0:4], wantBranch)
|
||||||
|
}
|
||||||
|
// done sits after 1100 ADDs: 4 + 4400, i.e. offset 4404 from the JMP at 4.
|
||||||
|
wantJmp := wordLE(riscvJType(0, 4404))
|
||||||
|
if !bytes.Equal(out[4:8], wantJmp) {
|
||||||
|
t.Errorf("inserted JMP = %x, want %x", out[4:8], wantJmp)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -928,7 +1067,7 @@ TEXT ·csrhi(SB), NOSPLIT, $0
|
|||||||
CSRRW $4096, X10, X11
|
CSRRW $4096, X10, X11
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
t.Error("expected an out-of-range error for CSR $4096, got none")
|
t.Error("expected an out-of-range error for CSR $4096, got none")
|
||||||
}
|
}
|
||||||
fn = firstTextRISCV(t, `#include "textflag.h"
|
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||||
@@ -936,25 +1075,24 @@ TEXT ·csrmax(SB), NOSPLIT, $0
|
|||||||
CSRRW $4095, X10, X11
|
CSRRW $4095, X10, X11
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
t.Errorf("CSR $4095 must assemble: %v", err)
|
t.Errorf("CSR $4095 must assemble: %v", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
|
// TestRISCV_Imm64Rejected checks that immediates outside the signed 32-bit
|
||||||
// span are diagnosed instead of silently truncated to their low 32 bits (the
|
// span are diagnosed instead of silently truncated to their low 32 bits for
|
||||||
// toolchain materialises such constants via SLLI expansion, which this
|
// the I-type arithmetic; the MOV forms materialise the wide constant instead
|
||||||
// assembler does not implement).
|
// (shift sequence or pooled load), like the toolchain.
|
||||||
func TestRISCV_Imm64Rejected(t *testing.T) {
|
func TestRISCV_Imm64Rejected(t *testing.T) {
|
||||||
cases := []string{
|
cases := []string{
|
||||||
"MOV $0x123456789, X10",
|
|
||||||
"ADDI $0x100000000, X10, X11",
|
"ADDI $0x100000000, X10, X11",
|
||||||
"ANDI $-0x800000001, X10, X11",
|
"ANDI $-0x800000001, X10, X11",
|
||||||
"SUB $0x100000000, X10, X11",
|
"SUB $0x100000000, X10, X11",
|
||||||
}
|
}
|
||||||
for _, src := range cases {
|
for _, src := range cases {
|
||||||
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
|
||||||
if _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
|
||||||
t.Errorf("%s: expected an out-of-range error, got none", src)
|
t.Errorf("%s: expected an out-of-range error, got none", src)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -967,7 +1105,238 @@ TEXT ·edge(SB), NOSPLIT, $0
|
|||||||
SUB $0x80000000, X12, X13
|
SUB $0x80000000, X12, X13
|
||||||
RET
|
RET
|
||||||
`)
|
`)
|
||||||
if _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
t.Errorf("int32-span immediates must assemble: %v", err)
|
t.Errorf("int32-span immediates must assemble: %v", err)
|
||||||
}
|
}
|
||||||
|
// Beyond the span the MOV forms materialise the constant like the
|
||||||
|
// toolchain instead of diagnosing it.
|
||||||
|
fn = firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·pool(SB), NOSPLIT, $0
|
||||||
|
MOV $0x123456789, X10
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
|
||||||
|
t.Errorf("MOV with a 64-bit immediate must assemble: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvWants decodes code as little-endian words and pins each one; the
|
||||||
|
// expected values below were read off GOARCH=riscv64 go tool objdump of
|
||||||
|
// kernels assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||||
|
// cross-checks the same words).
|
||||||
|
func riscvWants(t *testing.T, code []byte, want ...uint32) {
|
||||||
|
t.Helper()
|
||||||
|
got := make([]uint32, 0, len(code)/4)
|
||||||
|
for i := 0; i+4 <= len(code); i += 4 {
|
||||||
|
got = append(got, binary.LittleEndian.Uint32(code[i:]))
|
||||||
|
}
|
||||||
|
if len(got) < len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code)
|
||||||
|
}
|
||||||
|
// The RET (JALR) ends the sequence; only the pinned prefix is compared.
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// riscvWantsHex pins the exact hex encoding of a function's instruction
|
||||||
|
// bytes, including any 2-byte compressed instructions in the stream; the
|
||||||
|
// expected strings were read off GOARCH=riscv64 go tool objdump of kernels
|
||||||
|
// assembled with go tool asm (the toolchain's riscv64.s testdata
|
||||||
|
// cross-checks the same words).
|
||||||
|
func riscvWantsHex(t *testing.T, code []byte, wantHex string) {
|
||||||
|
t.Helper()
|
||||||
|
got := hex.EncodeToString(code)
|
||||||
|
if got != wantHex {
|
||||||
|
t.Errorf("code = %s, want %s", got, wantHex)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_extendedPseudos pins the toolchain-synthesised instructions:
|
||||||
|
// ANDN/ORN (XORI + AND/OR through the destination or TMP), the five-word
|
||||||
|
// MIN/MAX expansion, the four-word rotate, ROR's compressed reverse shift
|
||||||
|
// (C.SLLI when rd == rs1, both non-zero, 1 <= sll <= 63), the identical-
|
||||||
|
// input MIN/MAX fold to C.MV, FABSD (FSGNJX.D), SEQZ and RDTIME (csrrs with
|
||||||
|
// the time CSR).
|
||||||
|
func TestRISCV_extendedPseudos(t *testing.T) {
|
||||||
|
t.Run("logic and minmax", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·l(SB), NOSPLIT, $0
|
||||||
|
ANDN X19, X20, X21
|
||||||
|
ANDN X19, X20
|
||||||
|
ORN X20, X19
|
||||||
|
MAX X26, X28, X29
|
||||||
|
MIN X29, X30, X5
|
||||||
|
MAX X5, X5
|
||||||
|
MAX X5, X5, X6
|
||||||
|
SEQZ X5, X6
|
||||||
|
NEG X5, X6
|
||||||
|
NOT X5
|
||||||
|
RDTIME X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// Words 0-10 up to the folded C.MV pair (halfwords 96 82 and 16 83),
|
||||||
|
// then SEQZ, NEG, NOT and RDTIME.
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"93caf9ffb37a5a01"+"93cff9ff337afa01"+"934ffaffb3e9f901"+
|
||||||
|
"b32fae01b30ff041b34eae01b3fedf01b34ede01"+
|
||||||
|
"b3afee01b30ff041b342df01b3f25f00b3425f00"+
|
||||||
|
"9682"+"1683"+
|
||||||
|
"13b31200"+"33035040"+"93c2f2ff"+"f32210c0"+"67800000")
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("rotate", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·r(SB), NOSPLIT, $0
|
||||||
|
ROR X10, X11, X12
|
||||||
|
ROR X10, X11
|
||||||
|
ROR $63, X11
|
||||||
|
RORIW $31, X13, X14
|
||||||
|
RORIW $1, X14, X15
|
||||||
|
RORIW $3, X14
|
||||||
|
RORW X15, X16, X17
|
||||||
|
RORW $31, X13
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
// The third ROR carries the compressed C.SLLI (05 86) in mid-stream.
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"b30fa040b39ff50133d6a50033e6cf00"+
|
||||||
|
"b30fa040b39ff501b3d5a500b3e5bf00"+
|
||||||
|
"93dff5038605b3e5bf00"+
|
||||||
|
"9bdff6011b97160033e7ef00"+
|
||||||
|
"9b5f17009b17f701b3e7ff00"+
|
||||||
|
"9b5f37001b17d70133e7ef00"+
|
||||||
|
"b30ff040bb1ff801bb58f800b3e81f01"+
|
||||||
|
"9bdff6019b961600b3e6df00"+"67800000")
|
||||||
|
})
|
||||||
|
|
||||||
|
t.Run("fp and branches", func(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0
|
||||||
|
FABSD F1, F2
|
||||||
|
FSGNJD F1, F0, F2
|
||||||
|
FMADDD F1, F2, F3, F4
|
||||||
|
FMSUBD F1, F2, F3, F4
|
||||||
|
FNMSUBD F1, F2, F3, F4
|
||||||
|
BGT X5, X6, tgt
|
||||||
|
BLE X5, X6, tgt
|
||||||
|
BGTU X5, X6, tgt
|
||||||
|
BLEU X5, X6, tgt
|
||||||
|
tgt:
|
||||||
|
RDTIME X5
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWantsHex(t, code,
|
||||||
|
"53a11022"+"53011022"+"4382201a4782201a4b82201a"+
|
||||||
|
"63485300635653006364530063725300"+ // blt/bge/bltu/bgeu x6, x5
|
||||||
|
"f32210c0"+"67800000")
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_amoWords pins the full AMO family: every AMO carries aq and rl
|
||||||
|
// (funct7 |= 3), LR is acquire (funct7 |= 2) and SC release (funct7 |= 1),
|
||||||
|
// exactly as GOARCH=riscv64 go tool asm encodes them.
|
||||||
|
func TestRISCV_amoWords(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·amo(SB), NOSPLIT, $0
|
||||||
|
AMOSWAPW X5, (X6), X7
|
||||||
|
AMOSWAPD X5, (X6), X7
|
||||||
|
AMOADDW X5, (X6), X7
|
||||||
|
AMOADDD X5, (X6), X7
|
||||||
|
AMOANDW X5, (X6), X7
|
||||||
|
AMOANDD X5, (X6), X7
|
||||||
|
AMOORW X5, (X6), X7
|
||||||
|
AMOORD X5, (X6), X7
|
||||||
|
AMOXORW X5, (X6), X7
|
||||||
|
AMOXORD X5, (X6), X7
|
||||||
|
AMOMAXW X5, (X6), X7
|
||||||
|
AMOMAXD X5, (X6), X7
|
||||||
|
AMOMAXUW X5, (X6), X7
|
||||||
|
AMOMAXUD X5, (X6), X7
|
||||||
|
AMOMINUW X5, (X6), X7
|
||||||
|
AMOMINUD X5, (X6), X7
|
||||||
|
LRW (X5), X6
|
||||||
|
LRD (X5), X6
|
||||||
|
SCW X5, (X6), X7
|
||||||
|
SCD X5, (X6), X7
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0E5323AF, // amoswap.w
|
||||||
|
0x0E5333AF, // amoswap.d
|
||||||
|
0x065323AF, // amoaddd.w
|
||||||
|
0x065333AF, // amoadd.d
|
||||||
|
0x665323AF, // amoand.w
|
||||||
|
0x665333AF, // amoand.d
|
||||||
|
0x465323AF, // amoor.w
|
||||||
|
0x465333AF, // amoor.d
|
||||||
|
0x265323AF, // amoxor.w
|
||||||
|
0x265333AF, // amoxor.d
|
||||||
|
0xA65323AF, // amomax.w
|
||||||
|
0xA65333AF, // amomax.d
|
||||||
|
0xE65323AF, // amomaxu.w
|
||||||
|
0xE65333AF, // amomaxu.d
|
||||||
|
0xC65323AF, // amominu.w
|
||||||
|
0xC65333AF, // amominu.d
|
||||||
|
0x1402A32F, // lr.w (aq)
|
||||||
|
0x1402B32F, // lr.d
|
||||||
|
0x1A5323AF, // sc.w (rl)
|
||||||
|
0x1A5333AF, // sc.d
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRISCV_vectorWords pins the RVV slice and the VSET* encodings. The
|
||||||
|
// toolchain canonicalises an immediate avl to vsetivli even under the
|
||||||
|
// VSETVLI spelling (`VSETVLI $15` and `VSETIVLI $15` come out byte-
|
||||||
|
// identical), which is what the 0xC00 bit of the first word carries.
|
||||||
|
func TestRISCV_vectorWords(t *testing.T) {
|
||||||
|
fn := firstTextRISCV(t, `#include "textflag.h"
|
||||||
|
TEXT ·v(SB), NOSPLIT, $0
|
||||||
|
VSETVLI X5, E8, M8, TA, MA, X6
|
||||||
|
VSETIVLI $4, E32, M1, TA, MA, X0
|
||||||
|
VSETVLI $15, E32, M1, TA, MA, X12
|
||||||
|
VADDVV V1, V2, V3
|
||||||
|
VADDVX X12, V12, V12
|
||||||
|
VXORVV V8, V16, V24
|
||||||
|
VMSEQVX X12, V8, V0
|
||||||
|
VMSNEVV V8, V16, V0
|
||||||
|
VSLLVI $8, V28, V30
|
||||||
|
VSRLVI $25, V29, V29
|
||||||
|
VFIRSTM V0, X6
|
||||||
|
VIDV V12
|
||||||
|
VMV4RV V8, V24
|
||||||
|
VLE8V (X10), V8
|
||||||
|
VSE8V V24, (X10)
|
||||||
|
VSE32V V9, (X11)
|
||||||
|
VLSSEG4E32V (X14), X0, V0
|
||||||
|
VLSSEG8E32V (X10), X0, V4
|
||||||
|
RET
|
||||||
|
`)
|
||||||
|
code := assembleRISCVHelper(t, fn)
|
||||||
|
riscvWants(t, code,
|
||||||
|
0x0C32F357, // vsetvli x6, x5, vtype 0xc3 (E8, M8, TA, MA)
|
||||||
|
0xCD027057, // vsetivli x0, 4
|
||||||
|
0xCD07F657, // vsetivli x12, 15: VSETVLI $15 canonicalises to the same word
|
||||||
|
0x022081D7, // vadd.vv v3, v2, v1
|
||||||
|
0x02C64657, // vadd.vx v12, v12, x12
|
||||||
|
0x2F040C57, // vxor.vv v24, v16, v8
|
||||||
|
0x62864057, // vmseq.vx v0, v8, x12
|
||||||
|
0x67040057, // vmsne.vv v0, v16, v8
|
||||||
|
0x97C43F57, // vsll.vi v30, v28, 8
|
||||||
|
0xA3DCBED7, // vsrl.vi v29, v29, 25
|
||||||
|
0x4208A357, // vmfirst.m x6, v0
|
||||||
|
0x5208A657, // vid.v v12
|
||||||
|
0x9E81BC57, // vmv4r.v v24, v8
|
||||||
|
0x02050407, // vle8.v v8, (x10)
|
||||||
|
0x02050C27, // vse8.v v24, (x10)
|
||||||
|
0x0205E4A7, // vse32.v v9, (x11)
|
||||||
|
0x6A076007, // vlsseg4e32.v v0, (x14), x0
|
||||||
|
0xEA056207, // vlsseg8e32.v v4, (x10), x0
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
)
|
)
|
||||||
|
|
||||||
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ package asm
|
|||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
// TestRISCVFrameSpadjAndLines checks that a framed function records its
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
|
||||||
|
|||||||
+362
-5
@@ -41,7 +41,7 @@ const (
|
|||||||
vexExtract
|
vexExtract
|
||||||
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
// vexRMRev is the reversed two-operand form `OP src, dst` with the source
|
||||||
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
// in ModRM.reg and the destination in r/m, the layout of the EVEX
|
||||||
// narrowing stores (VPMOVDW, VPMOVQD).
|
// narrowing stores (VPMOVDW, VPMOVQD) and of the non-temporal VMOVNTDQ.
|
||||||
vexRMRev
|
vexRMRev
|
||||||
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
// vexRMSrcLen is the two-operand conversion form `OP src, dst` whose
|
||||||
// vector length follows the source: the packed-double → dword
|
// vector length follows the source: the packed-double → dword
|
||||||
@@ -52,6 +52,30 @@ const (
|
|||||||
vexRMSrcLen
|
vexRMSrcLen
|
||||||
// vexZero is the no-operand form (VZEROUPPER).
|
// vexZero is the no-operand form (VZEROUPPER).
|
||||||
vexZero
|
vexZero
|
||||||
|
// vexZeroAll is the no-operand form that zeroes the full upper state
|
||||||
|
// (VZEROALL, the L = 1 twin of VZEROUPPER).
|
||||||
|
vexZeroAll
|
||||||
|
// vexNDS3GPR is the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): reg = dst, vvvv = src1, rm = src2, L = 0.
|
||||||
|
vexNDS3GPR
|
||||||
|
// vexImmRMGPR is the immediate form over general-purpose registers
|
||||||
|
// (RORX): reg = dst, rm = src, imm8 = op0, L = 0.
|
||||||
|
vexImmRMGPR
|
||||||
|
// vexRMOpGPR is the two-operand /digit form over general-purpose
|
||||||
|
// registers (BLSI, BLSMSK, BLSR): ModRM.reg = /digit, ModRM.rm = src
|
||||||
|
// (op0), VEX.vvvv = dst (op1), L = 0.
|
||||||
|
vexRMOpGPR
|
||||||
|
// vexCountGPR is the three-operand count form over general-purpose
|
||||||
|
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): the first operand rides
|
||||||
|
// VEX.vvvv and the second is r/m, the opposite pairing of the ANDN
|
||||||
|
// family, with reg = dst (op2), L = 0.
|
||||||
|
vexCountGPR
|
||||||
|
// vexExtractGPR is the lane-extract-to-GPR form `OP $imm, xsrc, GPR/mem
|
||||||
|
// dst`: ModRM.reg = xsrc (op1), ModRM.rm = destination (op2), imm8 =
|
||||||
|
// op0, the VPEXTRB/W/D/Q layout. EVEX only; the destination never
|
||||||
|
// carries a vector length, so the register the L'L field follows is the
|
||||||
|
// XMM source.
|
||||||
|
vexExtractGPR
|
||||||
)
|
)
|
||||||
|
|
||||||
// vexSpec describes one VEX instruction's encoding parameters.
|
// vexSpec describes one VEX instruction's encoding parameters.
|
||||||
@@ -125,6 +149,12 @@ var vexTable = map[string]vexSpec{
|
|||||||
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
"VMAXSS": {1, 0x5F, 0, 2, -1, vexNDS3},
|
||||||
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
// VEX.128/256.66.0F38.W1, fused multiply-add (NDS form).
|
||||||
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
"VFMADD231PD": {2, 0xB8, 1, 1, -1, vexNDS3},
|
||||||
|
// Scalar fused multiply-add (NDS form). The Go assembler carries the
|
||||||
|
// same 66 prefix as the packed forms on every FMA row, and W1 on the
|
||||||
|
// double-precision spellings, so SD shares PD's prefix/W pair and the
|
||||||
|
// scalar width rides on the W bit.
|
||||||
|
"VFMADD213SD": {2, 0xA9, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231SD": {2, 0xBD, 1, 1, -1, vexNDS3},
|
||||||
|
|
||||||
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
// VEX.128/256.66.0F38.WIG, sign/zero extend and broadcast (reg=dst, rm=src,
|
||||||
// no vvvv).
|
// no vvvv).
|
||||||
@@ -192,6 +222,57 @@ var vexTable = map[string]vexSpec{
|
|||||||
|
|
||||||
// VEX.128.0F.W0, no operands.
|
// VEX.128.0F.W0, no operands.
|
||||||
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
"VZEROUPPER": {1, 0x77, 0, 0, -1, vexZero},
|
||||||
|
// VEX.256.0F.W0, zero all vector registers (the L = 1 twin).
|
||||||
|
"VZEROALL": {1, 0x77, 0, 0, -1, vexZeroAll},
|
||||||
|
// VEX.128/256.66.0F38, byte shuffle shifts and the packed byte compare.
|
||||||
|
"VPSLLDQ": {1, 0x73, 0, 1, 7, vexShiftImm},
|
||||||
|
"VPSRLDQ": {1, 0x73, 0, 1, 3, vexShiftImm},
|
||||||
|
"VPCMPEQB": {1, 0x74, 0, 1, -1, vexNDS3},
|
||||||
|
// VEX.128/256.0F.WIG, packed single XOR (NDS form).
|
||||||
|
"VXORPS": {1, 0x57, 0, 0, -1, vexNDS3},
|
||||||
|
// VEX.256.66.0F3A.W0, two-source permutes and blends with an imm8 control.
|
||||||
|
"VPERM2F128": {3, 0x06, 0, 1, -1, vexNDS3Imm},
|
||||||
|
"VPBLENDD": {3, 0x02, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.WIG, byte align (NDS + imm8); the ZMM spelling
|
||||||
|
// falls through to the EVEX table.
|
||||||
|
"VPALIGNR": {3, 0x0F, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W0, carry-less multiply ($imm, src2, src1, dst).
|
||||||
|
"VPCLMULQDQ": {3, 0x44, 0, 1, -1, vexNDS3Imm},
|
||||||
|
// VEX.128/256.66.0F3A.W1, GF(2^8) affine transform (NDS + imm8).
|
||||||
|
"VGF2P8AFFINEQB": {3, 0xCE, 1, 1, -1, vexNDS3Imm},
|
||||||
|
// BMI1/BMI2 general-register VEX forms (see vexNDS3GPR/vexImmRMGPR).
|
||||||
|
"ANDNL": {2, 0xF2, 0, 0, -1, vexNDS3GPR},
|
||||||
|
"ANDNQ": {2, 0xF2, 1, 0, -1, vexNDS3GPR},
|
||||||
|
"MULXL": {2, 0xF6, 0, 3, -1, vexNDS3GPR},
|
||||||
|
"MULXQ": {2, 0xF6, 1, 3, -1, vexNDS3GPR},
|
||||||
|
// VEX.NDS.LZ.0F38, the BMI2 three-operand bit ops: BEXTR and BZHI
|
||||||
|
// share the F7/F5 opcodes across W, the variable shifts carry their
|
||||||
|
// direction in the prefix (SHLX 66, SHRX F2, SARX F3) and PDEP/PEXT
|
||||||
|
// in F2/F3.
|
||||||
|
"BEXTRL": {2, 0xF7, 0, 0, -1, vexCountGPR},
|
||||||
|
"BEXTRQ": {2, 0xF7, 1, 0, -1, vexCountGPR},
|
||||||
|
"BZHIL": {2, 0xF5, 0, 0, -1, vexCountGPR},
|
||||||
|
"BZHIQ": {2, 0xF5, 1, 0, -1, vexCountGPR},
|
||||||
|
"SARXL": {2, 0xF7, 0, 2, -1, vexCountGPR},
|
||||||
|
"SARXQ": {2, 0xF7, 1, 2, -1, vexCountGPR},
|
||||||
|
"SHLXL": {2, 0xF7, 0, 1, -1, vexCountGPR},
|
||||||
|
"SHLXQ": {2, 0xF7, 1, 1, -1, vexCountGPR},
|
||||||
|
"SHRXL": {2, 0xF7, 0, 3, -1, vexCountGPR},
|
||||||
|
"SHRXQ": {2, 0xF7, 1, 3, -1, vexCountGPR},
|
||||||
|
"PDEPL": {2, 0xF5, 0, 3, -1, vexNDS3GPR},
|
||||||
|
"PDEPQ": {2, 0xF5, 1, 3, -1, vexNDS3GPR},
|
||||||
|
"PEXTL": {2, 0xF5, 0, 2, -1, vexNDS3GPR},
|
||||||
|
"PEXTQ": {2, 0xF5, 1, 2, -1, vexNDS3GPR},
|
||||||
|
// VEX.LZ.0F38.W, the BMI1 unary bit ops (src, dst: ModRM.reg = /digit,
|
||||||
|
// rm = src, vvvv = dst).
|
||||||
|
"BLSIL": {2, 0xF3, 0, 0, 3, vexRMOpGPR},
|
||||||
|
"BLSIQ": {2, 0xF3, 1, 0, 3, vexRMOpGPR},
|
||||||
|
"BLSMSKL": {2, 0xF3, 0, 0, 2, vexRMOpGPR},
|
||||||
|
"BLSMSKQ": {2, 0xF3, 1, 0, 2, vexRMOpGPR},
|
||||||
|
"BLSRL": {2, 0xF3, 0, 0, 1, vexRMOpGPR},
|
||||||
|
"BLSRQ": {2, 0xF3, 1, 0, 1, vexRMOpGPR},
|
||||||
|
"RORXL": {3, 0xF0, 0, 3, -1, vexImmRMGPR},
|
||||||
|
"RORXQ": {3, 0xF0, 1, 3, -1, vexImmRMGPR},
|
||||||
|
|
||||||
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
// VEX.128.0F.W0, mask-register test (KTESTW k1, k2: reg = dst, rm = src).
|
||||||
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
"KTESTW": {1, 0x99, 0, 0, -1, vexRM},
|
||||||
@@ -200,6 +281,14 @@ var vexTable = map[string]vexSpec{
|
|||||||
// rm=scalar memory; SD is 256-bit only).
|
// rm=scalar memory; SD is 256-bit only).
|
||||||
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
"VBROADCASTSS": {2, 0x18, 0, 1, -1, vexRM},
|
||||||
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
"VBROADCASTSD": {2, 0x19, 0, 1, -1, vexRM},
|
||||||
|
// VEX.256.66.0F38.W0, broadcast a 128-bit lane into both halves of a
|
||||||
|
// YMM (the encoder rejects an XMM destination, as go tool asm does).
|
||||||
|
"VBROADCASTI128": {2, 0x5A, 0, 1, -1, vexRM},
|
||||||
|
// VEX.128/256.66.0F.WIG, non-temporal store (vector source in reg,
|
||||||
|
// memory destination in rm).
|
||||||
|
"VMOVNTDQ": {1, 0xE7, 0, 1, -1, vexRMRev},
|
||||||
|
// VEX.128/256.66.0F38.W0, test (reg=dst, rm=src, no vvvv).
|
||||||
|
"VPTEST": {2, 0x17, 0, 1, -1, vexRM},
|
||||||
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
// VEX.66.0F38.W0, half-precision convert (reg=dst, rm=half-width
|
||||||
// source).
|
// source).
|
||||||
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
"VCVTPH2PS": {2, 0x13, 0, 1, -1, vexRM},
|
||||||
@@ -242,6 +331,128 @@ var vexTable = map[string]vexSpec{
|
|||||||
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
"VCVTPD2DQY": {1, 0xE6, 0, 3, -1, vexRMSrcLen},
|
||||||
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
"VCVTTPD2DQX": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||||
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
"VCVTTPD2DQY": {1, 0xE6, 0, 1, -1, vexRMSrcLen},
|
||||||
|
|
||||||
|
// --- the VEX forms the avx512enc corpus exercises alongside the EVEX
|
||||||
|
// spellings, read off the toolchain opcode tables ---
|
||||||
|
"VAESDEC": {2, 0xDE, 0, 1, -1, vexNDS3},
|
||||||
|
"VAESDECLAST": {2, 0xDF, 0, 1, -1, vexNDS3},
|
||||||
|
"VAESENC": {2, 0xDC, 0, 1, -1, vexNDS3},
|
||||||
|
"VAESENCLAST": {2, 0xDD, 0, 1, -1, vexNDS3},
|
||||||
|
"VANDNPD": {1, 0x55, 0, 1, -1, vexNDS3},
|
||||||
|
"VANDPD": {1, 0x54, 0, 1, -1, vexNDS3},
|
||||||
|
"VCOMISD": {1, 0x2F, 0, 1, -1, vexRM},
|
||||||
|
"VCVTSD2SS": {1, 0x5A, 0, 3, -1, vexNDS3},
|
||||||
|
"VCVTSS2SD": {1, 0x5A, 0, 2, -1, vexNDS3},
|
||||||
|
"VFMADD132PD": {2, 0x98, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD132PS": {2, 0x98, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD132SD": {2, 0x99, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD132SS": {2, 0x99, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD213PD": {2, 0xA8, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD213PS": {2, 0xA8, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD213SS": {2, 0xA9, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD231PS": {2, 0xB8, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADD231SD": {2, 0xB9, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADD231SS": {2, 0xB9, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB132PD": {2, 0x96, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB132PS": {2, 0x96, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB213PD": {2, 0xA6, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB213PS": {2, 0xA6, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB231PD": {2, 0xB6, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMADDSUB231PS": {2, 0xB6, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132PD": {2, 0x9A, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132PS": {2, 0x9A, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132SD": {2, 0x9B, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB132SS": {2, 0x9B, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213PD": {2, 0xAA, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213PS": {2, 0xAA, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213SD": {2, 0xAB, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB213SS": {2, 0xAB, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231PD": {2, 0xBA, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231PS": {2, 0xBA, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231SD": {2, 0xBB, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUB231SS": {2, 0xBB, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD132PD": {2, 0x97, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD132PS": {2, 0x97, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD213PD": {2, 0xA7, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD213PS": {2, 0xA7, 0, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD231PD": {2, 0xB7, 1, 1, -1, vexNDS3},
|
||||||
|
"VFMSUBADD231PS": {2, 0xB7, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132PD": {2, 0x9C, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132PS": {2, 0x9C, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132SD": {2, 0x9D, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD132SS": {2, 0x9D, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213PD": {2, 0xAC, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213PS": {2, 0xAC, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213SD": {2, 0xAD, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD213SS": {2, 0xAD, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231PD": {2, 0xBC, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231PS": {2, 0xBC, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMADD231SS": {2, 0xBD, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132PD": {2, 0x9E, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132PS": {2, 0x9E, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132SD": {2, 0x9F, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB132SS": {2, 0x9F, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213PD": {2, 0xAE, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213PS": {2, 0xAE, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213SD": {2, 0xAF, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB213SS": {2, 0xAF, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231PD": {2, 0xBE, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231PS": {2, 0xBE, 0, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231SD": {2, 0xBF, 1, 1, -1, vexNDS3},
|
||||||
|
"VFNMSUB231SS": {2, 0xBF, 0, 1, -1, vexNDS3},
|
||||||
|
"VGF2P8AFFINEINVQB": {3, 0xCF, 1, 1, -1, vexNDS3Imm},
|
||||||
|
"VGF2P8MULB": {2, 0xCF, 0, 1, -1, vexNDS3},
|
||||||
|
"VMOVNTDQA": {2, 0x2A, 0, 1, -1, vexRM},
|
||||||
|
"VMOVNTPD": {1, 0x2B, 0, 1, -1, vexRMRev},
|
||||||
|
"VORPD": {1, 0x56, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDSB": {1, 0xEC, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDSW": {1, 0xED, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDUSB": {1, 0xDC, 0, 1, -1, vexNDS3},
|
||||||
|
"VPADDUSW": {1, 0xDD, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPEQQ": {2, 0x29, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPEQW": {1, 0x75, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPGTB": {1, 0x64, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPGTD": {1, 0x66, 0, 1, -1, vexNDS3},
|
||||||
|
"VPCMPGTW": {1, 0x65, 0, 1, -1, vexNDS3},
|
||||||
|
"VPERMPS": {2, 0x16, 0, 1, -1, vexNDS3},
|
||||||
|
"VPEXTRB": {3, 0x14, 0, 1, -1, vexExtract},
|
||||||
|
"VPEXTRD": {3, 0x16, 0, 1, -1, vexExtract},
|
||||||
|
"VPEXTRQ": {3, 0x16, 1, 1, -1, vexExtract},
|
||||||
|
"VPINSRD": {3, 0x22, 0, 1, -1, vexNDS3Imm},
|
||||||
|
"VPINSRQ": {3, 0x22, 1, 1, -1, vexNDS3Imm},
|
||||||
|
"VPMULHRSW": {2, 0x0B, 0, 1, -1, vexNDS3},
|
||||||
|
"VPMULHW": {1, 0xE5, 0, 1, -1, vexNDS3},
|
||||||
|
"VPMULUDQ": {1, 0xF4, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSADBW": {1, 0xF6, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBSB": {1, 0xE8, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBSW": {1, 0xE9, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBUSB": {1, 0xD8, 0, 1, -1, vexNDS3},
|
||||||
|
"VPSUBUSW": {1, 0xD9, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKHBW": {1, 0x68, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKHQDQ": {1, 0x6D, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKHWD": {1, 0x69, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKLBW": {1, 0x60, 0, 1, -1, vexNDS3},
|
||||||
|
"VPUNPCKLWD": {1, 0x61, 0, 1, -1, vexNDS3},
|
||||||
|
"VSQRTPD": {1, 0x51, 0, 1, -1, vexRM},
|
||||||
|
"VSQRTSD": {1, 0x51, 0, 3, -1, vexNDS3},
|
||||||
|
"VSQRTSS": {1, 0x51, 0, 2, -1, vexNDS3},
|
||||||
|
"VUCOMISD": {1, 0x2E, 0, 1, -1, vexRM},
|
||||||
|
|
||||||
|
// VEX.0F.WIG, the plain-prefix single/double arithmetic and unpack
|
||||||
|
// spellings (no 66 prefix; WIG, so W = 0).
|
||||||
|
"VANDNPS": {1, 0x55, 0, 0, -1, vexNDS3},
|
||||||
|
"VANDPS": {1, 0x54, 0, 0, -1, vexNDS3},
|
||||||
|
"VORPS": {1, 0x56, 0, 0, -1, vexNDS3},
|
||||||
|
"VUNPCKLPS": {1, 0x14, 0, 0, -1, vexNDS3},
|
||||||
|
"VUNPCKHPS": {1, 0x15, 0, 0, -1, vexNDS3},
|
||||||
|
"VSQRTPS": {1, 0x51, 0, 0, -1, vexRM},
|
||||||
|
"VMOVNTPS": {1, 0x2B, 0, 0, -1, vexRMRev},
|
||||||
|
// VEX.128.66.0F, the scalar and packed compare forms.
|
||||||
|
"VCOMISS": {1, 0x2F, 0, 1, -1, vexRM},
|
||||||
|
"VUCOMISS": {1, 0x2E, 0, 0, -1, vexRM},
|
||||||
|
// VEX.128.0F.F3/F2.W0, the high/low word shuffles ($imm, src, dst).
|
||||||
|
"VPSHUFHW": {1, 0x70, 0, 2, -1, vexImmRM},
|
||||||
|
"VPSHUFLW": {1, 0x70, 0, 3, -1, vexImmRM},
|
||||||
}
|
}
|
||||||
|
|
||||||
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
// vexSrcLen maps a source-length conversion mnemonic (the X/Y spellings of
|
||||||
@@ -290,6 +501,8 @@ type vexMoveSpec struct {
|
|||||||
var vexMoveTable = map[string]vexMoveSpec{
|
var vexMoveTable = map[string]vexMoveSpec{
|
||||||
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
// VEX.128/256.F3.0F.WIG, unaligned integer move.
|
||||||
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
"VMOVDQU": {1, 2, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
|
// VEX.128/256.66.0F.WIG, aligned integer move.
|
||||||
|
"VMOVDQA": {1, 1, 0x6F, 0x7F, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
// VEX.128/256.66.0F.WIG, unaligned packed double move.
|
||||||
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
"VMOVUPD": {1, 1, 0x10, 0x11, 0, 0, 0, 0, true, false, false},
|
||||||
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
// VEX.128.66.0F.W0, 32-bit GPR/memory ↔ XMM.
|
||||||
@@ -324,6 +537,14 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
return fmt.Errorf("%s: vector register index %d needs an EVEX (AVX-512) instruction", mnemUpper, r.idx)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// VBROADCASTI128 broadcasts a 128-bit lane into a 256-bit destination
|
||||||
|
// only; an XMM destination is rejected exactly as go tool asm does.
|
||||||
|
if mnemUpper == "VBROADCASTI128" {
|
||||||
|
dstReg, ok := ops[len(ops)-1].(Reg)
|
||||||
|
if len(ops) != 2 || !ok || dstReg.size != 32 {
|
||||||
|
return fmt.Errorf("VBROADCASTI128 requires a YMM destination")
|
||||||
|
}
|
||||||
|
}
|
||||||
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
if ms, ok := vexMoveTable[mnemUpper]; ok {
|
||||||
return e.encodeVexMove(mnemUpper, ms, ops)
|
return e.encodeVexMove(mnemUpper, ms, ops)
|
||||||
}
|
}
|
||||||
@@ -356,6 +577,18 @@ func (e *enc) encodeVex(mnemUpper string, ops []Operand) error {
|
|||||||
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
return e.encodeVexRMSrcLen(mnemUpper, spec, ops)
|
||||||
case vexZero:
|
case vexZero:
|
||||||
return e.encodeVexZero(mnemUpper, spec, ops)
|
return e.encodeVexZero(mnemUpper, spec, ops)
|
||||||
|
case vexZeroAll:
|
||||||
|
return e.encodeVexZeroAll(mnemUpper, spec, ops)
|
||||||
|
case vexNDS3GPR:
|
||||||
|
return e.encodeVexNDS3GPR(spec, ops)
|
||||||
|
case vexImmRMGPR:
|
||||||
|
return e.encodeVexImmRMGPR(spec, ops)
|
||||||
|
case vexRMOpGPR:
|
||||||
|
return e.encodeVexRMOpGPR(spec, ops)
|
||||||
|
case vexCountGPR:
|
||||||
|
return e.encodeVexCountGPR(spec, ops)
|
||||||
|
case vexRMRev:
|
||||||
|
return e.encodeVexRMRev(spec, ops)
|
||||||
}
|
}
|
||||||
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
return fmt.Errorf("unhandled VEX form for %s", mnemUpper)
|
||||||
}
|
}
|
||||||
@@ -457,9 +690,10 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return fmt.Errorf("shift count must be an immediate")
|
return fmt.Errorf("shift count must be an immediate")
|
||||||
}
|
}
|
||||||
srcReg, ok := src.(Reg)
|
// The count source is a vector register or memory; the VEX length
|
||||||
if !ok || !srcReg.isVec() {
|
// follows the destination register either way.
|
||||||
return fmt.Errorf("shift source must be a vector register")
|
if !vecOrMem(src) {
|
||||||
|
return fmt.Errorf("shift source must be a vector register or memory")
|
||||||
}
|
}
|
||||||
dstReg, ok := dst.(Reg)
|
dstReg, ok := dst.(Reg)
|
||||||
if !ok || !dstReg.isVec() {
|
if !ok || !dstReg.isVec() {
|
||||||
@@ -467,7 +701,7 @@ func (e *enc) encodeVexShiftImm(spec vexSpec, ops []Operand) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
vvvvBar := 15 - (dstReg.idx & 15)
|
vvvvBar := 15 - (dstReg.idx & 15)
|
||||||
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, srcReg); err != nil {
|
if err := e.emitVexFields(spec, dstReg.vecLenBit(), spec.opdigit, 0, vvvvBar, src); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
immByte, err := imm8(int64(immVal))
|
immByte, err := imm8(int64(immVal))
|
||||||
@@ -607,6 +841,129 @@ func (e *enc) encodeVexZero(mnem string, spec vexSpec, ops []Operand) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeVexZeroAll encodes a no-operand instruction (VZEROALL), the L = 1
|
||||||
|
// twin of VZEROUPPER.
|
||||||
|
func (e *enc) encodeVexZeroAll(mnem string, spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return fmt.Errorf("%s expects no operands, got %d", mnem, len(ops))
|
||||||
|
}
|
||||||
|
// 2-byte VEX: R̄ = 1, v̄vvv = 1111 (unused), L = 1.
|
||||||
|
e.out = append(e.out, 0xC5, byte(1<<7|15<<3|1<<2|spec.pp), spec.opcode)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexNDS3GPR encodes the three-operand NDS form over general-purpose
|
||||||
|
// registers (ANDN, MULX): OP src2, src1, dst with reg = dst, vvvv = src1,
|
||||||
|
// rm = src2 and L = 0.
|
||||||
|
func (e *enc) encodeVexNDS3GPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("VEX NDS instruction expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src2, src1, dst := ops[0], ops[1], ops[2]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
vvvvReg, ok := src1.(Reg)
|
||||||
|
if !ok || vvvvReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX vvvv operand must be a general-purpose register")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if dstReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(vvvvReg.idx&15), src2)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexImmRMGPR encodes the immediate form over general-purpose
|
||||||
|
// registers (RORX): OP $imm, src, dst with reg = dst, rm = src, L = 0.
|
||||||
|
func (e *enc) encodeVexImmRMGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("instruction expects 3 operands ($imm, src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
imm, src, dst := ops[0], ops[1], ops[2]
|
||||||
|
immVal, ok := imm.(Imm)
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("shift control must be an immediate")
|
||||||
|
}
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
immByte, err := imm8(int64(immVal))
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := e.emitVexFields(spec, 0, dstReg.idx&7, 0, 15, src); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
e.out = append(e.out, immByte)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexRMOpGPR encodes the two-operand /digit form over general-purpose
|
||||||
|
// registers (BLSI, BLSMSK, BLSR): OP src, dst with ModRM.reg = /digit,
|
||||||
|
// ModRM.rm = src and VEX.vvvv = dst.
|
||||||
|
func (e *enc) encodeVexRMOpGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("instruction expects 2 operands (src, dst), got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, dst := ops[0], ops[1]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, spec.opdigit, 0, 15-(dstReg.idx&15), src)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexCountGPR encodes the three-operand count form over general-purpose
|
||||||
|
// registers (SHLX, SHRX, SARX, BEXTR, BZHI): OP src, count, dst with
|
||||||
|
// VEX.vvvv = src (op0), ModRM.rm = count (op1), ModRM.reg = dst (op2).
|
||||||
|
func (e *enc) encodeVexCountGPR(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 3 {
|
||||||
|
return fmt.Errorf("VEX count instruction expects 3 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
src, count, dst := ops[0], ops[1], ops[2]
|
||||||
|
dstReg, ok := dst.(Reg)
|
||||||
|
if !ok || dstReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX destination must be a general-purpose register")
|
||||||
|
}
|
||||||
|
countReg, ok := count.(Reg)
|
||||||
|
if !ok || countReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX count operand must be a general-purpose register")
|
||||||
|
}
|
||||||
|
srcReg, ok := src.(Reg)
|
||||||
|
if !ok || srcReg.isVec() {
|
||||||
|
return fmt.Errorf("VEX count source must be a general-purpose register")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if dstReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, 0, dstReg.idx&7, rBit, 15-(srcReg.idx&15), count)
|
||||||
|
}
|
||||||
|
|
||||||
|
// encodeVexRMRev encodes the reversed two-operand form: OP src, dst with the
|
||||||
|
// vector source in ModRM.reg and the memory destination in r/m (VMOVNTDQ,
|
||||||
|
// a store with no register-destination form).
|
||||||
|
func (e *enc) encodeVexRMRev(spec vexSpec, ops []Operand) error {
|
||||||
|
if len(ops) != 2 {
|
||||||
|
return fmt.Errorf("store expects 2 operands, got %d", len(ops))
|
||||||
|
}
|
||||||
|
srcReg, ok := ops[0].(Reg)
|
||||||
|
if !ok || !srcReg.isVec() {
|
||||||
|
return fmt.Errorf("store source must be a vector register")
|
||||||
|
}
|
||||||
|
if !memOperand(ops[1]) {
|
||||||
|
return fmt.Errorf("store destination must be memory")
|
||||||
|
}
|
||||||
|
rBit := 0
|
||||||
|
if srcReg.idx >= 8 {
|
||||||
|
rBit = 1
|
||||||
|
}
|
||||||
|
return e.emitVexFields(spec, srcReg.vecLenBit(), srcReg.idx&7, rBit, 15, ops[1])
|
||||||
|
}
|
||||||
|
|
||||||
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
// encodeVexMove encodes a two-operand move (VMOVDQU, VMOVUPD, VMOVD, VMOVQ,
|
||||||
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
// VMOVSD), picking the direction-specific opcode and VEX.W. A vector→vector
|
||||||
// move uses the store-form layout (reg = source, rm = destination), matching
|
// move uses the store-form layout (reg = source, rm = destination), matching
|
||||||
|
|||||||
+116
@@ -19,6 +19,65 @@ func vreg(t *testing.T, name string) Reg {
|
|||||||
return r
|
return r
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// x86asmUnrecognised lists the VEX mnemonics whose machine code the
|
||||||
|
// golang.org/x/arch decoder cannot resolve; their bytes are verified against
|
||||||
|
// go tool asm in the ground-truth tests instead.
|
||||||
|
var x86asmUnrecognised = map[string]bool{
|
||||||
|
"ANDNL": true,
|
||||||
|
"ANDNQ": true,
|
||||||
|
"MULXL": true,
|
||||||
|
"MULXQ": true,
|
||||||
|
"RORXL": true,
|
||||||
|
"RORXQ": true,
|
||||||
|
"VFMADD213SD": true,
|
||||||
|
"VFNMADD231SD": true,
|
||||||
|
// The scalar FMA spellings the decoder's tables lack entirely.
|
||||||
|
"VFMADD132SD": true,
|
||||||
|
"VFMADD132SS": true,
|
||||||
|
"VFMADD213SS": true,
|
||||||
|
"VFMADD231SD": true,
|
||||||
|
"VFMADD231SS": true,
|
||||||
|
"VFMSUB132SD": true,
|
||||||
|
"VFMSUB132SS": true,
|
||||||
|
"VFMSUB213SD": true,
|
||||||
|
"VFMSUB213SS": true,
|
||||||
|
"VFMSUB231SD": true,
|
||||||
|
"VFMSUB231SS": true,
|
||||||
|
"VFNMADD132SD": true,
|
||||||
|
"VFNMADD132SS": true,
|
||||||
|
"VFNMADD213SD": true,
|
||||||
|
"VFNMADD213SS": true,
|
||||||
|
"VFNMADD231SS": true,
|
||||||
|
"VFNMSUB132SD": true,
|
||||||
|
"VFNMSUB132SS": true,
|
||||||
|
"VFNMSUB213SD": true,
|
||||||
|
"VFNMSUB213SS": true,
|
||||||
|
"VFNMSUB231SD": true,
|
||||||
|
"VFNMSUB231SS": true,
|
||||||
|
// The BMI1 unary bit ops the decoder's AVX tables lack.
|
||||||
|
"BLSIL": true,
|
||||||
|
"BLSIQ": true,
|
||||||
|
"BLSMSKL": true,
|
||||||
|
"BLSMSKQ": true,
|
||||||
|
"BLSRL": true,
|
||||||
|
"BLSRQ": true,
|
||||||
|
// The BMI2 bit ops whose W1/LZ rows the decoder misses.
|
||||||
|
"BEXTRL": true,
|
||||||
|
"BEXTRQ": true,
|
||||||
|
"BZHIL": true,
|
||||||
|
"BZHIQ": true,
|
||||||
|
"PDEPL": true,
|
||||||
|
"PDEPQ": true,
|
||||||
|
"PEXTL": true,
|
||||||
|
"PEXTQ": true,
|
||||||
|
"SARXL": true,
|
||||||
|
"SARXQ": true,
|
||||||
|
"SHLXL": true,
|
||||||
|
"SHLXQ": true,
|
||||||
|
"SHRXL": true,
|
||||||
|
"SHRXQ": true,
|
||||||
|
}
|
||||||
|
|
||||||
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
// TestVexNDS3 encodes `mnem Y0, Y1, Y2` for every three-operand NDS
|
||||||
// instruction and verifies it round-trips through the x86 decoder to the same
|
// instruction and verifies it round-trips through the x86 decoder to the same
|
||||||
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
// mnemonic. A wrong opcode/map/pp surfaces as a different decoded instruction.
|
||||||
@@ -37,8 +96,15 @@ func TestVexNDS3(t *testing.T) {
|
|||||||
t.Errorf("%s: Encode: %v", mnem, err)
|
t.Errorf("%s: Encode: %v", mnem, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
// The x86 decoder's table lacks a handful of rows the Go assembler
|
||||||
|
// emits (the scalar 213/231 FMA spellings among them); those are
|
||||||
|
// pinned byte for byte against go tool asm in TestVexGroundTruth
|
||||||
|
// instead of round-tripped here.
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
if strings.Contains(err.Error(), "unrecognized instruction") && x86asmUnrecognised[mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
t.Errorf("%s: Decode(% x): %v", mnem, err, code)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
@@ -184,6 +250,50 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
{"VMULSD X0,X1,X1", "VMULSD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X1")}, "c5f359c8", ""},
|
||||||
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
{"VFMADD231PD Y14,Y12,Y8", "VFMADD231PD", []Operand{vreg(t, "Y14"), vreg(t, "Y12"), vreg(t, "Y8")}, "c4429db8c6", ""},
|
||||||
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
{"VFMADD231PD (DI),Y12,Y8", "VFMADD231PD", []Operand{Ptr(DI, 0, 32), vreg(t, "Y12"), vreg(t, "Y8")}, "c4629db807", ""},
|
||||||
|
{"VFMADD213SD X0,X1,X2", "VFMADD213SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1a9d0", ""},
|
||||||
|
{"VFNMADD231SD X0,X1,X2", "VFNMADD231SD", []Operand{vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e2f1bdd0", ""},
|
||||||
|
// Packed single XOR and byte compare (NDS form).
|
||||||
|
{"VXORPS Y0,Y1,Y2", "VXORPS", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f457d0", ""},
|
||||||
|
{"VPCMPEQB Y0,Y1,Y2", "VPCMPEQB", []Operand{vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c5f574d0", ""},
|
||||||
|
// Octa byte shifts (vvvv carries the destination).
|
||||||
|
{"VPSLLDQ $2,X0,X1", "VPSLLDQ", []Operand{Imm(2), vreg(t, "X0"), vreg(t, "X1")}, "c5f173f802", ""},
|
||||||
|
{"VPSRLDQ $2,Y0,Y1", "VPSRLDQ", []Operand{Imm(2), vreg(t, "Y0"), vreg(t, "Y1")}, "c5f573d802", ""},
|
||||||
|
// Two-source shuffle, blend and carry-less multiply (NDS + imm8).
|
||||||
|
{"VPERM2F128 $3,Y0,Y1,Y2", "VPERM2F128", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37506d003", ""},
|
||||||
|
{"VPBLENDD $3,X0,X1,X2", "VPBLENDD", []Operand{Imm(3), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37102d003", ""},
|
||||||
|
{"VPBLENDD $3,Y0,Y1,Y2", "VPBLENDD", []Operand{Imm(3), vreg(t, "Y0"), vreg(t, "Y1"), vreg(t, "Y2")}, "c4e37502d003", ""},
|
||||||
|
{"VPCLMULQDQ $0,X0,X1,X2", "VPCLMULQDQ", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e37144d000", ""},
|
||||||
|
{"VGF2P8AFFINEQB $0,X0,X1,X2", "VGF2P8AFFINEQB", []Operand{Imm(0), vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")}, "c4e3f1ced000", ""},
|
||||||
|
// Two-operand test and the non-temporal and broadcast stores.
|
||||||
|
{"VPTEST X0,X1", "VPTEST", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c4e27917c8", ""},
|
||||||
|
{"VPTEST Y0,Y1", "VPTEST", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c4e27d17c8", ""},
|
||||||
|
{"VMOVNTDQ Y0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "Y0"), Ptr(AX, 0, 32)}, "c5fde700", ""},
|
||||||
|
{"VMOVNTDQ X0,(AX)", "VMOVNTDQ", []Operand{vreg(t, "X0"), Ptr(AX, 0, 16)}, "c5f9e700", ""},
|
||||||
|
{"VBROADCASTI128 (AX),Y1", "VBROADCASTI128", []Operand{Ptr(AX, 0, 16), vreg(t, "Y1")}, "c4e27d5a08", ""},
|
||||||
|
// Aligned integer move and the full zeroing form.
|
||||||
|
{"VMOVDQA X0,X1", "VMOVDQA", []Operand{vreg(t, "X0"), vreg(t, "X1")}, "c5f97fc1", ""},
|
||||||
|
{"VMOVDQA (AX),X1", "VMOVDQA", []Operand{Ptr(AX, 0, 16), vreg(t, "X1")}, "c5f96f08", ""},
|
||||||
|
{"VMOVDQA Y0,Y1", "VMOVDQA", []Operand{vreg(t, "Y0"), vreg(t, "Y1")}, "c5fd7fc1", ""},
|
||||||
|
{"VZEROALL", "VZEROALL", []Operand{}, "c5fc77", ""},
|
||||||
|
// BMI1/BMI2 general-register VEX forms.
|
||||||
|
{"ANDNL AX,BX,CX", "ANDNL", []Operand{AX, BX, CX}, "c4e260f2c8", ""},
|
||||||
|
{"ANDNQ AX,BX,CX", "ANDNQ", []Operand{AX, BX, CX}, "c4e2e0f2c8", ""},
|
||||||
|
{"MULXL AX,BX,CX", "MULXL", []Operand{AX, BX, CX}, "c4e263f6c8", ""},
|
||||||
|
{"MULXQ AX,BX,CX", "MULXQ", []Operand{AX, BX, CX}, "c4e2e3f6c8", ""},
|
||||||
|
{"RORXL $3,AX,CX", "RORXL", []Operand{Imm(3), AX, CX}, "c4e37bf0c803", ""},
|
||||||
|
{"RORXQ $3,AX,CX", "RORXQ", []Operand{Imm(3), AX, CX}, "c4e3fbf0c803", ""},
|
||||||
|
// BMI2 variable shifts and bit ops (three general registers).
|
||||||
|
{"SHLXL AX,CX,R15", "SHLXL", []Operand{AX, CX, vreg(t, "R15")}, "c46279f7f9", ""},
|
||||||
|
{"SHRXQ R8,DX,AX", "SHRXQ", []Operand{vreg(t, "R8"), DX, AX}, "c4e2bbf7c2", ""},
|
||||||
|
{"SARXQ AX,DX,R9", "SARXQ", []Operand{AX, DX, vreg(t, "R9")}, "c462faf7ca", ""},
|
||||||
|
{"BEXTRL AX,CX,R15", "BEXTRL", []Operand{AX, CX, vreg(t, "R15")}, "c46278f7f9", ""},
|
||||||
|
{"BZHIQ AX,CX,R15", "BZHIQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f8f5f9", ""},
|
||||||
|
{"PDEPQ AX,CX,R15", "PDEPQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f3f5f8", ""},
|
||||||
|
{"PEXTQ AX,CX,R15", "PEXTQ", []Operand{AX, CX, vreg(t, "R15")}, "c462f2f5f8", ""},
|
||||||
|
// BMI1 unary bit ops (src, dst: /digit in ModRM.reg, dst in vvvv).
|
||||||
|
{"BLSIL AX,CX", "BLSIL", []Operand{AX, CX}, "c4e270f3d8", ""},
|
||||||
|
{"BLSRQ AX,CX", "BLSRQ", []Operand{AX, CX}, "c4e2f0f3c8", ""},
|
||||||
|
{"BLSMSKQ AX,CX", "BLSMSKQ", []Operand{AX, CX}, "c4e2f0f3d0", ""},
|
||||||
// Two-operand reg/rm form (v̄vvv must be 1111).
|
// Two-operand reg/rm form (v̄vvv must be 1111).
|
||||||
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
{"VPMOVSXDQ X0,Y4", "VPMOVSXDQ", []Operand{vreg(t, "X0"), vreg(t, "Y4")}, "c4e27d25e0", ""},
|
||||||
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
{"VPMOVSXWD (SI),Y0", "VPMOVSXWD", []Operand{Ptr(SI, 0, 8), vreg(t, "Y0")}, "c4e27d2306", ""},
|
||||||
@@ -287,6 +397,12 @@ func TestVexGroundTruth(t *testing.T) {
|
|||||||
}
|
}
|
||||||
inst, err := x86asm.Decode(code, 64)
|
inst, err := x86asm.Decode(code, 64)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
// The decoder's AVX/BMI table lacks a few rows the Go
|
||||||
|
// assembler emits (the GPR VEX forms and the scalar FMA
|
||||||
|
// spellings); their bytes are the ground truth here.
|
||||||
|
if x86asmUnrecognised[c.mnem] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
t.Errorf("%s: Decode(% x): %v", c.name, code, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
+18
-8
@@ -8,7 +8,7 @@
|
|||||||
// to the arch package; the AST records syntax only.
|
// to the arch package; the AST records syntax only.
|
||||||
package ast
|
package ast
|
||||||
|
|
||||||
import "sourcedock.dev/petrbalvin/gasm-devkit/token"
|
import "sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||||
|
|
||||||
// File is the parsed representation of one .s source file.
|
// File is the parsed representation of one .s source file.
|
||||||
type File struct {
|
type File struct {
|
||||||
@@ -151,11 +151,21 @@ type Immediate struct {
|
|||||||
// Address is a non-immediate operand: a register, a memory reference, a symbol
|
// Address is a non-immediate operand: a register, a memory reference, a symbol
|
||||||
// reference or a label. Fields are populated best-effort from the syntax.
|
// reference or a label. Fields are populated best-effort from the syntax.
|
||||||
type Address struct {
|
type Address struct {
|
||||||
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
Sym *Symbol // name reference (bare ident, or name+off(pseudo))
|
||||||
Base string // base register, from (base)
|
Base string // base register, from (base)
|
||||||
Index string // index register, from (index*scale)
|
Index string // index register, from (index*scale)
|
||||||
Scale int // index scale; 0 when absent
|
Scale int // index scale; 0 when absent
|
||||||
Offset int64 // leading displacement, from off(base)
|
Offset int64 // leading displacement, from off(base)
|
||||||
HasOff bool // a leading displacement is present
|
HasOff bool // a leading displacement is present
|
||||||
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
|
Shift string // verbatim arm64 shift suffix, e.g. "<< 2"
|
||||||
|
Range *RegRange // bracketed register range; nil for every other form
|
||||||
|
}
|
||||||
|
|
||||||
|
// RegRange is a bracketed register range, [Z0-Z3]: the amd64 spelling of
|
||||||
|
// the four-register source of the 4FMAPS/4VNNIW families. Lo and Hi carry
|
||||||
|
// the verbatim register spellings; the range is inclusive at both ends.
|
||||||
|
type RegRange struct {
|
||||||
|
Lo string
|
||||||
|
Hi string
|
||||||
|
Pos token.Position
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -6,7 +6,7 @@ package ast
|
|||||||
import (
|
import (
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
"sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||||
)
|
)
|
||||||
|
|
||||||
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
|
func pos(line, col int) token.Position { return token.Position{Line: line, Column: col} }
|
||||||
|
|||||||
@@ -0,0 +1,367 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"go/ast"
|
||||||
|
"go/build"
|
||||||
|
"go/constant"
|
||||||
|
"go/parser"
|
||||||
|
"go/token"
|
||||||
|
"go/types"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
)
|
||||||
|
|
||||||
|
// go_asm.h is the header the Go compiler writes for every package that
|
||||||
|
// carries assembly (the compiler's -asmhdr output): "#define const_NAME
|
||||||
|
// value" for each package constant, and for each named struct type
|
||||||
|
// "#define TYPE__size size" plus one "#define TYPE_field offset" per field.
|
||||||
|
// GOROOT assembly includes it, and a standalone assembler has no compiler
|
||||||
|
// to have produced it, so gasm generates the equivalent itself: the package
|
||||||
|
// the .s file lives in is parsed and type-checked here, with the target
|
||||||
|
// architecture's own sizes, and the same defines are written out. The
|
||||||
|
// type-checking GOOS is selected by the caller: a GOOS-specific file
|
||||||
|
// (sys_darwin_arm64.s) needs its platform's defines, which a header from
|
||||||
|
// the ambient GOOS silently omits.
|
||||||
|
//
|
||||||
|
// The emitter mirrors cmd/compile's dumpasmhdr exactly: constants come out
|
||||||
|
// as "const_NAME", struct entries as "NAME__size" followed by the fields in
|
||||||
|
// declaration order, blank names are skipped, and float and complex
|
||||||
|
// constants are omitted (the assembler carries integers, bools and strings
|
||||||
|
// only). Aliases to structs are emitted, generic types are not: they have
|
||||||
|
// no fixed size. A define the assembly references but this header does not
|
||||||
|
// carry surfaces later as the assembler's own "undefined" diagnostic naming
|
||||||
|
// the define, which is the honest failure.
|
||||||
|
|
||||||
|
// goAsmInclude matches the #include "go_asm.h" directive, tolerant of
|
||||||
|
// whitespace, so the wiring knows which files need a generated header
|
||||||
|
// before the preprocessor runs and would report the header as missing.
|
||||||
|
var goAsmInclude = regexp.MustCompile(`(?m)^\s*#\s*include\s+"go_asm\.h"`)
|
||||||
|
|
||||||
|
// needsGoAsmHeader reports whether src includes go_asm.h.
|
||||||
|
func needsGoAsmHeader(src string) bool {
|
||||||
|
return goAsmInclude.MatchString(src)
|
||||||
|
}
|
||||||
|
|
||||||
|
// goAsmHeaderResolved reports whether the include of go_asm.h from a file in
|
||||||
|
// asmDir already resolves: to a header in the package directory itself, or
|
||||||
|
// in one of the -I directories, the way the preprocessor searches. Only an
|
||||||
|
// unresolved include is generated for; a header someone placed by hand is
|
||||||
|
// the tool the author chose, and it also wins the preprocessor's own search
|
||||||
|
// order, so generating a second copy would be dead weight at best.
|
||||||
|
func goAsmHeaderResolved(asmDir string, dirs []string) bool {
|
||||||
|
candidates := []string{filepath.Join(asmDir, "go_asm.h")}
|
||||||
|
for _, d := range dirs {
|
||||||
|
candidates = append(candidates, filepath.Join(d, "go_asm.h"))
|
||||||
|
}
|
||||||
|
for _, candidate := range candidates {
|
||||||
|
if st, err := os.Stat(candidate); err == nil && !st.IsDir() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// generateGoAsmHeader type-checks the Go package in pkgDir for goos and
|
||||||
|
// goarch, writes its go_asm.h equivalent into dir, and returns dir. An
|
||||||
|
// empty goos means the ambient one. The caller owns the directory and its
|
||||||
|
// removal.
|
||||||
|
func generateGoAsmHeader(pkgDir, goos, goarch, dir string) (string, error) {
|
||||||
|
if goos == "" {
|
||||||
|
goos = build.Default.GOOS
|
||||||
|
}
|
||||||
|
imp := newSourceImporter(goos, goarch)
|
||||||
|
if imp.sizes == nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h: unknown GOARCH %q", goarch)
|
||||||
|
}
|
||||||
|
bp, err := imp.ctxt.ImportDir(pkgDir, 0)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||||
|
}
|
||||||
|
files, errs := imp.parse(bp)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %s", goarch, pkgDir, errorList(errs))
|
||||||
|
}
|
||||||
|
_, info, errs := imp.checkPackage(bp, files)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: package does not type-check: %s", goarch, pkgDir, errorList(errs))
|
||||||
|
}
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
fmt.Fprintf(&b, "// generated by gasm from package %s (GOOS %s, GOARCH %s)\n\n", bp.Name, goos, goarch)
|
||||||
|
// Files in the build's own order and declarations in source order: the
|
||||||
|
// same walk the compiler's reader makes, so the header reads the same
|
||||||
|
// way the toolchain's does. Order carries no meaning to the assembler
|
||||||
|
// (defines form a table), only to a human diffing against one.
|
||||||
|
for _, f := range files {
|
||||||
|
for _, decl := range f.Decls {
|
||||||
|
gd, ok := decl.(*ast.GenDecl)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, spec := range gd.Specs {
|
||||||
|
switch gd.Tok {
|
||||||
|
case token.CONST:
|
||||||
|
vs, ok := spec.(*ast.ValueSpec)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, name := range vs.Names {
|
||||||
|
emitConst(&b, info.Defs[name], name.Name)
|
||||||
|
}
|
||||||
|
case token.TYPE:
|
||||||
|
ts, ok := spec.(*ast.TypeSpec)
|
||||||
|
if !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
emitStruct(&b, imp.sizes, info.Defs[ts.Name], ts.Name.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||||
|
}
|
||||||
|
out := filepath.Join(dir, "go_asm.h")
|
||||||
|
if err := os.WriteFile(out, []byte(b.String()), 0o644); err != nil {
|
||||||
|
return "", fmt.Errorf("go_asm.h for GOARCH %s in %s: %w", goarch, pkgDir, err)
|
||||||
|
}
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// emitConst writes one const define, skipping what the toolchain skips:
|
||||||
|
// blank names, and float and complex values the assembler has no syntax for.
|
||||||
|
func emitConst(b *strings.Builder, obj types.Object, name string) {
|
||||||
|
c, ok := obj.(*types.Const)
|
||||||
|
if !ok || name == "_" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
switch c.Val().Kind() {
|
||||||
|
case constant.Float, constant.Complex, constant.Unknown:
|
||||||
|
return
|
||||||
|
}
|
||||||
|
fmt.Fprintf(b, "#define const_%s %s\n", name, c.Val().ExactString())
|
||||||
|
}
|
||||||
|
|
||||||
|
// emitStruct writes one named struct type's size and field offsets,
|
||||||
|
// skipping what the toolchain skips: blank names, non-struct types, and
|
||||||
|
// generic types, whose size depends on their instantiation.
|
||||||
|
func emitStruct(b *strings.Builder, sizes types.Sizes, obj types.Object, name string) {
|
||||||
|
tn, ok := obj.(*types.TypeName)
|
||||||
|
if !ok || name == "_" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t := types.Unalias(tn.Type())
|
||||||
|
// Generic types are spelled *types.Named with a type-parameter list;
|
||||||
|
// a plain struct type or an instantiated one carries none.
|
||||||
|
if named, ok := t.(*types.Named); ok && named.TypeParams().Len() > 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
st, ok := t.Underlying().(*types.Struct)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
fmt.Fprintf(b, "#define %s__size %d\n", name, sizes.Sizeof(t))
|
||||||
|
fields := make([]*types.Var, st.NumFields())
|
||||||
|
for i := range st.NumFields() {
|
||||||
|
fields[i] = st.Field(i)
|
||||||
|
}
|
||||||
|
for i, off := range sizes.Offsetsof(fields) {
|
||||||
|
fld := fields[i]
|
||||||
|
if fld.Name() == "_" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(b, "#define %s_%s %d\n", name, fld.Name(), off)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// errorList renders at most three errors, enough to say what is wrong
|
||||||
|
// without burying the diagnostic the caller actually reads.
|
||||||
|
func errorList(errs []error) string {
|
||||||
|
if len(errs) > 3 {
|
||||||
|
errs = errs[:3]
|
||||||
|
}
|
||||||
|
msgs := make([]string, len(errs))
|
||||||
|
for i, err := range errs {
|
||||||
|
msgs[i] = err.Error()
|
||||||
|
}
|
||||||
|
return strings.Join(msgs, "; ")
|
||||||
|
}
|
||||||
|
|
||||||
|
// sourceImporter type-checks imported packages from source with the target
|
||||||
|
// architecture's sizes. go/importer's "source" importer pins the host
|
||||||
|
// GOARCH, which would lay out imported types (internal/cpu, internal/abi)
|
||||||
|
// for the wrong target on a cross-architecture header, so the recursion is
|
||||||
|
// carried here with one build context and one sizes instance per
|
||||||
|
// architecture.
|
||||||
|
type sourceImporter struct {
|
||||||
|
fset *token.FileSet
|
||||||
|
ctxt *build.Context
|
||||||
|
sizes types.Sizes
|
||||||
|
pkgs map[string]*types.Package
|
||||||
|
}
|
||||||
|
|
||||||
|
// newSourceImporter returns the importer for one target GOOS and GOARCH.
|
||||||
|
// Cgo is disabled so the file set is deterministic and independent of the
|
||||||
|
// host's C toolchain: cgo-tagged files drop out of the build exactly as
|
||||||
|
// they do from a CGO_ENABLED=0 build, whose assembly is what gasm targets.
|
||||||
|
func newSourceImporter(goos, goarch string) *sourceImporter {
|
||||||
|
ctxt := new(build.Context)
|
||||||
|
*ctxt = build.Default
|
||||||
|
ctxt.GOOS = goos
|
||||||
|
ctxt.GOARCH = goarch
|
||||||
|
ctxt.CgoEnabled = false
|
||||||
|
return &sourceImporter{
|
||||||
|
fset: token.NewFileSet(),
|
||||||
|
ctxt: ctxt,
|
||||||
|
sizes: types.SizesFor("gc", goarch),
|
||||||
|
pkgs: map[string]*types.Package{},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Import type-checks one imported package and memoises it. "unsafe" must
|
||||||
|
// resolve to go/types' own package, never to the source in GOROOT/src/unsafe:
|
||||||
|
// the source declares Sizeof and Offsetof as ordinary functions over
|
||||||
|
// ArbitraryType, and checking against that signature rejects half the
|
||||||
|
// unsafe arithmetic the gc compiler accepts, which is exactly the divergence
|
||||||
|
// srcimporter guards against the same way.
|
||||||
|
func (im *sourceImporter) Import(path string) (*types.Package, error) {
|
||||||
|
if path == "unsafe" {
|
||||||
|
return types.Unsafe, nil
|
||||||
|
}
|
||||||
|
if p, ok := im.pkgs[path]; ok {
|
||||||
|
return p, nil
|
||||||
|
}
|
||||||
|
bp, err := im.ctxt.Import(path, "", 0)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
files, errs := im.parse(bp)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return nil, errors.New(errorList(errs))
|
||||||
|
}
|
||||||
|
pkg, _, _ := im.checkPackage(bp, files)
|
||||||
|
im.pkgs[path] = pkg
|
||||||
|
return pkg, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// parse reads the build package's Go files. Import-level failures (no Go
|
||||||
|
// files for the target, unreadable files) come back as errors, and the
|
||||||
|
// type-check decides the rest.
|
||||||
|
func (im *sourceImporter) parse(bp *build.Package) ([]*ast.File, []error) {
|
||||||
|
if len(bp.GoFiles) == 0 {
|
||||||
|
return nil, []error{fmt.Errorf("no Go source files for GOOS=%s GOARCH=%s", im.ctxt.GOOS, im.ctxt.GOARCH)}
|
||||||
|
}
|
||||||
|
var (
|
||||||
|
files []*ast.File
|
||||||
|
errs []error
|
||||||
|
)
|
||||||
|
for _, name := range bp.GoFiles {
|
||||||
|
f, err := parser.ParseFile(im.fset, filepath.Join(bp.Dir, name), nil, parser.SkipObjectResolution)
|
||||||
|
if err != nil {
|
||||||
|
errs = append(errs, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
files = append(files, f)
|
||||||
|
}
|
||||||
|
return files, errs
|
||||||
|
}
|
||||||
|
|
||||||
|
// checkPackage type-checks one package's files with the importer's sizes,
|
||||||
|
// recording every error: a header from a package that does not type-check
|
||||||
|
// could silently mis-state an offset, so the caller refuses the header
|
||||||
|
// rather than trusting it. The returned Defs map backs the root package's
|
||||||
|
// emission walk; imports only need the checked package itself.
|
||||||
|
func (im *sourceImporter) checkPackage(bp *build.Package, files []*ast.File) (*types.Package, *types.Info, []error) {
|
||||||
|
var errs []error
|
||||||
|
conf := &types.Config{
|
||||||
|
Importer: im,
|
||||||
|
Sizes: im.sizes,
|
||||||
|
Error: func(err error) { errs = append(errs, err) },
|
||||||
|
}
|
||||||
|
info := &types.Info{Defs: map[*ast.Ident]types.Object{}}
|
||||||
|
pkg, _ := conf.Check(bp.ImportPath, im.fset, files, info)
|
||||||
|
return pkg, info, errs
|
||||||
|
}
|
||||||
|
|
||||||
|
// asmhdrCache generates one go_asm.h per package directory and target
|
||||||
|
// architecture under one temp root, for callers that assemble many files
|
||||||
|
// (the corpus audit). Failures are cached too: a package that does not
|
||||||
|
// type-check must not be re-checked once per file.
|
||||||
|
type asmhdrCache struct {
|
||||||
|
root string
|
||||||
|
dirs map[string]string // "pkgDir\x00goos\x00goarch" -> directory holding go_asm.h
|
||||||
|
errs map[string]error
|
||||||
|
}
|
||||||
|
|
||||||
|
func newAsmhdrCache() (*asmhdrCache, error) {
|
||||||
|
root, err := os.MkdirTemp("", "gasm-asmhdr")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return &asmhdrCache{root: root, dirs: map[string]string{}, errs: map[string]error{}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// dirFor returns the directory holding the generated go_asm.h for pkgDir
|
||||||
|
// under goos and goarch, generating it on first use. An empty goos means
|
||||||
|
// the ambient one, resolved here so that one package cannot generate twice
|
||||||
|
// under an explicit and an implicit spelling of the same GOOS.
|
||||||
|
func (c *asmhdrCache) dirFor(pkgDir, goos, goarch string) (string, error) {
|
||||||
|
if goos == "" {
|
||||||
|
goos = build.Default.GOOS
|
||||||
|
}
|
||||||
|
key := pkgDir + "\x00" + goos + "\x00" + goarch
|
||||||
|
if dir, ok := c.dirs[key]; ok {
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
if err, ok := c.errs[key]; ok {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
dir := filepath.Join(c.root, fmt.Sprintf("h%d_%s_%s", len(c.dirs), goos, goarch))
|
||||||
|
if _, err := generateGoAsmHeader(pkgDir, goos, goarch, dir); err != nil {
|
||||||
|
c.errs[key] = err
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
c.dirs[key] = dir
|
||||||
|
return dir, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// close removes the temp root.
|
||||||
|
func (c *asmhdrCache) close() { os.RemoveAll(c.root) }
|
||||||
|
|
||||||
|
// ensureGoAsmHeader prepares the include directory a file that includes
|
||||||
|
// go_asm.h needs: the generated header for the package in path's directory,
|
||||||
|
// for the file's target GOOS and architecture. It reports a usage error
|
||||||
|
// when the architecture cannot be determined, and passes through the
|
||||||
|
// generator's diagnostics, which name the package.
|
||||||
|
func ensureGoAsmHeader(path string, target arch.Arch, goos string, cache *asmhdrCache) (string, func(), error) {
|
||||||
|
if path == "-" {
|
||||||
|
return "", nil, errors.New("cannot generate go_asm.h for standard input (no package directory)")
|
||||||
|
}
|
||||||
|
if target == arch.Unknown {
|
||||||
|
return "", nil, errors.New("a file that includes go_asm.h needs a target architecture: name the file _<arch>.s or pass -GOARCH")
|
||||||
|
}
|
||||||
|
if cache != nil {
|
||||||
|
dir, err := cache.dirFor(filepath.Dir(path), goos, goarchName(target))
|
||||||
|
return dir, func() {}, err
|
||||||
|
}
|
||||||
|
root, err := os.MkdirTemp("", "gasm-asmhdr")
|
||||||
|
if err != nil {
|
||||||
|
return "", nil, err
|
||||||
|
}
|
||||||
|
dir, err := generateGoAsmHeader(filepath.Dir(path), goos, goarchName(target), root)
|
||||||
|
if err != nil {
|
||||||
|
os.RemoveAll(root)
|
||||||
|
return "", nil, err
|
||||||
|
}
|
||||||
|
return dir, func() { os.RemoveAll(root) }, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,476 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// writePkg lays out a minimal Go package in a temp directory.
|
||||||
|
func writePkg(t *testing.T, files map[string]string) string {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
for name, src := range files {
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dir
|
||||||
|
}
|
||||||
|
|
||||||
|
// generateFor generates the header for dir and returns its text. An empty
|
||||||
|
// goos means the ambient one.
|
||||||
|
func generateFor(t *testing.T, dir, goos, goarch string) string {
|
||||||
|
t.Helper()
|
||||||
|
hdrDir, err := generateGoAsmHeader(dir, goos, goarch, t.TempDir())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("generateGoAsmHeader(%q, %s, %s): %v", dir, goos, goarch, err)
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(filepath.Join(hdrDir, "go_asm.h"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return string(b)
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGenerateGoAsmHeaderShape(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{"sample.go": `package sample
|
||||||
|
|
||||||
|
const bufSize = 1024
|
||||||
|
|
||||||
|
const (
|
||||||
|
a = iota * 8
|
||||||
|
b
|
||||||
|
c
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
strConst = "hello"
|
||||||
|
boolConst = true
|
||||||
|
floatConst = 1.5
|
||||||
|
_ = "the blank identifier is skipped"
|
||||||
|
)
|
||||||
|
|
||||||
|
const shift = 1 << 20
|
||||||
|
|
||||||
|
type reader struct {
|
||||||
|
r int64
|
||||||
|
w int64
|
||||||
|
_ [4]byte
|
||||||
|
name string
|
||||||
|
}
|
||||||
|
|
||||||
|
type scalar int
|
||||||
|
|
||||||
|
type aliased struct {
|
||||||
|
k uint32
|
||||||
|
v uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
type alias = aliased
|
||||||
|
`})
|
||||||
|
hdr := generateFor(t, dir, "", "amd64")
|
||||||
|
want := []string{
|
||||||
|
"#define const_bufSize 1024",
|
||||||
|
// iota resolves through go/types, one define per name.
|
||||||
|
"#define const_a 0",
|
||||||
|
"#define const_b 8",
|
||||||
|
"#define const_c 16",
|
||||||
|
`#define const_strConst "hello"`,
|
||||||
|
"#define const_boolConst true",
|
||||||
|
// Floats are the toolchain's own skip, as are blank names.
|
||||||
|
"#define const_shift 1048576",
|
||||||
|
// The blank field still occupies its bytes: the pad after w runs to
|
||||||
|
// the string's 8-byte alignment.
|
||||||
|
"#define reader__size 40",
|
||||||
|
"#define reader_r 0",
|
||||||
|
"#define reader_w 8",
|
||||||
|
"#define reader_name 24",
|
||||||
|
// Non-struct named types carry no defines; aliases to structs do.
|
||||||
|
"#define aliased__size 8",
|
||||||
|
"#define aliased_k 0",
|
||||||
|
"#define aliased_v 4",
|
||||||
|
"#define alias__size 8",
|
||||||
|
"#define alias_k 0",
|
||||||
|
"#define alias_v 4",
|
||||||
|
}
|
||||||
|
for _, w := range want {
|
||||||
|
if !strings.Contains(hdr, w+"\n") {
|
||||||
|
t.Errorf("header misses %q\ngot:\n%s", w, hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, banned := range []string{"#define const_floatConst", "#define _ ", "#define scalar"} {
|
||||||
|
if strings.Contains(hdr, banned) {
|
||||||
|
t.Errorf("header must not carry %s\ngot:\n%s", banned, hdr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGenerateGoAsmHeaderPerArch(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{
|
||||||
|
"common.go": `package perarch
|
||||||
|
|
||||||
|
type layout struct {
|
||||||
|
a int32
|
||||||
|
p uintptr
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
// The build-tagged file set is part of the contract: a per-arch
|
||||||
|
// package is exactly how internal/cpu declares its layouts.
|
||||||
|
"const_amd64.go": `//go:build amd64
|
||||||
|
|
||||||
|
package perarch
|
||||||
|
|
||||||
|
const flavour = 1
|
||||||
|
`,
|
||||||
|
"const_arm64.go": `//go:build arm64
|
||||||
|
|
||||||
|
package perarch
|
||||||
|
|
||||||
|
const flavour = 2
|
||||||
|
`,
|
||||||
|
})
|
||||||
|
amd64 := generateFor(t, dir, "", "amd64")
|
||||||
|
arm64 := generateFor(t, dir, "", "arm64")
|
||||||
|
if !strings.Contains(amd64, "#define const_flavour 1\n") {
|
||||||
|
t.Errorf("amd64 header misses const_flavour 1:\n%s", amd64)
|
||||||
|
}
|
||||||
|
if !strings.Contains(arm64, "#define const_flavour 2\n") {
|
||||||
|
t.Errorf("arm64 header misses const_flavour 2:\n%s", arm64)
|
||||||
|
}
|
||||||
|
if strings.Contains(arm64, "#define const_flavour 1\n") {
|
||||||
|
t.Errorf("arm64 header must not carry the amd64 file's value")
|
||||||
|
}
|
||||||
|
// SizesFor makes the layout the target's: uintptr is 4 bytes wide on
|
||||||
|
// 386 and 8 on amd64, which must move p and grow the struct.
|
||||||
|
if !strings.Contains(amd64, "#define layout__size 16\n") || !strings.Contains(amd64, "#define layout_p 8\n") {
|
||||||
|
t.Errorf("amd64 layout wrong:\n%s", amd64)
|
||||||
|
}
|
||||||
|
w386 := generateFor(t, dir, "", "386")
|
||||||
|
if !strings.Contains(w386, "#define layout__size 8\n") || !strings.Contains(w386, "#define layout_p 4\n") {
|
||||||
|
t.Errorf("386 layout wrong:\n%s", w386)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGenerateGoAsmHeaderGOOS pins the GOOS half of the target: only the
|
||||||
|
// platform's own files type-check into the header, which is why
|
||||||
|
// sys_darwin_arm64.s cannot assemble against a linux-generated one.
|
||||||
|
func TestGenerateGoAsmHeaderGOOS(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{
|
||||||
|
"common.go": `package goosaware
|
||||||
|
|
||||||
|
type shared struct {
|
||||||
|
a int32
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
"plat_darwin.go": `//go:build darwin
|
||||||
|
|
||||||
|
package goosaware
|
||||||
|
|
||||||
|
type platform struct {
|
||||||
|
trampoline_numer int64
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
"plat_windows.go": `//go:build windows
|
||||||
|
|
||||||
|
package goosaware
|
||||||
|
|
||||||
|
type platform struct {
|
||||||
|
callbackArgs__size int32
|
||||||
|
}
|
||||||
|
`,
|
||||||
|
})
|
||||||
|
darwin := generateFor(t, dir, "darwin", "arm64")
|
||||||
|
if !strings.Contains(darwin, "#define platform__size 8\n") || !strings.Contains(darwin, "#define platform_trampoline_numer 0\n") {
|
||||||
|
t.Errorf("darwin header misses the darwin layout:\n%s", darwin)
|
||||||
|
}
|
||||||
|
if strings.Contains(darwin, "callbackArgs") {
|
||||||
|
t.Errorf("darwin header must not carry the windows layout:\n%s", darwin)
|
||||||
|
}
|
||||||
|
windows := generateFor(t, dir, "windows", "arm64")
|
||||||
|
if !strings.Contains(windows, "#define platform_callbackArgs__size 0\n") {
|
||||||
|
t.Errorf("windows header misses the windows layout:\n%s", windows)
|
||||||
|
}
|
||||||
|
if strings.Contains(windows, "trampoline_numer") {
|
||||||
|
t.Errorf("windows header must not carry the darwin layout:\n%s", windows)
|
||||||
|
}
|
||||||
|
// The ambient GOOS is neither of the two, so only shared's defines are
|
||||||
|
// emitted; the shared type keeps its layout there.
|
||||||
|
ambient := generateFor(t, dir, "", "arm64")
|
||||||
|
if !strings.Contains(ambient, "#define shared__size 4\n") {
|
||||||
|
t.Errorf("ambient header misses the shared layout:\n%s", ambient)
|
||||||
|
}
|
||||||
|
if strings.Contains(ambient, "#define platform_") {
|
||||||
|
t.Errorf("ambient header must not carry either platform layout:\n%s", ambient)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGoosFromFilename(t *testing.T) {
|
||||||
|
for path, want := range map[string]string{
|
||||||
|
"/x/sys_darwin_arm64.s": "darwin",
|
||||||
|
"/x/sys_windows_arm64.s": "windows",
|
||||||
|
"/x/asm_linux_amd64.s": "linux",
|
||||||
|
"/x/rt0_darwin_arm64.s": "darwin",
|
||||||
|
"/x/vgetrandom_zos_s390x.s": "zos",
|
||||||
|
"/x/rt0_js_wasm.s": "js",
|
||||||
|
"/x/memmove_amd64.s": "",
|
||||||
|
"/x/vlop_arm.s": "",
|
||||||
|
"/x/stubs.s": "",
|
||||||
|
} {
|
||||||
|
if got := goosFromFilename(path); got != want {
|
||||||
|
t.Errorf("goosFromFilename(%q) = %q, want %q", path, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGenerateGoAsmHeaderErrors(t *testing.T) {
|
||||||
|
t.Run("type error", func(t *testing.T) {
|
||||||
|
dir := writePkg(t, map[string]string{"bad.go": `package bad
|
||||||
|
|
||||||
|
const x = undefinedIdent
|
||||||
|
`})
|
||||||
|
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("generation must fail for a package that does not type-check")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), dir) {
|
||||||
|
t.Errorf("error must name the package directory: %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "type-check") {
|
||||||
|
t.Errorf("error must say the package does not type-check: %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
t.Run("no go files", func(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
_, err := generateGoAsmHeader(dir, "", "amd64", t.TempDir())
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("generation must fail without Go files")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), dir) {
|
||||||
|
t.Errorf("error must name the package directory: %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNeedsGoAsmHeader(t *testing.T) {
|
||||||
|
yes := "#include \"go_asm.h\"\n#include \"textflag.h\"\n"
|
||||||
|
no := "#include \"textflag.h\"\n#include \"funcdata.h\"\n"
|
||||||
|
if !needsGoAsmHeader(yes) {
|
||||||
|
t.Error("needsGoAsmHeader(missing on a go_asm.h include)")
|
||||||
|
}
|
||||||
|
if needsGoAsmHeader(no) {
|
||||||
|
t.Error("needsGoAsmHeader claims other headers need generation")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestGoAsmHeaderResolved(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
if goAsmHeaderResolved(dir, nil) {
|
||||||
|
t.Error("resolved with no header anywhere")
|
||||||
|
}
|
||||||
|
other := t.TempDir()
|
||||||
|
if goAsmHeaderResolved(dir, []string{other}) {
|
||||||
|
t.Error("resolved with an empty -I directory")
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, "go_asm.h"), nil, 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !goAsmHeaderResolved(dir, nil) {
|
||||||
|
t.Error("not resolved with the header in the package directory")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOtherGOOSFile(t *testing.T) {
|
||||||
|
for path, want := range map[string]bool{
|
||||||
|
"/x/sys_windows_amd64.s": true,
|
||||||
|
"/x/rt0_js_wasm.s": true,
|
||||||
|
"/x/sys_darwin_arm64.s": true,
|
||||||
|
"/x/sys_linux_amd64.s": false,
|
||||||
|
"/x/time_linux_amd64.s": false,
|
||||||
|
"/x/memmove_amd64.s": false,
|
||||||
|
"/x/generic.s": false,
|
||||||
|
} {
|
||||||
|
if got := otherGOOSFile(path); got != want {
|
||||||
|
t.Errorf("otherGOOSFile(%q) = %v, want %v", path, got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAuditGoAsm covers the audit wiring end to end: a package
|
||||||
|
// beside its kernel, the kernel living off the generated defines, and the
|
||||||
|
// histogram recording a generation failure as its own reason.
|
||||||
|
func TestRunCorpusAuditGoAsm(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("pkg.go", `package corpus
|
||||||
|
|
||||||
|
const pageSize = 4096
|
||||||
|
|
||||||
|
type header struct {
|
||||||
|
magic uint64
|
||||||
|
flags uint64
|
||||||
|
}
|
||||||
|
`)
|
||||||
|
write("kern_amd64.s", "#include \"go_asm.h\"\nTEXT \xc2\xb7f(SB), NOSPLIT, $0-16\n\tMOVQ\t$const_pageSize, AX\n\tMOVQ\t$header__size, BX\n\tRET\n")
|
||||||
|
// The defines live in the file's own package; a kernel in a directory
|
||||||
|
// without Go files has no package to generate from.
|
||||||
|
if err := os.MkdirAll(filepath.Join(dir, "sub"), 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
write(filepath.Join("sub", "lonely_arm64.s"), "#include \"go_asm.h\"\nTEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
get := func(name string) *corpusTally {
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == name {
|
||||||
|
return stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no tally for %s", name)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if a := get("amd64"); a.attempted != 1 || a.assembled != 1 {
|
||||||
|
t.Errorf("amd64 = %d/%d, want 1/1", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
// lonely_arm64.s is an arm64 file whose package cannot be generated.
|
||||||
|
if a := get("arm64"); a.attempted != 1 || a.assembled != 0 {
|
||||||
|
t.Errorf("arm64 = %d/%d, want 0/1", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
if r := get("arm64").reasons["go_asm.h generation failed"]; r != 1 {
|
||||||
|
t.Errorf("arm64 go_asm.h failure count = %d, want 1", r)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAuditGOOS covers the filename-derived GOOS end to end: a
|
||||||
|
// kernel whose name names darwin must have its header type-checked with
|
||||||
|
// GOOS=darwin, so the darwin-only constant it offsets with is defined. The
|
||||||
|
// operand mirrors sys_darwin_arm64.s's trampoline, where a missing define
|
||||||
|
// leaves an unexpanded symbol in the offset and fails.
|
||||||
|
func TestRunCorpusAuditGOOS(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("pkg.go", "package corpus\n")
|
||||||
|
write("plat_darwin.go", "//go:build darwin\n\npackage corpus\n\nconst trampolineNumer = 8\n")
|
||||||
|
write("kern_darwin_arm64.s", "#include \"go_asm.h\"\n"+
|
||||||
|
"GLOBL timebase<>(SB), NOPTR, $16\n"+
|
||||||
|
"TEXT \xc2\xb7g(SB), NOSPLIT, $0-0\n"+
|
||||||
|
"\tMOVD\ttimebase<>+const_trampolineNumer(SB), R0\n"+
|
||||||
|
"\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
var arm *corpusTally
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == "arm64" {
|
||||||
|
arm = stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if arm == nil {
|
||||||
|
t.Fatal("no arm64 tally")
|
||||||
|
}
|
||||||
|
if arm.attempted != 1 || arm.assembled != 1 {
|
||||||
|
t.Errorf("arm64 = %d/%d, want 1/1; reasons: %v", arm.assembled, arm.attempted, arm.reasons)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestRunCorpusAuditBuildConstraint covers the //go:build classification end
|
||||||
|
// to end: a generic-named file whose constraint admits one target is
|
||||||
|
// attempted there alone (cpu_x86.s on amd64), and a file whose constraint
|
||||||
|
// admits none of the four targets is never attempted (the msan and
|
||||||
|
// goexperiment trees).
|
||||||
|
func TestRunCorpusAuditBuildConstraint(t *testing.T) {
|
||||||
|
dir := t.TempDir()
|
||||||
|
write := func(name, src string) {
|
||||||
|
t.Helper()
|
||||||
|
if err := os.WriteFile(filepath.Join(dir, name), []byte(src), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
write("x86.s", "//go:build 386 || amd64\n\nTEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
write("racey.s", "//go:build race\n\nTEXT \xc2\xb7r(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
write("plain.s", "TEXT \xc2\xb7p(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
|
||||||
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
|
}
|
||||||
|
tally := func(name string) *corpusTally {
|
||||||
|
for i, tg := range stats.targets {
|
||||||
|
if tg.name == name {
|
||||||
|
return stats.tallies[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no tally for %s", name)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if stats.narrowed != 1 || stats.excluded != 1 || stats.generic != 1 {
|
||||||
|
t.Errorf("buckets = narrowed %d, excluded %d, generic %d; want 1, 1, 1", stats.narrowed, stats.excluded, stats.generic)
|
||||||
|
}
|
||||||
|
if a := tally("amd64"); a.attempted != 2 || a.assembled != 2 {
|
||||||
|
t.Errorf("amd64 = %d/%d, want 2/2 (x86.s and plain.s)", a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
for _, name := range []string{"arm64", "riscv64", "loong64"} {
|
||||||
|
if a := tally(name); a.attempted != 1 || a.assembled != 1 {
|
||||||
|
t.Errorf("%s = %d/%d, want 1/1 (plain.s only)", name, a.assembled, a.attempted)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if stats.full != 2 {
|
||||||
|
t.Errorf("full = %d, want 2 (x86.s over its one target, plain.s over all four)", stats.full)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestGenerateGoAsmHeaderRuntime pins the generator against the real thing:
|
||||||
|
// the runtime package of the ambient toolchain, whose header the toolchain's
|
||||||
|
// own -asmhdr output was sampled from. Skipped in short mode: it type-checks
|
||||||
|
// the whole package. The GOROOT comes from the go command itself, so the
|
||||||
|
// test follows whatever toolchain the host provides.
|
||||||
|
func TestGenerateGoAsmHeaderRuntime(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("type-checks the whole runtime package")
|
||||||
|
}
|
||||||
|
out, err := exec.Command("go", "env", "GOROOT").Output()
|
||||||
|
if err != nil {
|
||||||
|
t.Skipf("no Go toolchain: %v", err)
|
||||||
|
}
|
||||||
|
runtimeDir := filepath.Join(strings.TrimSpace(string(out)), "src", "runtime")
|
||||||
|
dir, err := generateGoAsmHeader(runtimeDir, "", "amd64", t.TempDir())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("generateGoAsmHeader(runtime): %v", err)
|
||||||
|
}
|
||||||
|
b, err := os.ReadFile(dir + "/go_asm.h")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
hdr := string(b)
|
||||||
|
for _, want := range []string{
|
||||||
|
"#define const_hashSize 8\n",
|
||||||
|
"#define const_avxSupported 1\n",
|
||||||
|
"#define const_pageSize 8192\n",
|
||||||
|
"#define g_stackguard0 16\n",
|
||||||
|
"#define m__size ",
|
||||||
|
} {
|
||||||
|
if !strings.Contains(hdr, want) {
|
||||||
|
t.Errorf("runtime header misses %q", want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+426
-36
@@ -5,6 +5,8 @@ package main
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"go/build/constraint"
|
||||||
|
"maps"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
@@ -14,9 +16,9 @@ import (
|
|||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
|
// cmdAuditInstructions cross-checks a gasm encoder against the Go toolchain's
|
||||||
@@ -37,7 +39,7 @@ import (
|
|||||||
// construction and are excluded from the diff; the other architectures list
|
// construction and are excluded from the diff; the other architectures list
|
||||||
// their conditional branches outright.
|
// their conditional branches outright.
|
||||||
func cmdAuditInstructions(args []string) error {
|
func cmdAuditInstructions(args []string) error {
|
||||||
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [amd64|arm64|riscv64|loong64]", `
|
fs := newCommand("audit-instructions", "gasm audit-instructions [--corpus [dir]] [--list] [-I dir] [amd64|arm64|riscv64|loong64]", `
|
||||||
Compare the gasm encoder for the given architecture (default amd64) against
|
Compare the gasm encoder for the given architecture (default amd64) against
|
||||||
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
go tool asm and print the diff: superset encodings (gasm-only, shippable via
|
||||||
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
|
gasm asm --format goobj) and known-but-unencodable names (the backlog). The
|
||||||
@@ -54,14 +56,19 @@ toolchain probing. A file whose name carries a recognisable _arch suffix is
|
|||||||
attempted for that architecture; a file without one is attempted for all
|
attempted for that architecture; a file without one is attempted for all
|
||||||
four, exactly as a GOARCH build would compile it. The report gives the
|
four, exactly as a GOARCH build would compile it. The report gives the
|
||||||
per-architecture pass rates and the most common failure reasons, which drive
|
per-architecture pass rates and the most common failure reasons, which drive
|
||||||
the encodability backlog by frequency rather than by table order.
|
the encodability backlog by frequency rather than by table order. With
|
||||||
|
-list the report also prints every failing file with its reason, per
|
||||||
|
architecture.
|
||||||
`)
|
`)
|
||||||
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
|
corpus := fs.Bool("corpus", false, "assemble a corpus of .s files and report pass rates and failure reasons")
|
||||||
|
list := fs.Bool("list", false, "with --corpus, list every failing file with its reason, per architecture")
|
||||||
|
var dirs includeDirs
|
||||||
|
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if *corpus {
|
if *corpus {
|
||||||
return cmdAuditCorpus(fs.Args())
|
return cmdAuditCorpus(fs.Args(), dirs, *list)
|
||||||
}
|
}
|
||||||
archName := "amd64"
|
archName := "amd64"
|
||||||
switch n := len(fs.Args()); {
|
switch n := len(fs.Args()); {
|
||||||
@@ -266,18 +273,76 @@ func probeShapes(a arch.Arch) []string {
|
|||||||
// and takes R register spellings.
|
// and takes R register spellings.
|
||||||
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
"EQ, R0, R1, R2", "EQ, R0, R1", "EQ, R0",
|
||||||
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
"GE, F0, F1, F2", "NE, F0, F1, $0",
|
||||||
|
// Pairs, acquire/release and exclusive atomics, LSE-AL forms.
|
||||||
|
"(R0), R1", "R0, (R1)", "R1, (R2), R3", "(R2, R3), 8(R1)",
|
||||||
|
"8(R1), (R2, R3)", "R1, R2, (R3)", "(R0)",
|
||||||
|
// System operations and their register/operand names.
|
||||||
|
"$4, R1, p2", "$35943", "$1", "$1, SPSel", "SPSel, R0",
|
||||||
|
"IVAC, R0", "(R0), PLDL1KEEP", "R1, R2, R3, R4",
|
||||||
|
// SIMD element, structure and literal-pool forms.
|
||||||
|
"(R0), [V1.B16]", "[V1.B16], (R0)", "V13.S[0], R1",
|
||||||
|
"R1, V2.B[3]", "$4, V1.B16, V2.B16", "V1.B16, (R0)",
|
||||||
|
"(R0), V1.B16", "",
|
||||||
|
// The spellings GOROOT's own kernels use, from the
|
||||||
|
// differential kernels this table was proven against.
|
||||||
|
"R0, p2", "R0, R1", "F0, F1, F2, F3", "$4, V1.B16, V2.B16, V3.B16, V4.B16",
|
||||||
|
"(R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "$1, $2, V1",
|
||||||
|
"R0, R1, p2", "p2, R1", "$1234, R1", "DCZID_EL0, R1",
|
||||||
|
"$0", "R1, $4, EQ", "$33, R1, $25, R2", "$4, R1, p2",
|
||||||
|
"$4, V1.B8, V2.B8, V3.B8", "$63, V1.D2, V2.D2, V3.D2",
|
||||||
|
"V1.B16, [V2.B16], V3.B16", "V1.B8, [V2.B16, V3.B16], V4.B8",
|
||||||
|
"$4, V1.B16, V2.B16, V3.B16", "$15, V1", "V1, V2, p2",
|
||||||
|
"R0, R1, $1, $4, p2",
|
||||||
|
// The landing-pad kind, the compiler's PCDATA
|
||||||
|
// bookkeeping and the four-operand bitfield
|
||||||
|
// insert/extract family, as the toolchain's own
|
||||||
|
// testdata spells them.
|
||||||
|
"C", "$1, $0", "$0, R1, $1, R2",
|
||||||
}
|
}
|
||||||
case arch.RISCV:
|
case arch.RISCV:
|
||||||
return []string{
|
return []string{
|
||||||
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
"X5, X6, X7", "X5, X6", "X5", "$1, X5", "X5, (X6)", "$1, X5, X6",
|
||||||
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
"(X5), X6", "F0, F1, F2", "F0, F1", "p2", "X1, p2", "X0, p2",
|
||||||
"X5, X6, p2", "p2(SB)",
|
"X5, X6, p2", "p2(SB)",
|
||||||
|
// AMO atomics: destination, base, source.
|
||||||
|
"R5, (R4), R6", "X5, (X4), X6",
|
||||||
|
// Segment stores take the first vector register aligned
|
||||||
|
// to the segment count, as the toolchain requires.
|
||||||
|
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||||
|
// The FP multiply-add family takes four registers.
|
||||||
|
"F0, F1, F2, F3",
|
||||||
|
// The RVV slice: register, vector-register and vtype forms.
|
||||||
|
"V1, V2, V3", "V1, X5, V2", "V1", "V1, (X5)", "(X5), V1",
|
||||||
|
"$15, V1", "$15", "V1, V2", "V1, X5",
|
||||||
|
"X5, X6, p2", "R5, R6, p2",
|
||||||
|
"X5, E8, M8, TA, MA, X6", "$4, E32, M1, TA, MA, X1",
|
||||||
|
"(X5), X6, V1, V2",
|
||||||
|
// The CSR immediate forms the toolchain's testdata spells:
|
||||||
|
// immediate, CSR name, destination.
|
||||||
|
"$2, TIME, X5",
|
||||||
|
"",
|
||||||
}
|
}
|
||||||
case arch.LOONG64:
|
case arch.LOONG64:
|
||||||
return []string{
|
return []string{
|
||||||
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
"R4, R5, R6", "R4, R5", "R4", "$1, R4", "R4, (R5)", "(R4), R5",
|
||||||
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
"F0, F1, F2", "F0, F1", "p2", "R1, p2", "R4, p2",
|
||||||
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
"$1, R4, R5, R6", "$65536, R4", "R4, R5, p2", "p2(SB)",
|
||||||
|
// AMO atomics: destination, base, source.
|
||||||
|
"R5, (R4), R6", "X5, (X4), X6",
|
||||||
|
// Segment stores take the first vector register aligned
|
||||||
|
// to the segment count, as the toolchain requires.
|
||||||
|
"(X5), X6, V0, V8", "(X5), X6, V0", "(X5), X0, V4",
|
||||||
|
// The LSX and LASX banks share the 5-bit numbering with F.
|
||||||
|
"V1, V2, V3", "X1, X2, X3", "V1, V2", "X1, X2", "V1", "X1",
|
||||||
|
// The vector compare-to-flag forms land in an FCC register.
|
||||||
|
"V1, FCC0", "X1, FCC0",
|
||||||
|
// The compiler's bookkeeping pair and the raw spellings the
|
||||||
|
// toolchain's own testdata carries: JIRL rd, rj, offset (the
|
||||||
|
// form RET lowers to), the prefetch with a 32-bit address and
|
||||||
|
// hint, and the byte-shuffle quads.
|
||||||
|
"$1, $0", "R1, R5, 0", "0(R7), $5, $0", "(R7), $5, $0",
|
||||||
|
"V1, V2, V3, V4", "X1, X2, X3, X4",
|
||||||
|
"",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return nil
|
return nil
|
||||||
@@ -342,17 +407,30 @@ type corpusTally struct {
|
|||||||
assembled int
|
assembled int
|
||||||
reasons map[string]int // failure reason → count
|
reasons map[string]int // failure reason → count
|
||||||
example map[string]string // failure reason → one representative file
|
example map[string]string // failure reason → one representative file
|
||||||
|
fails []corpusFailure // every failure, in file order, for --list
|
||||||
}
|
}
|
||||||
|
|
||||||
func (t *corpusTally) fail(path, reason string) {
|
// corpusFailure is one failed attempt, recorded for the --list report.
|
||||||
|
type corpusFailure struct {
|
||||||
|
path string
|
||||||
|
reason string
|
||||||
|
detail string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (t *corpusTally) fail(path string, err error) {
|
||||||
|
reason := corpusReason(err)
|
||||||
t.reasons[reason]++
|
t.reasons[reason]++
|
||||||
if t.example[reason] == "" {
|
if t.example[reason] == "" {
|
||||||
t.example[reason] = path
|
t.example[reason] = path
|
||||||
}
|
}
|
||||||
|
t.fails = append(t.fails, corpusFailure{path: path, reason: reason, detail: firstLine(err.Error())})
|
||||||
}
|
}
|
||||||
|
|
||||||
// cmdAuditCorpus implements audit-instructions --corpus.
|
// cmdAuditCorpus implements audit-instructions --corpus. The include
|
||||||
func cmdAuditCorpus(args []string) error {
|
// directories carry #include resolution over a corpus whose files refer to
|
||||||
|
// headers such as GOROOT/pkg/include, the same -I a toolchain comparison
|
||||||
|
// needs.
|
||||||
|
func cmdAuditCorpus(args []string, dirs includeDirs, list bool) error {
|
||||||
if len(args) > 1 {
|
if len(args) > 1 {
|
||||||
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
|
return &usageError{fmt.Errorf("audit-instructions --corpus takes at most one directory argument")}
|
||||||
}
|
}
|
||||||
@@ -366,26 +444,190 @@ func cmdAuditCorpus(args []string) error {
|
|||||||
}
|
}
|
||||||
root = filepath.Join(strings.TrimSpace(string(out)), "src")
|
root = filepath.Join(strings.TrimSpace(string(out)), "src")
|
||||||
}
|
}
|
||||||
stats, err := runCorpusAudit(root)
|
// The toolchain's shipped headers (funcdata.h and friends) define the
|
||||||
|
// macros GOROOT files include; a corpus audit measures those files, so
|
||||||
|
// the header directory joins the search path automatically. go_asm.h
|
||||||
|
// is compiler-generated per package, so it is not resolved from here:
|
||||||
|
// files that include it get one generated per target architecture,
|
||||||
|
// which runCorpusAudit arranges.
|
||||||
|
if out, err := exec.Command("go", "env", "GOROOT").Output(); err == nil {
|
||||||
|
pkgInclude := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
|
||||||
|
if fi, err := os.Stat(pkgInclude); err == nil && fi.IsDir() {
|
||||||
|
seen := false
|
||||||
|
for _, d := range dirs {
|
||||||
|
if d == pkgInclude {
|
||||||
|
seen = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !seen {
|
||||||
|
dirs = append(dirs, pkgInclude)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stats, err := runCorpusAudit(root, dirs)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
printCorpusStats(stats)
|
printCorpusStats(stats, list)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// corpusStats is the outcome of one corpus audit run.
|
// corpusStats is the outcome of one corpus audit run.
|
||||||
type corpusStats struct {
|
type corpusStats struct {
|
||||||
root string
|
root string
|
||||||
files int
|
files int
|
||||||
generic int // files attempted for all four architectures
|
generic int // files attempted for all four architectures
|
||||||
full int // files that assembled for every target architecture
|
narrowed int // files whose //go:build admits a proper subset of the four
|
||||||
targets []corpusTarget
|
excluded int // files whose //go:build admits none of the four: never compiled
|
||||||
tallies []*corpusTally
|
otherPort int // files named for another Go port: never attempted
|
||||||
|
full int // files that assembled for every applicable target architecture
|
||||||
|
targets []corpusTarget
|
||||||
|
tallies []*corpusTally
|
||||||
}
|
}
|
||||||
|
|
||||||
// runCorpusAudit assembles every .s file under root and returns the stats.
|
// runCorpusAudit assembles every .s file under root and returns the stats.
|
||||||
func runCorpusAudit(root string) (*corpusStats, error) {
|
// goPortSuffixes lists every architecture the Go project ports to. A file
|
||||||
|
// named for one of them belongs to that port's build, not to the generic
|
||||||
|
// set, even when gasm does not support the architecture.
|
||||||
|
var goPortSuffixes = []string{
|
||||||
|
"386", "amd64", "arm", "arm64", "loong64", "mips", "mips64",
|
||||||
|
"mips64le", "mipsle", "mips64x", "mipsx", "ppc64", "ppc64le",
|
||||||
|
"ppc64x", "riscv", "riscv64", "s390x", "wasm",
|
||||||
|
}
|
||||||
|
|
||||||
|
// otherPortFile reports whether the file belongs to a build no supported
|
||||||
|
// target ever compiles: either its name carries a Go-architecture suffix
|
||||||
|
// gasm does not support, or, for a file with no architecture suffix at all,
|
||||||
|
// it names another GOOS, which go/build drops from the file set
|
||||||
|
// (rt0_js_wasm.s is a javascript build, not a generic one).
|
||||||
|
func otherPortFile(path string) bool {
|
||||||
|
if otherGOOSFile(path) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
for _, sfx := range goPortSuffixes {
|
||||||
|
if strings.HasSuffix(base, "_"+sfx+".s") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// goOSNames are the GOOS values go/build recognises in file names.
|
||||||
|
var goOSNames = map[string]bool{
|
||||||
|
"aix": true, "android": true, "darwin": true, "dragonfly": true,
|
||||||
|
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
|
||||||
|
"js": true, "linux": true, "nacl": true, "netbsd": true,
|
||||||
|
"openbsd": true, "plan9": true, "solaris": true, "wasip1": true,
|
||||||
|
"windows": true, "zos": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveGOOS validates a -GOOS flag value, mirroring the architecture
|
||||||
|
// check's surface: a usage error naming what the tool accepts.
|
||||||
|
func resolveGOOS(name string) (string, error) {
|
||||||
|
lower := strings.ToLower(name)
|
||||||
|
if goOSNames[lower] {
|
||||||
|
return lower, nil
|
||||||
|
}
|
||||||
|
return "", &usageError{fmt.Errorf("unknown GOOS %q: want one of %s", name, strings.Join(slices.Sorted(maps.Keys(goOSNames)), ", "))}
|
||||||
|
}
|
||||||
|
|
||||||
|
// goosFromFilename returns the GOOS the file's name carries, by go/build's
|
||||||
|
// goodOSArchFile rule: the GOOS segment sits last, or last before the
|
||||||
|
// architecture segment (sys_darwin_arm64.s, vlop_arm.s carries none). An
|
||||||
|
// empty result means the name names no GOOS and the ambient one applies.
|
||||||
|
func goosFromFilename(path string) string {
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
base = strings.TrimSuffix(base, ".s")
|
||||||
|
// go/build ignores everything before the first underscore, so a GOOS
|
||||||
|
// segment is only ever looked for from there on.
|
||||||
|
i := strings.IndexByte(base, '_')
|
||||||
|
if i < 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
segs := strings.Split(base[i:], "_")
|
||||||
|
if n := len(segs); n >= 2 && goOSNames[segs[n-2]] && slices.Contains(goPortSuffixes, segs[n-1]) {
|
||||||
|
return segs[n-2]
|
||||||
|
}
|
||||||
|
if goOSNames[segs[len(segs)-1]] {
|
||||||
|
return segs[len(segs)-1]
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// otherGOOSFile reports whether the file's name names a GOOS other than the
|
||||||
|
// host's, by go/build's file-name rules.
|
||||||
|
func otherGOOSFile(path string) bool {
|
||||||
|
base := path
|
||||||
|
if i := strings.LastIndexByte(base, '/'); i >= 0 {
|
||||||
|
base = base[i+1:]
|
||||||
|
}
|
||||||
|
for seg := range strings.SplitSeq(strings.TrimSuffix(base, ".s"), "_") {
|
||||||
|
if goOSNames[seg] && seg != runtime.GOOS {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildConstraint returns the file's leading //go:build expression, or nil
|
||||||
|
// when the file carries none. The constraint governs the same header block
|
||||||
|
// go/build reads: blank lines and comments may precede it, and the first
|
||||||
|
// line that is neither ends the block. A constraint that does not parse
|
||||||
|
// narrows nothing, so the file stays in the attempted set: the audit must
|
||||||
|
// never exclude a file the toolchain would compile.
|
||||||
|
func buildConstraint(src string) constraint.Expr {
|
||||||
|
for line := range strings.SplitSeq(src, "\n") {
|
||||||
|
t := strings.TrimSpace(line)
|
||||||
|
switch {
|
||||||
|
case t == "":
|
||||||
|
continue
|
||||||
|
case strings.HasPrefix(t, "//"):
|
||||||
|
if constraint.IsGoBuild(t) {
|
||||||
|
e, err := constraint.Parse(t)
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return e
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
default:
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// unixOS is go/build's unixOS set: the GOOSes the unix build tag admits.
|
||||||
|
var unixOS = map[string]bool{
|
||||||
|
"aix": true, "android": true, "darwin": true, "dragonfly": true,
|
||||||
|
"freebsd": true, "hurd": true, "illumos": true, "ios": true,
|
||||||
|
"linux": true, "netbsd": true, "openbsd": true, "solaris": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// constraintTags answers the build tags a plain `go build` sets for a
|
||||||
|
// target: the GOOS and GOARCH, gc, and unix on the unix-like GOOSes. No
|
||||||
|
// experiment, sanitiser or cgo tag is ever true: the audit models the
|
||||||
|
// default build, and no GOROOT assembly file's constraint hinges on cgo.
|
||||||
|
func constraintTags(goarch, goos string) func(string) bool {
|
||||||
|
return func(tag string) bool {
|
||||||
|
switch tag {
|
||||||
|
case goarch, goos, "gc":
|
||||||
|
return true
|
||||||
|
case "unix":
|
||||||
|
return unixOS[goos]
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func runCorpusAudit(root string, dirs includeDirs) (*corpusStats, error) {
|
||||||
files, err := asmFiles(root)
|
files, err := asmFiles(root)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
@@ -402,43 +644,172 @@ func runCorpusAudit(root string) (*corpusStats, error) {
|
|||||||
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
|
tallies[i] = &corpusTally{reasons: map[string]int{}, example: map[string]string{}}
|
||||||
}
|
}
|
||||||
// full is the north-star number: a file counts when every architecture
|
// full is the north-star number: a file counts when every architecture
|
||||||
// its name allows assembles it.
|
// its build admits assembles it.
|
||||||
full, generic := 0, 0
|
full, generic, otherPort, narrowedCount, excluded := 0, 0, 0, 0, 0
|
||||||
|
|
||||||
|
// Header generation is created on first use, so a corpus with no
|
||||||
|
// go_asm.h includes never pays for a temp directory.
|
||||||
|
var hdr *asmhdrCache
|
||||||
|
defer func() {
|
||||||
|
if hdr != nil {
|
||||||
|
hdr.close()
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
for _, path := range files {
|
for _, path := range files {
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
f, errs := parser.Parse(path, src)
|
|
||||||
|
|
||||||
|
// The GOOS the header generation type-checks under follows the
|
||||||
|
// file's name when the name carries one; the ambient GOOS is the
|
||||||
|
// honest guess otherwise (a build tag naming another GOOS is
|
||||||
|
// invisible to a file-name rule).
|
||||||
|
goos := goosFromFilename(path)
|
||||||
|
|
||||||
|
// The GOOS the header generation type-checks under follows the
|
||||||
|
// file's name when the name carries one; the ambient GOOS is the
|
||||||
|
// honest guess otherwise.
|
||||||
|
namedArch := arch.FromFilename(path)
|
||||||
var wanted []int // indexes into targets
|
var wanted []int // indexes into targets
|
||||||
if a := arch.FromFilename(path); a != arch.Unknown {
|
other := false
|
||||||
|
switch {
|
||||||
|
case namedArch != arch.Unknown:
|
||||||
for i, tg := range targets {
|
for i, tg := range targets {
|
||||||
if tg.a == a {
|
if tg.a == namedArch {
|
||||||
wanted = append(wanted, i)
|
wanted = append(wanted, i)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
case otherPortFile(path):
|
||||||
generic++
|
// A file named for a Go port gasm does not support (arm,
|
||||||
|
// 386, s390x, ...) or for another GOOS is compiled by no
|
||||||
|
// supported-arch build, so it is neither generic nor a
|
||||||
|
// per-arch attempt: counting it as generic would make the
|
||||||
|
// headline unreachably low for reasons no supported target
|
||||||
|
// can fix.
|
||||||
|
other = true
|
||||||
|
otherPort++
|
||||||
|
default:
|
||||||
for i := range targets {
|
for i := range targets {
|
||||||
wanted = append(wanted, i)
|
wanted = append(wanted, i)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A //go:build constraint narrows the set of targets the file is
|
||||||
|
// assembled for, the way the go command compiles the file only for
|
||||||
|
// the targets the expression admits: cpu_x86.s belongs to the x86
|
||||||
|
// build alone, and a file whose constraint admits none of the four
|
||||||
|
// targets (the goexperiment.runtimesecret and msan trees) is
|
||||||
|
// compiled by no supported build. The tags mirror what a plain
|
||||||
|
// `go build` sets: the GOOS and GOARCH, gc, and unix on the
|
||||||
|
// unix-like GOOSes; no experiment, sanitiser or cgo tag is ever
|
||||||
|
// true. The GOOS is the file's own when the name carries one,
|
||||||
|
// else the ambient one.
|
||||||
|
goosForEval := goos
|
||||||
|
if goosForEval == "" {
|
||||||
|
goosForEval = runtime.GOOS
|
||||||
|
}
|
||||||
|
narrowed := false
|
||||||
|
if len(wanted) > 0 {
|
||||||
|
if ce := buildConstraint(src); ce != nil {
|
||||||
|
kept := make([]int, 0, len(wanted))
|
||||||
|
for _, i := range wanted {
|
||||||
|
tg := targets[i]
|
||||||
|
if ce.Eval(constraintTags(goarchName(tg.a), goosForEval)) {
|
||||||
|
kept = append(kept, i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(kept) < len(wanted) {
|
||||||
|
narrowed = true
|
||||||
|
}
|
||||||
|
wanted = kept
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
switch {
|
||||||
|
case other:
|
||||||
|
// already tallied above
|
||||||
|
case len(wanted) == 0:
|
||||||
|
excluded++
|
||||||
|
case namedArch != arch.Unknown:
|
||||||
|
// a per-arch attempt over the constraint's subset
|
||||||
|
case narrowed:
|
||||||
|
narrowedCount++
|
||||||
|
default:
|
||||||
|
generic++
|
||||||
|
}
|
||||||
|
|
||||||
|
// A file that includes go_asm.h parses against a per-target header:
|
||||||
|
// the defines differ per architecture (internal/cpu's layout, for
|
||||||
|
// one) and per GOOS (sys_darwin_arm64.s's trampoline constants,
|
||||||
|
// for another), so the parse cannot be shared the way a
|
||||||
|
// header-free file's can. A generation failure is a failure for
|
||||||
|
// every target, named for the package rather than a bare "include
|
||||||
|
// not found". A header already resolvable in the package
|
||||||
|
// directory or the -I list is left alone.
|
||||||
|
if len(wanted) > 0 && needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
|
||||||
|
if hdr == nil {
|
||||||
|
if hdr, err = newAsmhdrCache(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pkgDir := filepath.Dir(path)
|
||||||
|
ok := true
|
||||||
|
for _, i := range wanted {
|
||||||
|
tg, t := targets[i], tallies[i]
|
||||||
|
t.attempted++
|
||||||
|
hdrDir, err := hdr.dirFor(pkgDir, goos, goarchName(tg.a))
|
||||||
|
if err != nil {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{
|
||||||
|
Expand: true,
|
||||||
|
IncludeDirs: append(slices.Clone(dirs), hdrDir),
|
||||||
|
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
|
||||||
|
})
|
||||||
|
if len(errs) > 0 {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, errs[0])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if _, err := assembleFile(tg.a, f, goos); err != nil {
|
||||||
|
ok = false
|
||||||
|
t.fail(path, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
t.assembled++
|
||||||
|
}
|
||||||
|
if ok && len(wanted) > 0 {
|
||||||
|
full++
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
ok := true
|
ok := true
|
||||||
for _, i := range wanted {
|
for _, i := range wanted {
|
||||||
tg, t := targets[i], tallies[i]
|
tg, t := targets[i], tallies[i]
|
||||||
t.attempted++
|
t.attempted++
|
||||||
|
// The parse carries the target's platform predefines, so it
|
||||||
|
// cannot be shared across targets the way a header-free file's
|
||||||
|
// could: a #ifdef GOARCH_arm block must be live on arm64 and
|
||||||
|
// dead everywhere else.
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{
|
||||||
|
Expand: true,
|
||||||
|
IncludeDirs: dirs,
|
||||||
|
Predefines: platformPredefinesFor(goarchName(tg.a), goos),
|
||||||
|
})
|
||||||
var err error
|
var err error
|
||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
err = errs[0] // a parse failure is a failure for every target
|
err = errs[0] // a parse failure is a failure for every target
|
||||||
} else {
|
} else {
|
||||||
_, err = assembleFile(tg.a, f)
|
_, err = assembleFile(tg.a, f, goos)
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
ok = false
|
ok = false
|
||||||
t.fail(path, corpusReason(err))
|
t.fail(path, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
t.assembled++
|
t.assembled++
|
||||||
@@ -449,19 +820,29 @@ func runCorpusAudit(root string) (*corpusStats, error) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
return &corpusStats{
|
return &corpusStats{
|
||||||
root: root,
|
root: root,
|
||||||
files: len(files),
|
files: len(files),
|
||||||
generic: generic,
|
generic: generic,
|
||||||
full: full,
|
narrowed: narrowedCount,
|
||||||
targets: targets,
|
excluded: excluded,
|
||||||
tallies: tallies,
|
otherPort: otherPort,
|
||||||
|
full: full,
|
||||||
|
targets: targets,
|
||||||
|
tallies: tallies,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// printCorpusStats renders the corpus audit report.
|
// printCorpusStats renders the corpus audit report.
|
||||||
func printCorpusStats(s *corpusStats) {
|
func printCorpusStats(s *corpusStats, list bool) {
|
||||||
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures)\n", s.root, s.files, s.generic)
|
fmt.Printf("corpus %s: %d files (%d generic, attempted for all architectures; %d narrowed by //go:build; %d excluded by //go:build; %d named for other Go ports, never attempted)\n",
|
||||||
fmt.Printf(" assemble for every target architecture: %d (%.1f%%)\n", s.full, 100*float64(s.full)/float64(max(s.files, 1)))
|
s.root, s.files, s.generic, s.narrowed, s.excluded, s.otherPort)
|
||||||
|
// The rate is over the files a supported build would attempt: the
|
||||||
|
// other ports' files and the ones no supported target compiles sit in
|
||||||
|
// the count for completeness but can never assemble, so counting them
|
||||||
|
// in the denominator would report the gap of platforms gasm
|
||||||
|
// deliberately does not target.
|
||||||
|
attemptable := max(s.files-s.otherPort-s.excluded, 1)
|
||||||
|
fmt.Printf(" assemble for every applicable target: %d of %d attemptable (%.1f%%)\n", s.full, attemptable, 100*float64(s.full)/float64(attemptable))
|
||||||
for i, tg := range s.targets {
|
for i, tg := range s.targets {
|
||||||
t := s.tallies[i]
|
t := s.tallies[i]
|
||||||
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
|
fmt.Printf(" %s: %d/%d attempted\n", tg.name, t.assembled, t.attempted)
|
||||||
@@ -469,6 +850,13 @@ func printCorpusStats(s *corpusStats) {
|
|||||||
fmt.Printf(" %4d %s\n", t.reasons[r], r)
|
fmt.Printf(" %4d %s\n", t.reasons[r], r)
|
||||||
fmt.Printf(" e.g. %s\n", t.example[r])
|
fmt.Printf(" e.g. %s\n", t.example[r])
|
||||||
}
|
}
|
||||||
|
if !list {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, f := range t.fails {
|
||||||
|
fmt.Printf(" FAIL %s\n", f.path)
|
||||||
|
fmt.Printf(" %s: %s\n", f.reason, f.detail)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -476,6 +864,8 @@ func printCorpusStats(s *corpusStats) {
|
|||||||
func corpusReason(err error) string {
|
func corpusReason(err error) string {
|
||||||
msg := err.Error()
|
msg := err.Error()
|
||||||
switch {
|
switch {
|
||||||
|
case strings.Contains(msg, "go_asm.h for GOARCH"):
|
||||||
|
return "go_asm.h generation failed"
|
||||||
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
|
case strings.Contains(msg, "unsupported"), strings.Contains(msg, "cannot encode"):
|
||||||
return "instruction not encodable"
|
return "instruction not encodable"
|
||||||
case strings.Contains(msg, "undefined label"):
|
case strings.Contains(msg, "undefined label"):
|
||||||
|
|||||||
+61
-1
@@ -4,9 +4,10 @@
|
|||||||
package main
|
package main
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"runtime"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
)
|
)
|
||||||
|
|
||||||
func TestDerivedFamily(t *testing.T) {
|
func TestDerivedFamily(t *testing.T) {
|
||||||
@@ -64,3 +65,62 @@ func TestGasmEncodable(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestBuildConstraint pins the //go:build reader: the constraint governs the
|
||||||
|
// leading comment block, the first non-comment line ends it (a tag below a
|
||||||
|
// #include governs nothing, exactly as go/build drops it), and a file
|
||||||
|
// without one admits every target.
|
||||||
|
func TestBuildConstraint(t *testing.T) {
|
||||||
|
admits := func(src, goarch, goos string) bool {
|
||||||
|
t.Helper()
|
||||||
|
e := buildConstraint(src)
|
||||||
|
if e == nil {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return e.Eval(constraintTags(goarch, goos))
|
||||||
|
}
|
||||||
|
const ret = "TEXT \xc2\xb7f(SB), NOSPLIT, $0\n\tRET\n"
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
src string
|
||||||
|
amd64, arm64 bool
|
||||||
|
}{
|
||||||
|
{"no constraint", ret, true, true},
|
||||||
|
{"x86 only", "//go:build 386 || amd64\n\n" + ret, true, false},
|
||||||
|
{"arm64 and linux", "//go:build arm64 && linux\n\n" + ret, false, true},
|
||||||
|
{"msan never", "//go:build msan\n\n" + ret, false, false},
|
||||||
|
{"experiment never", "//go:build goexperiment.runtimesecret\n\n" + ret, false, false},
|
||||||
|
{"below an include governs nothing", "#include \"textflag.h\"\n//go:build amd64\n" + ret, true, true},
|
||||||
|
{"unparsable narrows nothing", "//go:build (amd64\n" + ret, true, true},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := admits(c.src, "amd64", runtime.GOOS); got != c.amd64 {
|
||||||
|
t.Errorf("amd64 admission = %v, want %v", got, c.amd64)
|
||||||
|
}
|
||||||
|
if got := admits(c.src, "arm64", runtime.GOOS); got != c.arm64 {
|
||||||
|
t.Errorf("arm64 admission = %v, want %v", got, c.arm64)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestConstraintTags pins the tag set a plain `go build` sets: the GOOS and
|
||||||
|
// GOARCH, gc, unix on the unix-like GOOSes; nothing else is ever true.
|
||||||
|
func TestConstraintTags(t *testing.T) {
|
||||||
|
ok := constraintTags("amd64", "linux")
|
||||||
|
for _, tag := range []string{"amd64", "linux", "gc", "unix"} {
|
||||||
|
if !ok(tag) {
|
||||||
|
t.Errorf("tag %q = false, want true", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, tag := range []string{"arm64", "freebsd", "darwin", "cgo", "race", "msan", "goexperiment.runtimesecret"} {
|
||||||
|
if ok(tag) {
|
||||||
|
t.Errorf("tag %q = true, want false", tag)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fb := constraintTags("arm64", "freebsd")
|
||||||
|
if !fb("unix") {
|
||||||
|
t.Error("unix on freebsd = false, want true")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build !linux
|
//go:build !(linux || (freebsd && (amd64 || arm64 || riscv64)))
|
||||||
|
|
||||||
package main
|
package main
|
||||||
|
|
||||||
@@ -11,6 +11,6 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
func cmdDebug(args []string) int {
|
func cmdDebug(args []string) int {
|
||||||
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux (ptrace)")
|
fmt.Fprintln(os.Stderr, "gasm debug: the interactive debugger requires Linux or FreeBSD (ptrace)")
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build linux
|
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||||
|
|
||||||
package main
|
package main
|
||||||
|
|
||||||
@@ -13,8 +13,8 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/debug"
|
"sourcedock.dev/petrbalvin/gasm-sdk/debug"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
func cmdDebug(args []string) int {
|
func cmdDebug(args []string) int {
|
||||||
+4
-4
@@ -9,9 +9,9 @@ import (
|
|||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdDis disassembles machine code: either a raw binary (standard input with
|
// cmdDis disassembles machine code: either a raw binary (standard input with
|
||||||
@@ -84,7 +84,7 @@ func disSource(path string, target arch.Arch) int {
|
|||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img, err := assembleFile(target, f)
|
img, err := assembleFile(target, f, "")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
|
fmt.Fprintf(os.Stderr, "gasm dis: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
+103
-29
@@ -27,15 +27,15 @@ import (
|
|||||||
"sync"
|
"sync"
|
||||||
"syscall"
|
"syscall"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/format"
|
"sourcedock.dev/petrbalvin/gasm-sdk/format"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
"sourcedock.dev/petrbalvin/gasm-sdk/lexer"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lint"
|
"sourcedock.dev/petrbalvin/gasm-sdk/lint"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/lsp"
|
"sourcedock.dev/petrbalvin/gasm-sdk/lsp"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// version reports the release the toolchain recorded for this build: the
|
// version reports the release the toolchain recorded for this build: the
|
||||||
@@ -240,6 +240,16 @@ func readSource(path string) (string, error) {
|
|||||||
return string(b), err
|
return string(b), err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// includeDirs collects repeatable -I flags: the directories searched for
|
||||||
|
// #include files during macro expansion and include splicing.
|
||||||
|
type includeDirs []string
|
||||||
|
|
||||||
|
func (d *includeDirs) String() string { return strings.Join(*d, ",") }
|
||||||
|
func (d *includeDirs) Set(v string) error {
|
||||||
|
*d = append(*d, v)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
func cmdTokens(args []string) int {
|
func cmdTokens(args []string) int {
|
||||||
fs := newCommand("tokens", "gasm tokens <file>", `
|
fs := newCommand("tokens", "gasm tokens <file>", `
|
||||||
Print the lexical token stream of FILE: position, token kind and text, one
|
Print the lexical token stream of FILE: position, token kind and text, one
|
||||||
@@ -476,7 +486,7 @@ hover, document symbols, diagnostics and semantic-token highlighting.
|
|||||||
}
|
}
|
||||||
|
|
||||||
func cmdAsm(args []string) int {
|
func cmdAsm(args []string) int {
|
||||||
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>", `
|
fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-GOOS os] [-o out] <file>", `
|
||||||
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
Assemble FILE without the Go toolchain: every TEXT function is encoded to
|
||||||
machine code and printed as a hex dump. Supported architectures: amd64
|
machine code and printed as a hex dump. Supported architectures: amd64
|
||||||
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
(including VEX/AVX2 and EVEX/AVX-512), arm64 (AArch64 integer, FP,
|
||||||
@@ -493,14 +503,26 @@ system toolchain; goobj emits the Go toolchain's own object format, which
|
|||||||
cmd/link consumes directly (it requires -p, the package path, and the
|
cmd/link consumes directly (it requires -p, the package path, and the
|
||||||
installed Go toolchain: the object preamble is captured from go tool asm
|
installed Go toolchain: the object preamble is captured from go tool asm
|
||||||
and the format version from go version).
|
and the format version from go version).
|
||||||
|
|
||||||
|
A file that includes go_asm.h gets that header generated automatically from
|
||||||
|
the package it lives in (the .go files beside it, type-checked for the
|
||||||
|
target architecture, the toolchain's own defines), so GOROOT assembly
|
||||||
|
assembles without a compiler. -GOOS selects the type-checking GOOS for
|
||||||
|
that header: a GOOS-specific file (sys_darwin_arm64.s) needs its platform's
|
||||||
|
defines, which a header from the ambient GOOS silently omits. A package
|
||||||
|
that has no Go files for the target or does not type-check is a hard error
|
||||||
|
naming the package.
|
||||||
`)
|
`)
|
||||||
out := fs.String("o", "", "write the output to this file")
|
out := fs.String("o", "", "write the output to this file")
|
||||||
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
format := fs.String("format", "raw", "output format: raw (concatenated image), elf or goobj (Go object)")
|
||||||
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
pkg := fs.String("p", "", "package path for --format goobj (qualifies the exported symbols)")
|
||||||
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
|
archName := fs.String("GOARCH", "", "target architecture: amd64, arm64, riscv64 or loong64 (overrides the file-name suffix)")
|
||||||
|
goosName := fs.String("GOOS", "", "operating system for go_asm.h generation: a GOOS go/build recognises (default: the host's)")
|
||||||
|
var dirs includeDirs
|
||||||
|
fs.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||||
fs.Parse(args)
|
fs.Parse(args)
|
||||||
if fs.NArg() != 1 {
|
if fs.NArg() != 1 {
|
||||||
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-p pkg] [-GOARCH arch] [-o out] <file>")
|
fmt.Fprintln(os.Stderr, "usage: gasm asm [--format raw|elf|goobj] [-I dir] [-p pkg] [-GOARCH arch] [-GOOS os] [-o out] <file>")
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
// The format is validated before anything else, so a bogus value exits 2
|
// The format is validated before anything else, so a bogus value exits 2
|
||||||
@@ -521,12 +543,38 @@ and the format version from go version).
|
|||||||
}
|
}
|
||||||
targetArch = a
|
targetArch = a
|
||||||
}
|
}
|
||||||
|
goos := ""
|
||||||
|
if *goosName != "" {
|
||||||
|
g, err := resolveGOOS(*goosName)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintf(os.Stderr, "gasm asm: %v\n", err)
|
||||||
|
return 2
|
||||||
|
}
|
||||||
|
goos = g
|
||||||
|
}
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, "gasm:", err)
|
fmt.Fprintln(os.Stderr, "gasm:", err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
f, errs := parser.Parse(path, src)
|
// A file that includes go_asm.h cannot assemble without the package's
|
||||||
|
// defines, and without a compiler nothing else has generated them, so
|
||||||
|
// gasm produces the equivalent itself: automatic, because the compiler
|
||||||
|
// behaves the same way and a flag would only ever be forgotten. A
|
||||||
|
// generation failure is fatal and names the package: assembling against
|
||||||
|
// a missing header would fail later with a bare "undefined" instead.
|
||||||
|
// A go_asm.h that already resolves (placed by hand, or passed with -I)
|
||||||
|
// is left alone.
|
||||||
|
if needsGoAsmHeader(src) && !goAsmHeaderResolved(filepath.Dir(path), dirs) {
|
||||||
|
hdrDir, cleanup, err := ensureGoAsmHeader(path, targetArch, goos, nil)
|
||||||
|
if err != nil {
|
||||||
|
fmt.Fprintln(os.Stderr, "gasm asm:", err)
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
defer cleanup()
|
||||||
|
dirs = append(dirs, hdrDir)
|
||||||
|
}
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(targetArch), goos)})
|
||||||
for _, e := range errs {
|
for _, e := range errs {
|
||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||||
}
|
}
|
||||||
@@ -534,7 +582,7 @@ and the format version from go version).
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
img, err := assembleFile(targetArch, f)
|
img, err := assembleFile(targetArch, f, goos)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, err)
|
||||||
return 1
|
return 1
|
||||||
@@ -634,7 +682,7 @@ and the format version from go version).
|
|||||||
|
|
||||||
// cmdDiff compares the machine code of two assembly files.
|
// cmdDiff compares the machine code of two assembly files.
|
||||||
func cmdDiff(args []string) int {
|
func cmdDiff(args []string) int {
|
||||||
set := newCommand("diff", "gasm diff [-GOARCH arch] <file1.s> <file2.s>", `
|
set := newCommand("diff", "gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>", `
|
||||||
Compare the machine code produced by assembling two files.
|
Compare the machine code produced by assembling two files.
|
||||||
Shows which functions differ and the byte-level differences.
|
Shows which functions differ and the byte-level differences.
|
||||||
Useful for verifying that two implementations produce identical code,
|
Useful for verifying that two implementations produce identical code,
|
||||||
@@ -645,9 +693,11 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
|||||||
`)
|
`)
|
||||||
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
mapSpec := set.String("map", "", "comma-separated old=new pairs to match functions with different names")
|
||||||
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
|
archName := set.String("GOARCH", "", "target architecture for both files: amd64, arm64, riscv64 or loong64")
|
||||||
|
var dirs includeDirs
|
||||||
|
set.Var(&dirs, "I", "directory to search for #include files (may be repeated)")
|
||||||
set.Parse(args)
|
set.Parse(args)
|
||||||
if set.NArg() != 2 {
|
if set.NArg() != 2 {
|
||||||
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] <file1.s> <file2.s>")
|
fmt.Fprintln(os.Stderr, "usage: gasm diff [-GOARCH arch] [-I dir] <file1.s> <file2.s>")
|
||||||
return 2
|
return 2
|
||||||
}
|
}
|
||||||
path1, path2 := set.Arg(0), set.Arg(1)
|
path1, path2 := set.Arg(0), set.Arg(1)
|
||||||
@@ -675,12 +725,12 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Assemble both files.
|
// Assemble both files.
|
||||||
img1, err := assemblePath(path1, forced)
|
img1, err := assemblePath(path1, forced, dirs)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path1, err)
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img2, err := assemblePath(path2, forced)
|
img2, err := assemblePath(path2, forced, dirs)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
fmt.Fprintf(os.Stderr, "gasm diff: %s: %v\n", path2, err)
|
||||||
return 1
|
return 1
|
||||||
@@ -739,11 +789,34 @@ e.g. --map wideCopyAVX2=wideCopyAVX512 pairs the two regardless of suffix.
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// platformPredefines mirrors the go command's assembler invocation, which
|
||||||
|
// defines GOOS_<goos> and GOARCH_<arch> as -D macros: GOROOT headers
|
||||||
|
// (go_tls.h, asm_riscv64.h) select their platform blocks with #ifdef on
|
||||||
|
// exactly those names, so an assembler without them cannot see the platform
|
||||||
|
// definitions at all.
|
||||||
|
func platformPredefines(goarch, goos string) map[string]string {
|
||||||
|
return map[string]string{
|
||||||
|
"GOARCH_" + goarch: "1",
|
||||||
|
"GOOS_" + goos: "1",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// platformPredefinesFor resolves the ambient GOOS the way a build would: a
|
||||||
|
// file whose name carries one (sys_darwin_arm64.s) is compiled for that GOOS
|
||||||
|
// and nothing else.
|
||||||
|
func platformPredefinesFor(goarch string, fileGoos string) map[string]string {
|
||||||
|
goos := fileGoos
|
||||||
|
if goos == "" {
|
||||||
|
goos = runtime.GOOS
|
||||||
|
}
|
||||||
|
return platformPredefines(goarch, goos)
|
||||||
|
}
|
||||||
|
|
||||||
// assembleFile assembles a parsed file for the given architecture and returns the image.
|
// assembleFile assembles a parsed file for the given architecture and returns the image.
|
||||||
func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
|
func assembleFile(targetArch arch.Arch, f *ast.File, goos string) (*asm.Image, error) {
|
||||||
switch targetArch {
|
switch targetArch {
|
||||||
case arch.AMD64:
|
case arch.AMD64:
|
||||||
return asm.AssembleFile(f)
|
return asm.AssembleFile(f, asm.WithGOOS(goos))
|
||||||
case arch.RISCV:
|
case arch.RISCV:
|
||||||
return asm.AssembleFileRISCV(f)
|
return asm.AssembleFileRISCV(f)
|
||||||
case arch.ARM64:
|
case arch.ARM64:
|
||||||
@@ -755,25 +828,26 @@ func assembleFile(targetArch arch.Arch, f *ast.File) (*asm.Image, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// assemblePath reads, parses and assembles a file (used by cmdDiff). A
|
// assemblePath reads, preprocesses, parses and assembles a file (used by
|
||||||
// non-Unknown forced architecture overrides the file-name suffix.
|
// cmdDiff). A non-Unknown forced architecture overrides the file-name
|
||||||
func assemblePath(path string, forced arch.Arch) (*asm.Image, error) {
|
// suffix.
|
||||||
|
func assemblePath(path string, forced arch.Arch, dirs includeDirs) (*asm.Image, error) {
|
||||||
src, err := readSource(path)
|
src, err := readSource(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
f, errs := parser.Parse(path, src)
|
target := forced
|
||||||
|
if target == arch.Unknown {
|
||||||
|
target = arch.FromFilename(path)
|
||||||
|
}
|
||||||
|
f, errs := parser.ParseWithOptions(path, src, parser.Options{Expand: true, IncludeDirs: dirs, Predefines: platformPredefinesFor(string(target), "")})
|
||||||
for _, e := range errs {
|
for _, e := range errs {
|
||||||
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
fmt.Fprintf(os.Stderr, "%s: %v\n", path, e)
|
||||||
}
|
}
|
||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return nil, fmt.Errorf("parse errors")
|
return nil, fmt.Errorf("parse errors")
|
||||||
}
|
}
|
||||||
target := forced
|
return assembleFile(target, f, "")
|
||||||
if target == arch.Unknown {
|
|
||||||
target = arch.FromFilename(path)
|
|
||||||
}
|
|
||||||
return assembleFile(target, f)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// printByteDiff shows the first few byte differences between two code blocks.
|
// printByteDiff shows the first few byte differences between two code blocks.
|
||||||
@@ -869,7 +943,7 @@ func cmdVerifyNonJIT(path string, targetArch arch.Arch, groundTruth, profile boo
|
|||||||
if len(errs) > 0 {
|
if len(errs) > 0 {
|
||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
img, err := assembleFile(targetArch, f)
|
img, err := assembleFile(targetArch, f, "")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err)
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
@@ -14,8 +14,8 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
)
|
)
|
||||||
|
|
||||||
const clean = "#include \"textflag.h\"\n" +
|
const clean = "#include \"textflag.h\"\n" +
|
||||||
@@ -403,7 +403,7 @@ func TestRunCorpusAudit(t *testing.T) {
|
|||||||
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
|
write("generic.s", "#include \"textflag.h\"\nTEXT ·g(SB), NOSPLIT, $0-0\n\tRET\n")
|
||||||
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
|
write("broken.s", "#include \"textflag.h\"\nTEXT ·b(SB), NOSPLIT, $0-0\n\tJMP nowhere\n\tRET\n")
|
||||||
|
|
||||||
stats, err := runCorpusAudit(dir)
|
stats, err := runCorpusAudit(dir, nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("runCorpusAudit: %v", err)
|
t.Fatalf("runCorpusAudit: %v", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,109 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// writeTree writes a directory of files and returns its root.
|
||||||
|
func writeTree(t *testing.T, files map[string]string) string {
|
||||||
|
t.Helper()
|
||||||
|
dir := t.TempDir()
|
||||||
|
for name, content := range files {
|
||||||
|
path := filepath.Join(dir, name)
|
||||||
|
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dir
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAsmMacroAndIncludeEndToEnd drives `gasm asm` over a source with an
|
||||||
|
// in-file parameterised macro and an include resolved through -I, and checks
|
||||||
|
// the assembled bytes came from the expansion (the loop body counts six
|
||||||
|
// increments, two per expanded iteration).
|
||||||
|
func TestAsmMacroAndIncludeEndToEnd(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs the assembler end to end")
|
||||||
|
}
|
||||||
|
dir := writeTree(t, map[string]string{
|
||||||
|
"inc/consts.h": "#define NITER 3\n",
|
||||||
|
"main_amd64.s": "#include \"textflag.h\"\n" +
|
||||||
|
"#include \"consts.h\"\n" +
|
||||||
|
"#define STEP(r) ADDQ $1, r; ADDQ $1, r\n" +
|
||||||
|
"TEXT ·f(SB), NOSPLIT, $0-8\n" +
|
||||||
|
"\tXORQ AX, AX\n" +
|
||||||
|
"\tMOVQ $NITER, CX\n" +
|
||||||
|
"loop:\n" +
|
||||||
|
"\tSTEP(AX)\n" +
|
||||||
|
"\tDECQ CX\n" +
|
||||||
|
"\tJNZ loop\n" +
|
||||||
|
"\tMOVQ AX, ret+0(FP)\n" +
|
||||||
|
"\tRET\n",
|
||||||
|
})
|
||||||
|
stdout, stderr, code := capture(func() int {
|
||||||
|
return cmdAsm([]string{"-I", filepath.Join(dir, "inc"), "-GOARCH", "amd64", filepath.Join(dir, "main_amd64.s")})
|
||||||
|
})
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||||
|
}
|
||||||
|
// The macro expanded to two ADDQ $1 encodings in the static body; the
|
||||||
|
// iteration count lives in the runtime loop.
|
||||||
|
if n := strings.Count(stdout, "83 c0 01"); n != 2 {
|
||||||
|
t.Errorf("found %d ADDQ $1 encodings in the image, want 2:\n%s", n, stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAsmIncludeResolutionOrder pins the -I search order end to end: the
|
||||||
|
// including file's directory wins over the -I directories.
|
||||||
|
func TestAsmIncludeResolutionOrder(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs the assembler end to end")
|
||||||
|
}
|
||||||
|
dir := writeTree(t, map[string]string{
|
||||||
|
"src/main_amd64.s": "#include \"textflag.h\"\n" +
|
||||||
|
"#include \"vals.h\"\n" +
|
||||||
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||||
|
"\tMOVQ $VAL, AX\n" +
|
||||||
|
"\tRET\n",
|
||||||
|
"src/vals.h": "#define VAL 1\n",
|
||||||
|
"late/vals.h": "#define VAL 2\n",
|
||||||
|
"early/vals.h": "#define VAL 3\n",
|
||||||
|
})
|
||||||
|
stdout, stderr, code := capture(func() int {
|
||||||
|
return cmdAsm([]string{"-I", filepath.Join(dir, "early"), "-I", filepath.Join(dir, "late"),
|
||||||
|
"-GOARCH", "amd64", filepath.Join(dir, "src", "main_amd64.s")})
|
||||||
|
})
|
||||||
|
if code != 0 {
|
||||||
|
t.Fatalf("gasm asm exited %d: %s%s", code, stdout, stderr)
|
||||||
|
}
|
||||||
|
// VAL came from src/vals.h, not from either -I directory: the image
|
||||||
|
// loads the immediate 1.
|
||||||
|
if !strings.Contains(stdout, "b8 01 00 00 00") {
|
||||||
|
t.Errorf("expected the source-directory VAL (immediate 1) in:\n%s", stdout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAsmMissingIncludeIsAnError pins the diagnostic for an include that
|
||||||
|
// resolves nowhere on the assembly path.
|
||||||
|
func TestAsmMissingIncludeIsAnError(t *testing.T) {
|
||||||
|
if testing.Short() {
|
||||||
|
t.Skip("runs the assembler end to end")
|
||||||
|
}
|
||||||
|
path := writeTemp(t, "main_amd64.s", "#include \"textflag.h\"\n#include \"nothere.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
|
||||||
|
_, stderr, code := capture(func() int { return cmdAsm([]string{"-GOARCH", "amd64", path}) })
|
||||||
|
if code == 0 {
|
||||||
|
t.Fatal("gasm asm accepted a file whose include resolves nowhere")
|
||||||
|
}
|
||||||
|
if !strings.Contains(stderr, `#include "nothere.h"`) {
|
||||||
|
t.Errorf("stderr does not name the failing include: %s", stderr)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,8 +11,8 @@ import (
|
|||||||
"os"
|
"os"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
gasmast "sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
gasmast "sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
||||||
gasmparser "sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
gasmparser "sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
)
|
)
|
||||||
|
|
||||||
// cmdScaffold generates a differential test skeleton for every kernel in a
|
// cmdScaffold generates a differential test skeleton for every kernel in a
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build linux
|
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||||
|
|
||||||
package debug
|
package debug
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
mem, err := s.ReadMemory(addr, 15)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
ins, err := disasm.Decode(arch.AMD64, mem, addr)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result strings.Builder
|
||||||
|
pc := addr
|
||||||
|
for range n {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (x86asm.IntelSyntax) is a
|
||||||
|
// call. The first token must match exactly: a prefix test would also catch
|
||||||
|
// unrelated mnemonics.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
return strings.ToLower(m) == "call"
|
||||||
|
}
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && arm64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
mem, err := s.ReadMemory(addr, 4)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
ins, err := disasm.Decode(arch.ARM64, mem, addr)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result strings.Builder
|
||||||
|
pc := addr
|
||||||
|
for range n {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (arm64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders bl as CALL; the native mnemonic is accepted too.
|
||||||
|
// The first token must match exactly so branches never match.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "bl":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && riscv64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
// memory and returns its text representation and length in bytes.
|
||||||
|
func (s *Session) Disassemble(addr uint64) (string, int, error) {
|
||||||
|
mem, err := s.ReadMemory(addr, 4)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
ins, err := disasm.Decode(arch.RISCV, mem, addr)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, err
|
||||||
|
}
|
||||||
|
return ins.Text, ins.Len, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// DisassembleN decodes up to n instructions starting at addr and returns
|
||||||
|
// them as a formatted string with addresses and byte offsets.
|
||||||
|
func (s *Session) DisassembleN(addr uint64, n int) string {
|
||||||
|
var result strings.Builder
|
||||||
|
pc := addr
|
||||||
|
for range n {
|
||||||
|
text, length, err := s.Disassemble(pc)
|
||||||
|
if err != nil {
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: <error: %v>\n", pc, err))
|
||||||
|
break
|
||||||
|
}
|
||||||
|
result.WriteString(fmt.Sprintf(" %#08x: %s\n", pc, text))
|
||||||
|
if length == 0 {
|
||||||
|
length = 1
|
||||||
|
}
|
||||||
|
pc += uint64(length)
|
||||||
|
}
|
||||||
|
return result.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isCallInsn reports whether disassembled text (riscv64asm.GoSyntax) is a
|
||||||
|
// call. GoSyntax renders jal and jalr calls as CALL; the native mnemonics
|
||||||
|
// are accepted too. The first token must match exactly: a prefix test on
|
||||||
|
// "bl" would catch branches on other architectures, and jalr as ret prints
|
||||||
|
// RET, which must not be stepped over.
|
||||||
|
func isCallInsn(text string) bool {
|
||||||
|
m, _, _ := strings.Cut(text, " ")
|
||||||
|
switch strings.ToLower(m) {
|
||||||
|
case "call", "jal", "jalr":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
@@ -9,8 +9,8 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
|||||||
@@ -9,8 +9,8 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
|||||||
@@ -9,8 +9,8 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
|||||||
@@ -9,8 +9,8 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/disasm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/disasm"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Disassemble decodes the instruction at the given address in the debuggee's
|
// Disassemble decodes the instruction at the given address in the debuggee's
|
||||||
|
|||||||
@@ -0,0 +1,136 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||||
|
fmt.Printf(" RIP = %#016x (func+%#x)\n", regs.RIP, regs.RIP-codeBase-funcOff)
|
||||||
|
fmt.Printf(" RSP = %#016x RBP = %#016x\n", regs.RSP, regs.RBP)
|
||||||
|
fmt.Printf(" RAX = %#016x RBX = %#016x\n", regs.RAX, regs.RBX)
|
||||||
|
fmt.Printf(" RCX = %#016x RDX = %#016x\n", regs.RCX, regs.RDX)
|
||||||
|
fmt.Printf(" RSI = %#016x RDI = %#016x\n", regs.RSI, regs.RDI)
|
||||||
|
fmt.Printf(" R8 = %#016x R9 = %#016x\n", regs.R8, regs.R9)
|
||||||
|
fmt.Printf(" R10 = %#016x R11 = %#016x\n", regs.R10, regs.R11)
|
||||||
|
fmt.Printf(" R12 = %#016x R13 = %#016x\n", regs.R12, regs.R13)
|
||||||
|
fmt.Printf(" R14 = %#016x R15 = %#016x\n", regs.R14, regs.R15)
|
||||||
|
fmt.Printf(" RFLAGS = %#x [%s]\n", regs.RFLAGS, decodeRflags(regs.RFLAGS))
|
||||||
|
}
|
||||||
|
|
||||||
|
func printVectorRegs(v *VectorRegs) {
|
||||||
|
fmt.Println("\n Vector registers (YMM):")
|
||||||
|
for i := 0; i < 16; i += 2 {
|
||||||
|
fmt.Printf(" YMM%-2d = ", i)
|
||||||
|
printYMM(v.YMM[i][:])
|
||||||
|
fmt.Printf(" YMM%-2d = ", i+1)
|
||||||
|
printYMM(v.YMM[i+1][:])
|
||||||
|
fmt.Println()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func printYMM(b []byte) {
|
||||||
|
for j := 0; j < 32; j += 4 {
|
||||||
|
v := uint32(b[j]) | uint32(b[j+1])<<8 | uint32(b[j+2])<<16 | uint32(b[j+3])<<24
|
||||||
|
fmt.Printf("%08x ", v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func decodeRflags(f uint64) string {
|
||||||
|
var flags string
|
||||||
|
if f&1 != 0 {
|
||||||
|
flags += "CF "
|
||||||
|
}
|
||||||
|
if f&(1<<2) != 0 {
|
||||||
|
flags += "PF "
|
||||||
|
}
|
||||||
|
if f&(1<<4) != 0 {
|
||||||
|
flags += "AF "
|
||||||
|
}
|
||||||
|
if f&(1<<6) != 0 {
|
||||||
|
flags += "ZF "
|
||||||
|
}
|
||||||
|
if f&(1<<7) != 0 {
|
||||||
|
flags += "SF "
|
||||||
|
}
|
||||||
|
if f&(1<<8) != 0 {
|
||||||
|
flags += "TF "
|
||||||
|
}
|
||||||
|
if f&(1<<9) != 0 {
|
||||||
|
flags += "IF "
|
||||||
|
}
|
||||||
|
if f&(1<<10) != 0 {
|
||||||
|
flags += "DF "
|
||||||
|
}
|
||||||
|
if f&(1<<11) != 0 {
|
||||||
|
flags += "OF "
|
||||||
|
}
|
||||||
|
if flags == "" {
|
||||||
|
return "none"
|
||||||
|
}
|
||||||
|
return flags[:len(flags)-1]
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetReg modifies a register value in the debuggee.
|
||||||
|
func (s *Session) SetReg(name string, value uint64) error {
|
||||||
|
regs, err := s.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
switch name {
|
||||||
|
case "rax", "eax", "ax", "al":
|
||||||
|
regs.RAX = value
|
||||||
|
case "rbx", "ebx", "bx", "bl":
|
||||||
|
regs.RBX = value
|
||||||
|
case "rcx", "ecx", "cx", "cl":
|
||||||
|
regs.RCX = value
|
||||||
|
case "rdx", "edx", "dx", "dl":
|
||||||
|
regs.RDX = value
|
||||||
|
case "rsi", "esi", "si":
|
||||||
|
regs.RSI = value
|
||||||
|
case "rdi", "edi", "di":
|
||||||
|
regs.RDI = value
|
||||||
|
case "rbp", "ebp", "bp":
|
||||||
|
regs.RBP = value
|
||||||
|
case "rsp", "esp", "sp":
|
||||||
|
regs.RSP = value
|
||||||
|
case "r8":
|
||||||
|
regs.R8 = value
|
||||||
|
case "r9":
|
||||||
|
regs.R9 = value
|
||||||
|
case "r10":
|
||||||
|
regs.R10 = value
|
||||||
|
case "r11":
|
||||||
|
regs.R11 = value
|
||||||
|
case "r12":
|
||||||
|
regs.R12 = value
|
||||||
|
case "r13":
|
||||||
|
regs.R13 = value
|
||||||
|
case "r14":
|
||||||
|
regs.R14 = value
|
||||||
|
case "r15":
|
||||||
|
regs.R15 = value
|
||||||
|
case "rip", "eip":
|
||||||
|
regs.RIP = value
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("debug: unknown register %q", name)
|
||||||
|
}
|
||||||
|
return s.SetRegs(®s)
|
||||||
|
}
|
||||||
|
|
||||||
|
// archReturnAddr reads the return address of the current frame (amd64
|
||||||
|
// ABI0 convention). A function that contains a CALL (or has a frame) is
|
||||||
|
// assembled with the prologue PUSHQ BP; MOVQ SP, BP, so mid-function the
|
||||||
|
// word at SP is the saved caller BP, a stack address, and the return
|
||||||
|
// address sits further up. Walk the stack from SP and take the first word
|
||||||
|
// that lies in an executable mapping: stack and data words never do, a
|
||||||
|
// return address always does. FreeBSD exposes no mapping list, so the
|
||||||
|
// walk degenerates to the raw entry convention, [SP] before any push.
|
||||||
|
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||||
|
return s.Peek(regs.RSP)
|
||||||
|
}
|
||||||
|
|
||||||
|
// archSPLabel returns the SP register name for display.
|
||||||
|
func archSPLabel() string { return "RSP" }
|
||||||
@@ -0,0 +1,127 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && arm64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
)
|
||||||
|
|
||||||
|
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||||
|
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
|
||||||
|
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.SP, regs.X29)
|
||||||
|
fmt.Printf(" LR = %#016x\n", regs.X30)
|
||||||
|
fmt.Printf(" X0 = %#016x X1 = %#016x\n", regs.X0, regs.X1)
|
||||||
|
fmt.Printf(" X2 = %#016x X3 = %#016x\n", regs.X2, regs.X3)
|
||||||
|
fmt.Printf(" X4 = %#016x X5 = %#016x\n", regs.X4, regs.X5)
|
||||||
|
fmt.Printf(" X6 = %#016x X7 = %#016x\n", regs.X6, regs.X7)
|
||||||
|
fmt.Printf(" X8 = %#016x X9 = %#016x\n", regs.X8, regs.X9)
|
||||||
|
fmt.Printf(" X10 = %#016x X11 = %#016x\n", regs.X10, regs.X11)
|
||||||
|
fmt.Printf(" X12 = %#016x X13 = %#016x\n", regs.X12, regs.X13)
|
||||||
|
fmt.Printf(" X14 = %#016x X15 = %#016x\n", regs.X14, regs.X15)
|
||||||
|
fmt.Printf(" X16 = %#016x X17 = %#016x\n", regs.X16, regs.X17)
|
||||||
|
fmt.Printf(" X18 = %#016x X19 = %#016x\n", regs.X18, regs.X19)
|
||||||
|
fmt.Printf(" X20 = %#016x X21 = %#016x\n", regs.X20, regs.X21)
|
||||||
|
fmt.Printf(" X22 = %#016x X23 = %#016x\n", regs.X22, regs.X23)
|
||||||
|
fmt.Printf(" X24 = %#016x X25 = %#016x\n", regs.X24, regs.X25)
|
||||||
|
fmt.Printf(" X26 = %#016x X27 = %#016x\n", regs.X26, regs.X27)
|
||||||
|
fmt.Printf(" X28 = %#016x PSTATE = %#x\n", regs.X28, regs.PSTATE)
|
||||||
|
}
|
||||||
|
|
||||||
|
func printVectorRegs(v *VectorRegs) {
|
||||||
|
fmt.Println("\n Vector registers (V0-V31):")
|
||||||
|
for i := 0; i < 32; i += 2 {
|
||||||
|
fmt.Printf(" V%-2d = %016x%016x\n", i, binary.LittleEndian.Uint64(v.V[i][8:16]), binary.LittleEndian.Uint64(v.V[i][0:8]))
|
||||||
|
fmt.Printf(" V%-2d = %016x%016x\n", i+1, binary.LittleEndian.Uint64(v.V[i+1][8:16]), binary.LittleEndian.Uint64(v.V[i+1][0:8]))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetReg modifies a register value in the debuggee.
|
||||||
|
func (s *Session) SetReg(name string, value uint64) error {
|
||||||
|
regs, err := s.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
switch name {
|
||||||
|
case "x0":
|
||||||
|
regs.X0 = value
|
||||||
|
case "x1":
|
||||||
|
regs.X1 = value
|
||||||
|
case "x2":
|
||||||
|
regs.X2 = value
|
||||||
|
case "x3":
|
||||||
|
regs.X3 = value
|
||||||
|
case "x4":
|
||||||
|
regs.X4 = value
|
||||||
|
case "x5":
|
||||||
|
regs.X5 = value
|
||||||
|
case "x6":
|
||||||
|
regs.X6 = value
|
||||||
|
case "x7":
|
||||||
|
regs.X7 = value
|
||||||
|
case "x8":
|
||||||
|
regs.X8 = value
|
||||||
|
case "x9":
|
||||||
|
regs.X9 = value
|
||||||
|
case "x10":
|
||||||
|
regs.X10 = value
|
||||||
|
case "x11":
|
||||||
|
regs.X11 = value
|
||||||
|
case "x12":
|
||||||
|
regs.X12 = value
|
||||||
|
case "x13":
|
||||||
|
regs.X13 = value
|
||||||
|
case "x14":
|
||||||
|
regs.X14 = value
|
||||||
|
case "x15":
|
||||||
|
regs.X15 = value
|
||||||
|
case "x16":
|
||||||
|
regs.X16 = value
|
||||||
|
case "x17":
|
||||||
|
regs.X17 = value
|
||||||
|
case "x18":
|
||||||
|
regs.X18 = value
|
||||||
|
case "x19":
|
||||||
|
regs.X19 = value
|
||||||
|
case "x20":
|
||||||
|
regs.X20 = value
|
||||||
|
case "x21":
|
||||||
|
regs.X21 = value
|
||||||
|
case "x22":
|
||||||
|
regs.X22 = value
|
||||||
|
case "x23":
|
||||||
|
regs.X23 = value
|
||||||
|
case "x24":
|
||||||
|
regs.X24 = value
|
||||||
|
case "x25":
|
||||||
|
regs.X25 = value
|
||||||
|
case "x26":
|
||||||
|
regs.X26 = value
|
||||||
|
case "x27":
|
||||||
|
regs.X27 = value
|
||||||
|
case "x28":
|
||||||
|
regs.X28 = value
|
||||||
|
case "x29", "fp":
|
||||||
|
regs.X29 = value
|
||||||
|
case "x30", "lr":
|
||||||
|
regs.X30 = value
|
||||||
|
case "sp":
|
||||||
|
regs.SP = value
|
||||||
|
case "pc":
|
||||||
|
regs.PC = value
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("debug: unknown register %q", name)
|
||||||
|
}
|
||||||
|
return s.SetRegs(®s)
|
||||||
|
}
|
||||||
|
|
||||||
|
// archReturnAddr reads the return address from LR (arm64 convention).
|
||||||
|
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||||
|
return regs.X30, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// archSPLabel returns the SP register name for display.
|
||||||
|
func archSPLabel() string { return "SP" }
|
||||||
@@ -0,0 +1,121 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && riscv64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import "fmt"
|
||||||
|
|
||||||
|
func printRegs(regs *Regs, codeBase, funcOff uint64) {
|
||||||
|
fmt.Printf(" PC = %#016x (func+%#x)\n", regs.PC, regs.PC-codeBase-funcOff)
|
||||||
|
fmt.Printf(" SP = %#016x FP = %#016x\n", regs.Sp, regs.S0)
|
||||||
|
fmt.Printf(" RA = %#016x\n", regs.Ra)
|
||||||
|
fmt.Printf(" A0 = %#016x A1 = %#016x\n", regs.A0, regs.A1)
|
||||||
|
fmt.Printf(" A2 = %#016x A3 = %#016x\n", regs.A2, regs.A3)
|
||||||
|
fmt.Printf(" A4 = %#016x A5 = %#016x\n", regs.A4, regs.A5)
|
||||||
|
fmt.Printf(" A6 = %#016x A7 = %#016x\n", regs.A6, regs.A7)
|
||||||
|
fmt.Printf(" T0 = %#016x T1 = %#016x\n", regs.T0, regs.T1)
|
||||||
|
fmt.Printf(" T2 = %#016x T3 = %#016x\n", regs.T2, regs.T3)
|
||||||
|
fmt.Printf(" T4 = %#016x T5 = %#016x\n", regs.T4, regs.T5)
|
||||||
|
fmt.Printf(" T6 = %#016x\n", regs.T6)
|
||||||
|
fmt.Printf(" S1 = %#016x S2 = %#016x\n", regs.S1, regs.S2)
|
||||||
|
fmt.Printf(" S3 = %#016x S4 = %#016x\n", regs.S3, regs.S4)
|
||||||
|
fmt.Printf(" S5 = %#016x S6 = %#016x\n", regs.S5, regs.S6)
|
||||||
|
fmt.Printf(" S7 = %#016x S8 = %#016x\n", regs.S7, regs.S8)
|
||||||
|
fmt.Printf(" S9 = %#016x S10 = %#016x\n", regs.S9, regs.S10)
|
||||||
|
fmt.Printf(" S11 = %#016x\n", regs.S11)
|
||||||
|
}
|
||||||
|
|
||||||
|
func printVectorRegs(v *VectorRegs) {
|
||||||
|
fmt.Println("\n FP registers (F0-F31):")
|
||||||
|
for i := 0; i < 32; i += 2 {
|
||||||
|
fmt.Printf(" F%-2d = %#018x F%-2d = %#018x\n", i, v.F[i], i+1, v.F[i+1])
|
||||||
|
}
|
||||||
|
fmt.Printf(" FCSR = %#x\n", v.FCSR)
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetReg modifies a register value in the debuggee.
|
||||||
|
func (s *Session) SetReg(name string, value uint64) error {
|
||||||
|
regs, err := s.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
switch name {
|
||||||
|
case "pc":
|
||||||
|
regs.PC = value
|
||||||
|
case "ra", "x1":
|
||||||
|
regs.Ra = value
|
||||||
|
case "sp", "x2":
|
||||||
|
regs.Sp = value
|
||||||
|
case "gp", "x3":
|
||||||
|
regs.Gp = value
|
||||||
|
case "tp", "x4":
|
||||||
|
regs.Tp = value
|
||||||
|
case "t0", "x5":
|
||||||
|
regs.T0 = value
|
||||||
|
case "t1", "x6":
|
||||||
|
regs.T1 = value
|
||||||
|
case "t2", "x7":
|
||||||
|
regs.T2 = value
|
||||||
|
case "s0", "fp", "x8":
|
||||||
|
regs.S0 = value
|
||||||
|
case "s1", "x9":
|
||||||
|
regs.S1 = value
|
||||||
|
case "a0", "x10":
|
||||||
|
regs.A0 = value
|
||||||
|
case "a1", "x11":
|
||||||
|
regs.A1 = value
|
||||||
|
case "a2", "x12":
|
||||||
|
regs.A2 = value
|
||||||
|
case "a3", "x13":
|
||||||
|
regs.A3 = value
|
||||||
|
case "a4", "x14":
|
||||||
|
regs.A4 = value
|
||||||
|
case "a5", "x15":
|
||||||
|
regs.A5 = value
|
||||||
|
case "a6", "x16":
|
||||||
|
regs.A6 = value
|
||||||
|
case "a7", "x17":
|
||||||
|
regs.A7 = value
|
||||||
|
case "s2", "x18":
|
||||||
|
regs.S2 = value
|
||||||
|
case "s3", "x19":
|
||||||
|
regs.S3 = value
|
||||||
|
case "s4", "x20":
|
||||||
|
regs.S4 = value
|
||||||
|
case "s5", "x21":
|
||||||
|
regs.S5 = value
|
||||||
|
case "s6", "x22":
|
||||||
|
regs.S6 = value
|
||||||
|
case "s7", "x23":
|
||||||
|
regs.S7 = value
|
||||||
|
case "s8", "x24":
|
||||||
|
regs.S8 = value
|
||||||
|
case "s9", "x25":
|
||||||
|
regs.S9 = value
|
||||||
|
case "s10", "x26":
|
||||||
|
regs.S10 = value
|
||||||
|
case "s11", "x27":
|
||||||
|
regs.S11 = value
|
||||||
|
case "t3", "x28":
|
||||||
|
regs.T3 = value
|
||||||
|
case "t4", "x29":
|
||||||
|
regs.T4 = value
|
||||||
|
case "t5", "x30":
|
||||||
|
regs.T5 = value
|
||||||
|
case "t6", "x31":
|
||||||
|
regs.T6 = value
|
||||||
|
default:
|
||||||
|
return fmt.Errorf("debug: unknown register %q", name)
|
||||||
|
}
|
||||||
|
return s.SetRegs(®s)
|
||||||
|
}
|
||||||
|
|
||||||
|
// archReturnAddr reads the return address from RA (riscv64 convention).
|
||||||
|
func archReturnAddr(s *Session, regs *Regs) (uint64, error) {
|
||||||
|
return regs.Ra, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// archSPLabel returns the SP register name for display.
|
||||||
|
func archSPLabel() string { return "SP" }
|
||||||
@@ -17,8 +17,8 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
|
// Integration tests beyond the basic entry breakpoint: hardware watchpoints,
|
||||||
|
|||||||
@@ -0,0 +1,288 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && (amd64 || arm64 || riscv64)
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"strings"
|
||||||
|
"syscall"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Session is a ptrace debugging session controlling one debuggee process.
|
||||||
|
// The FreeBSD implementation sits behind the same surface as the Linux one:
|
||||||
|
// PT_TRACE_ME from the debuggee, PT_CONTINUE/PT_STEP from the tracer, and
|
||||||
|
// tracee memory through PT_IO (FreeBSD has no /proc/pid/mem to fall back
|
||||||
|
// on, so PT_IO is the only supported route).
|
||||||
|
type Session struct {
|
||||||
|
pid int
|
||||||
|
cmd *exec.Cmd
|
||||||
|
stopped bool
|
||||||
|
exited bool
|
||||||
|
codeBase uint64 // base address of the JIT code in the debuggee
|
||||||
|
tmpDir string // scratch directory of the session, removed on Kill
|
||||||
|
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 dbw 0-15)
|
||||||
|
// lastSignal holds the signal of the most recent stop when that stop
|
||||||
|
// was a genuine signal-delivery-stop the caller must see (a fault such
|
||||||
|
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
|
||||||
|
// single-steps, SIGSTOP and suppressed runtime signals.
|
||||||
|
lastSignal syscall.Signal
|
||||||
|
}
|
||||||
|
|
||||||
|
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
||||||
|
// attaches to it via ptrace.
|
||||||
|
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
|
||||||
|
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
|
||||||
|
return sess, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
|
||||||
|
//
|
||||||
|
// It pins the calling goroutine to its OS thread and leaves it pinned: the
|
||||||
|
// debuggee's PT_TRACE_ME binds the tracer relation to the forking thread,
|
||||||
|
// and every ptrace request on the session must come from that same thread.
|
||||||
|
// All Session methods must therefore be called from the goroutine that
|
||||||
|
// launched the session (the REPL and coverage loops do exactly that).
|
||||||
|
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
|
||||||
|
runtime.LockOSThread() // ptrace requests must stay on the forking thread
|
||||||
|
self, err := os.Executable()
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
|
||||||
|
}
|
||||||
|
if gasmBin != "" {
|
||||||
|
self = gasmBin
|
||||||
|
}
|
||||||
|
|
||||||
|
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
|
||||||
|
if err != nil {
|
||||||
|
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
|
||||||
|
}
|
||||||
|
argsFile := filepath.Join(tmpDir, "args.bin")
|
||||||
|
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: write args: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if bufSpec != "" {
|
||||||
|
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
|
||||||
|
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
|
||||||
|
cmd.Stdout = nil
|
||||||
|
cmd.Stderr = os.Stderr
|
||||||
|
cmd.SysProcAttr = &syscall.SysProcAttr{}
|
||||||
|
|
||||||
|
if err := cmd.Start(); err != nil {
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
|
||||||
|
|
||||||
|
readyFile := filepath.Join(tmpDir, "ready")
|
||||||
|
for range 500 {
|
||||||
|
if _, err := os.Stat(readyFile); err == nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
time.Sleep(5 * time.Millisecond)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
|
||||||
|
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
|
||||||
|
// so the wait loops until a stop the debugger cares about instead of
|
||||||
|
// assuming the first event is the SIGSTOP.
|
||||||
|
if _, err := s.waitStopped(); err != nil {
|
||||||
|
cmd.Process.Kill()
|
||||||
|
os.RemoveAll(tmpDir)
|
||||||
|
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
|
||||||
|
}
|
||||||
|
s.stopped = true
|
||||||
|
|
||||||
|
// The debuggee reports its JIT mapping in the codebase file; that is
|
||||||
|
// the supported path on FreeBSD, where no /proc/pid/maps exists to
|
||||||
|
// scan for the RWX region as a fallback.
|
||||||
|
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
|
||||||
|
fmt.Sscanf(string(data), "%d", &s.codeBase)
|
||||||
|
}
|
||||||
|
|
||||||
|
var bufAddrs []uint64
|
||||||
|
if bufSpec != "" {
|
||||||
|
addrFile := filepath.Join(tmpDir, "bufaddrs")
|
||||||
|
if data, err := os.ReadFile(addrFile); err == nil {
|
||||||
|
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
|
||||||
|
var addr uint64
|
||||||
|
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
|
||||||
|
bufAddrs = append(bufAddrs, addr)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return s, bufAddrs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// waitStopped consumes ptrace-stop events until one the debugger cares
|
||||||
|
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
|
||||||
|
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
|
||||||
|
// runtime raises SIGURG for asynchronous preemption, and every signal on a
|
||||||
|
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
|
||||||
|
// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS,
|
||||||
|
// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0
|
||||||
|
// would restart the faulting instruction and fault forever, so a faulting
|
||||||
|
// kernel must surface as a stop the caller reports.
|
||||||
|
func (s *Session) waitStopped() (syscall.Signal, error) {
|
||||||
|
for {
|
||||||
|
var ws syscall.WaitStatus
|
||||||
|
if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
if ws.Exited() {
|
||||||
|
s.exited = true
|
||||||
|
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
||||||
|
}
|
||||||
|
if ws.Signaled() {
|
||||||
|
s.exited = true
|
||||||
|
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
|
||||||
|
}
|
||||||
|
switch sig := ws.StopSignal(); sig {
|
||||||
|
case syscall.SIGTRAP, syscall.SIGSTOP:
|
||||||
|
s.stopped = true
|
||||||
|
s.lastSignal = 0
|
||||||
|
return sig, nil
|
||||||
|
case syscall.SIGURG:
|
||||||
|
// Go runtime asynchronous preemption: resume the tracee
|
||||||
|
// without delivering the signal.
|
||||||
|
s.lastSignal = 0
|
||||||
|
if err := unix.PtraceCont(s.pid, 0); err != nil {
|
||||||
|
return 0, fmt.Errorf("debug: PT_CONTINUE: %w", err)
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
// A genuine signal-delivery-stop. Report it; the caller
|
||||||
|
// decides how to proceed.
|
||||||
|
s.stopped = true
|
||||||
|
s.lastSignal = sig
|
||||||
|
return sig, nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// LastSignal returns the signal of the most recent stop when that stop was
|
||||||
|
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
|
||||||
|
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
|
||||||
|
// suppressed runtime signals.
|
||||||
|
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
|
||||||
|
|
||||||
|
// Peek reads a word (8 bytes) from the debuggee's memory at addr, through
|
||||||
|
// PT_IO with PIOD_READ_D.
|
||||||
|
func (s *Session) Peek(addr uint64) (uint64, error) {
|
||||||
|
var buf [8]byte
|
||||||
|
if _, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), buf[:], len(buf)); err != nil {
|
||||||
|
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
|
||||||
|
}
|
||||||
|
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
|
||||||
|
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Poke writes a word (8 bytes) to the debuggee's memory at addr, through
|
||||||
|
// PT_IO with PIOD_WRITE_D.
|
||||||
|
func (s *Session) Poke(addr, val uint64) error {
|
||||||
|
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
|
||||||
|
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
|
||||||
|
if _, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), buf, len(buf)); err != nil {
|
||||||
|
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// ReadMemory reads len bytes from the debuggee's memory at addr in one
|
||||||
|
// PT_IO request, the shape the request is built for.
|
||||||
|
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
|
||||||
|
out := make([]byte, length)
|
||||||
|
n, err := unix.PtraceIO(unix.PIOD_READ_D, s.pid, uintptr(addr), out, length)
|
||||||
|
return out[:n], err
|
||||||
|
}
|
||||||
|
|
||||||
|
// WriteMemory writes bytes to the debuggee's memory at addr in one PT_IO
|
||||||
|
// request.
|
||||||
|
func (s *Session) WriteMemory(addr uint64, data []byte) error {
|
||||||
|
_, err := unix.PtraceIO(unix.PIOD_WRITE_D, s.pid, uintptr(addr), data, len(data))
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Step executes a single instruction in the debuggee.
|
||||||
|
func (s *Session) Step() error {
|
||||||
|
if s.exited {
|
||||||
|
return fmt.Errorf("debug: debuggee has exited")
|
||||||
|
}
|
||||||
|
if err := unix.PtraceSingleStep(s.pid); err != nil {
|
||||||
|
return fmt.Errorf("debug: PT_STEP: %w", err)
|
||||||
|
}
|
||||||
|
_, err := s.waitStopped()
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Continue resumes execution until the next breakpoint or exit.
|
||||||
|
func (s *Session) Continue() error {
|
||||||
|
if s.exited {
|
||||||
|
return fmt.Errorf("debug: debuggee has exited")
|
||||||
|
}
|
||||||
|
if err := unix.PtraceCont(s.pid, 0); err != nil {
|
||||||
|
return fmt.Errorf("debug: PT_CONTINUE: %w", err)
|
||||||
|
}
|
||||||
|
_, err := s.waitStopped()
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Exited returns true if the debuggee has terminated.
|
||||||
|
func (s *Session) Exited() bool { return s.exited }
|
||||||
|
|
||||||
|
// Pid returns the debuggee's process ID.
|
||||||
|
func (s *Session) Pid() int { return s.pid }
|
||||||
|
|
||||||
|
// CodeBase returns the base address of the JIT code in the debuggee.
|
||||||
|
func (s *Session) CodeBase() uint64 { return s.codeBase }
|
||||||
|
|
||||||
|
// Kill terminates the debuggee and removes the session's scratch
|
||||||
|
// directory, so a successful session leaves no gasm-debug-* debris behind.
|
||||||
|
func (s *Session) Kill() {
|
||||||
|
if !s.exited {
|
||||||
|
syscall.Kill(s.pid, syscall.SIGKILL)
|
||||||
|
syscall.Wait4(s.pid, nil, 0, nil)
|
||||||
|
s.exited = true
|
||||||
|
}
|
||||||
|
if s.cmd != nil && s.cmd.Process != nil {
|
||||||
|
s.cmd.Wait()
|
||||||
|
}
|
||||||
|
if s.tmpDir != "" {
|
||||||
|
os.RemoveAll(s.tmpDir)
|
||||||
|
s.tmpDir = ""
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// execRange is one executable mapping of the debuggee.
|
||||||
|
type execRange struct {
|
||||||
|
lo, hi uint64
|
||||||
|
}
|
||||||
|
|
||||||
|
// execRanges is a stub on FreeBSD: there is no /proc/pid/maps to parse,
|
||||||
|
// and procfs(5) is not guaranteed to be mounted. The callers degrade
|
||||||
|
// gracefully: archReturnAddr falls back to the raw stack convention and
|
||||||
|
// the mapping scan is skipped.
|
||||||
|
func execRanges(pid int) []execRange { return nil }
|
||||||
|
|
||||||
|
// findRWXMapping is a stub on FreeBSD for the same reason: the codebase
|
||||||
|
// handshake file is the supported way the JIT region is located.
|
||||||
|
func findRWXMapping(pid int) uint64 { return 0 }
|
||||||
@@ -0,0 +1,160 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"fmt"
|
||||||
|
"unsafe"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// GetRegs reads the general-purpose registers of the stopped debuggee and
|
||||||
|
// converts the FreeBSD struct reg into the portable layout.
|
||||||
|
func (s *Session) GetRegs() (Regs, error) {
|
||||||
|
var ur unix.Reg
|
||||||
|
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
|
||||||
|
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
|
||||||
|
}
|
||||||
|
return Regs{
|
||||||
|
R15: uint64(ur.R15),
|
||||||
|
R14: uint64(ur.R14),
|
||||||
|
R13: uint64(ur.R13),
|
||||||
|
R12: uint64(ur.R12),
|
||||||
|
R11: uint64(ur.R11),
|
||||||
|
R10: uint64(ur.R10),
|
||||||
|
R9: uint64(ur.R9),
|
||||||
|
R8: uint64(ur.R8),
|
||||||
|
RDI: uint64(ur.Rdi),
|
||||||
|
RSI: uint64(ur.Rsi),
|
||||||
|
RBP: uint64(ur.Rbp),
|
||||||
|
RBX: uint64(ur.Rbx),
|
||||||
|
RDX: uint64(ur.Rdx),
|
||||||
|
RCX: uint64(ur.Rcx),
|
||||||
|
RAX: uint64(ur.Rax),
|
||||||
|
RIP: uint64(ur.Rip),
|
||||||
|
CS: uint64(ur.Cs),
|
||||||
|
RFLAGS: uint64(ur.Rflags),
|
||||||
|
RSP: uint64(ur.Rsp),
|
||||||
|
SS: uint64(ur.Ss),
|
||||||
|
FS: uint64(ur.Fs),
|
||||||
|
GS: uint64(ur.Gs),
|
||||||
|
DS: uint64(ur.Ds),
|
||||||
|
ES: uint64(ur.Es),
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||||
|
func (s *Session) SetRegs(regs *Regs) error {
|
||||||
|
// Read-modify-write keeps the fields FreeBSD owns (trapno, err) intact.
|
||||||
|
var ur unix.Reg
|
||||||
|
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
|
||||||
|
return fmt.Errorf("debug: PT_GETREGS: %w", err)
|
||||||
|
}
|
||||||
|
ur.R15 = int64(regs.R15)
|
||||||
|
ur.R14 = int64(regs.R14)
|
||||||
|
ur.R13 = int64(regs.R13)
|
||||||
|
ur.R12 = int64(regs.R12)
|
||||||
|
ur.R11 = int64(regs.R11)
|
||||||
|
ur.R10 = int64(regs.R10)
|
||||||
|
ur.R9 = int64(regs.R9)
|
||||||
|
ur.R8 = int64(regs.R8)
|
||||||
|
ur.Rdi = int64(regs.RDI)
|
||||||
|
ur.Rsi = int64(regs.RSI)
|
||||||
|
ur.Rbp = int64(regs.RBP)
|
||||||
|
ur.Rbx = int64(regs.RBX)
|
||||||
|
ur.Rdx = int64(regs.RDX)
|
||||||
|
ur.Rcx = int64(regs.RCX)
|
||||||
|
ur.Rax = int64(regs.RAX)
|
||||||
|
ur.Rip = int64(regs.RIP)
|
||||||
|
ur.Cs = int64(regs.CS)
|
||||||
|
ur.Rflags = int64(regs.RFLAGS)
|
||||||
|
ur.Rsp = int64(regs.RSP)
|
||||||
|
ur.Ss = int64(regs.SS)
|
||||||
|
return unix.PtraceSetRegs(s.pid, &ur)
|
||||||
|
}
|
||||||
|
|
||||||
|
// FPRegs holds the x87 FPU and SSE (XMM) register state, the FXSAVE image
|
||||||
|
// the FreeBSD struct fpreg mirrors: XMM0-15 at the same offsets.
|
||||||
|
type FPRegs struct {
|
||||||
|
XMM [16][16]byte // XMM0-15
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetFPRegs retrieves the FPU/SSE register state via PT_GETFPREGS. The
|
||||||
|
// FreeBSD struct fpreg mirrors the FXSAVE image: the x87 environment and
|
||||||
|
// stack in Env/Acc, XMM0-15 in Xacc.
|
||||||
|
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||||
|
var fp FPRegs
|
||||||
|
var fr unix.FpReg
|
||||||
|
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
|
||||||
|
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
|
||||||
|
}
|
||||||
|
for i := range 16 {
|
||||||
|
copy(fp.XMM[i][:], fr.Xacc[i][:])
|
||||||
|
}
|
||||||
|
return fp, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// VectorRegs holds the YMM register state.
|
||||||
|
type VectorRegs struct {
|
||||||
|
YMM [16][32]byte // YMM0-15 (full 256-bit values)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The XSAVE area the PT_GETXSTATE request returns follows the architectural
|
||||||
|
// layout (Intel SDM vol 1, "XSAVE"): the 512-byte legacy FXSAVE image (x87
|
||||||
|
// state in 0-159, XMM0-15 in 160-511), then the 64-byte xsave header whose
|
||||||
|
// first 8 bytes are xstate_bv, then one component per set feature bit, each
|
||||||
|
// 64-byte aligned. The YMM high halves are the first extended component,
|
||||||
|
// at offset 576; XFEATURE_STATE_BIT_AVX is bit 2 of xstate_bv.
|
||||||
|
const (
|
||||||
|
xsaveXMMOffset = 160
|
||||||
|
xsaveHeaderOffset = 512
|
||||||
|
xsaveBVOffset = xsaveHeaderOffset
|
||||||
|
ymmOffset = xsaveHeaderOffset + 64 // 576
|
||||||
|
ymmSize = 256 // 16 registers, 16 bytes each
|
||||||
|
xfeatureMaskYMM = 1 << 2
|
||||||
|
xstateMaxBuffer = 4096 // PT_GETXSTATE_INFO bounds the size far below this
|
||||||
|
)
|
||||||
|
|
||||||
|
// GetVectorRegs retrieves the YMM registers via PT_GETXSTATE. The low
|
||||||
|
// (XMM) halves always come from the legacy image; the high halves are
|
||||||
|
// copied only when xstate_bv reports the AVX state, and read as zero
|
||||||
|
// otherwise. When the request fails the FP image still provides correct
|
||||||
|
// XMM halves, so that is the fallback.
|
||||||
|
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||||
|
var v VectorRegs
|
||||||
|
buf := make([]byte, xstateMaxBuffer)
|
||||||
|
n, _, errno := unix.Syscall6(
|
||||||
|
unix.SYS_PTRACE,
|
||||||
|
uintptr(unix.PT_GETXSTATE),
|
||||||
|
uintptr(s.pid),
|
||||||
|
0,
|
||||||
|
uintptr(unsafe.Pointer(&buf[0])),
|
||||||
|
0, 0,
|
||||||
|
)
|
||||||
|
if errno != 0 {
|
||||||
|
fp, err := s.GetFPRegs()
|
||||||
|
if err != nil {
|
||||||
|
return v, err
|
||||||
|
}
|
||||||
|
for i := range 16 {
|
||||||
|
copy(v.YMM[i][:16], fp.XMM[i][:])
|
||||||
|
}
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
|
for i := range 16 {
|
||||||
|
copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16])
|
||||||
|
}
|
||||||
|
if int(n) >= ymmOffset+ymmSize {
|
||||||
|
if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 {
|
||||||
|
for i := range 16 {
|
||||||
|
copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,114 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && arm64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// GetRegs reads the general-purpose registers of the stopped debuggee and
|
||||||
|
// converts the FreeBSD struct reg (x[30], lr, sp, elr, spsr) into the
|
||||||
|
// portable layout.
|
||||||
|
func (s *Session) GetRegs() (Regs, error) {
|
||||||
|
var ur unix.Reg
|
||||||
|
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
|
||||||
|
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
|
||||||
|
}
|
||||||
|
return Regs{
|
||||||
|
X0: ur.X[0],
|
||||||
|
X1: ur.X[1],
|
||||||
|
X2: ur.X[2],
|
||||||
|
X3: ur.X[3],
|
||||||
|
X4: ur.X[4],
|
||||||
|
X5: ur.X[5],
|
||||||
|
X6: ur.X[6],
|
||||||
|
X7: ur.X[7],
|
||||||
|
X8: ur.X[8],
|
||||||
|
X9: ur.X[9],
|
||||||
|
X10: ur.X[10],
|
||||||
|
X11: ur.X[11],
|
||||||
|
X12: ur.X[12],
|
||||||
|
X13: ur.X[13],
|
||||||
|
X14: ur.X[14],
|
||||||
|
X15: ur.X[15],
|
||||||
|
X16: ur.X[16],
|
||||||
|
X17: ur.X[17],
|
||||||
|
X18: ur.X[18],
|
||||||
|
X19: ur.X[19],
|
||||||
|
X20: ur.X[20],
|
||||||
|
X21: ur.X[21],
|
||||||
|
X22: ur.X[22],
|
||||||
|
X23: ur.X[23],
|
||||||
|
X24: ur.X[24],
|
||||||
|
X25: ur.X[25],
|
||||||
|
X26: ur.X[26],
|
||||||
|
X27: ur.X[27],
|
||||||
|
X28: ur.X[28],
|
||||||
|
X29: ur.X[29],
|
||||||
|
X30: ur.Lr,
|
||||||
|
SP: ur.Sp,
|
||||||
|
PC: ur.Elr,
|
||||||
|
PSTATE: uint64(ur.Spsr),
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||||
|
func (s *Session) SetRegs(regs *Regs) error {
|
||||||
|
var ur unix.Reg
|
||||||
|
ur.X = [30]uint64{
|
||||||
|
regs.X0, regs.X1, regs.X2, regs.X3, regs.X4, regs.X5, regs.X6,
|
||||||
|
regs.X7, regs.X8, regs.X9, regs.X10, regs.X11, regs.X12, regs.X13,
|
||||||
|
regs.X14, regs.X15, regs.X16, regs.X17, regs.X18, regs.X19, regs.X20,
|
||||||
|
regs.X21, regs.X22, regs.X23, regs.X24, regs.X25, regs.X26, regs.X27,
|
||||||
|
regs.X28, regs.X29,
|
||||||
|
}
|
||||||
|
ur.Lr = regs.X30
|
||||||
|
ur.Sp = regs.SP
|
||||||
|
ur.Elr = regs.PC
|
||||||
|
ur.Spsr = uint32(regs.PSTATE)
|
||||||
|
return unix.PtraceSetRegs(s.pid, &ur)
|
||||||
|
}
|
||||||
|
|
||||||
|
// FPRegs holds the arm64 FP/NEON register state: the 32 128-bit V
|
||||||
|
// registers, then FPSR and FPCR (the user_fpsimd shape).
|
||||||
|
type FPRegs struct {
|
||||||
|
V [32][16]byte // V0-V31 (128-bit NEON/FP registers)
|
||||||
|
FPSR uint32
|
||||||
|
FPCR uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetFPRegs retrieves the FP/NEON register state via PT_GETFPREGS. The
|
||||||
|
// FreeBSD struct fpreg holds the 32 128-bit V registers followed by FPSR
|
||||||
|
// and FPCR, the user_fpsimd shape.
|
||||||
|
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||||
|
var fp FPRegs
|
||||||
|
var fr unix.FpReg
|
||||||
|
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
|
||||||
|
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
|
||||||
|
}
|
||||||
|
for i := range 32 {
|
||||||
|
copy(fp.V[i][:], fr.Q[i][:])
|
||||||
|
}
|
||||||
|
return fp, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// VectorRegs holds the full SIMD register state.
|
||||||
|
type VectorRegs struct {
|
||||||
|
V [32][16]byte // V0-V31 (128-bit)
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetVectorRegs retrieves the SIMD registers.
|
||||||
|
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||||
|
var v VectorRegs
|
||||||
|
fp, err := s.GetFPRegs()
|
||||||
|
if err != nil {
|
||||||
|
return v, err
|
||||||
|
}
|
||||||
|
copy(v.V[:][:], fp.V[:][:])
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && riscv64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// GetRegs reads the general-purpose registers of the stopped debuggee and
|
||||||
|
// converts the FreeBSD struct reg into the portable layout. Sstatus rides
|
||||||
|
// the kernel's struct but the portable surface carries the GPRs and PC.
|
||||||
|
func (s *Session) GetRegs() (Regs, error) {
|
||||||
|
var ur unix.Reg
|
||||||
|
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
|
||||||
|
return Regs{}, fmt.Errorf("debug: PT_GETREGS: %w", err)
|
||||||
|
}
|
||||||
|
return Regs{
|
||||||
|
PC: ur.Sepc,
|
||||||
|
Ra: ur.Ra,
|
||||||
|
Sp: ur.Sp,
|
||||||
|
Gp: ur.Gp,
|
||||||
|
Tp: ur.Tp,
|
||||||
|
T0: ur.T[0],
|
||||||
|
T1: ur.T[1],
|
||||||
|
T2: ur.T[2],
|
||||||
|
S0: ur.S[0],
|
||||||
|
S1: ur.S[1],
|
||||||
|
A0: ur.A[0],
|
||||||
|
A1: ur.A[1],
|
||||||
|
A2: ur.A[2],
|
||||||
|
A3: ur.A[3],
|
||||||
|
A4: ur.A[4],
|
||||||
|
A5: ur.A[5],
|
||||||
|
A6: ur.A[6],
|
||||||
|
A7: ur.A[7],
|
||||||
|
S2: ur.S[2],
|
||||||
|
S3: ur.S[3],
|
||||||
|
S4: ur.S[4],
|
||||||
|
S5: ur.S[5],
|
||||||
|
S6: ur.S[6],
|
||||||
|
S7: ur.S[7],
|
||||||
|
S8: ur.S[8],
|
||||||
|
S9: ur.S[9],
|
||||||
|
S10: ur.S[10],
|
||||||
|
S11: ur.S[11],
|
||||||
|
T3: ur.T[3],
|
||||||
|
T4: ur.T[4],
|
||||||
|
T5: ur.T[5],
|
||||||
|
T6: ur.T[6],
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// SetRegs writes the general-purpose registers of the stopped debuggee.
|
||||||
|
// Read-modify-write keeps sstatus, which the kernel owns, intact.
|
||||||
|
func (s *Session) SetRegs(regs *Regs) error {
|
||||||
|
var ur unix.Reg
|
||||||
|
if err := unix.PtraceGetRegs(s.pid, &ur); err != nil {
|
||||||
|
return fmt.Errorf("debug: PT_GETREGS: %w", err)
|
||||||
|
}
|
||||||
|
ur.Sepc = regs.PC
|
||||||
|
ur.Ra = regs.Ra
|
||||||
|
ur.Sp = regs.Sp
|
||||||
|
ur.Gp = regs.Gp
|
||||||
|
ur.Tp = regs.Tp
|
||||||
|
ur.T = [7]uint64{regs.T0, regs.T1, regs.T2, regs.T3, regs.T4, regs.T5, regs.T6}
|
||||||
|
ur.S = [12]uint64{regs.S0, regs.S1, regs.S2, regs.S3, regs.S4, regs.S5,
|
||||||
|
regs.S6, regs.S7, regs.S8, regs.S9, regs.S10, regs.S11}
|
||||||
|
ur.A = [8]uint64{regs.A0, regs.A1, regs.A2, regs.A3, regs.A4, regs.A5, regs.A6, regs.A7}
|
||||||
|
return unix.PtraceSetRegs(s.pid, &ur)
|
||||||
|
}
|
||||||
|
|
||||||
|
// FPRegs holds the RISC-V FP register state (32 64-bit FP registers plus
|
||||||
|
// fcsr).
|
||||||
|
type FPRegs struct {
|
||||||
|
F [32]uint64 // F0-F31 (64-bit FP registers)
|
||||||
|
FCSR uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetFPRegs retrieves the FP register state via PT_GETFPREGS. The FreeBSD
|
||||||
|
// struct fpreg carries each 64-bit FP register in a 128-bit slot (fp_x is
|
||||||
|
// the flat [64]-word area the x/sys type renders as [32][2]); the low word
|
||||||
|
// holds the register, and FCSR rides the tail.
|
||||||
|
func (s *Session) GetFPRegs() (FPRegs, error) {
|
||||||
|
var fp FPRegs
|
||||||
|
var fr unix.FpReg
|
||||||
|
if err := unix.PtraceGetFpRegs(s.pid, &fr); err != nil {
|
||||||
|
return fp, fmt.Errorf("debug: PT_GETFPREGS: %w", err)
|
||||||
|
}
|
||||||
|
for i := range 32 {
|
||||||
|
fp.F[i] = fr.X[i][0]
|
||||||
|
}
|
||||||
|
fp.FCSR = uint32(fr.Fcsr)
|
||||||
|
return fp, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// VectorRegs holds the FP register state shown by the regs command
|
||||||
|
// (riscv64 has 32 64-bit FP registers and fcsr).
|
||||||
|
type VectorRegs struct {
|
||||||
|
F [32]uint64
|
||||||
|
FCSR uint32
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetVectorRegs retrieves the FP registers.
|
||||||
|
func (s *Session) GetVectorRegs() (VectorRegs, error) {
|
||||||
|
fp, err := s.GetFPRegs()
|
||||||
|
if err != nil {
|
||||||
|
return VectorRegs{}, err
|
||||||
|
}
|
||||||
|
return VectorRegs{F: fp.F, FCSR: fp.FCSR}, nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os/exec"
|
||||||
|
"path/filepath"
|
||||||
|
"runtime"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestLaunchAndBreakpoint is the FreeBSD twin of the Linux integration
|
||||||
|
// test: it drives the whole launch, breakpoint, trap and register-rewind
|
||||||
|
// flow end to end. It needs a real FreeBSD kernel (ptrace does not work
|
||||||
|
// under emulation), so it only runs where it can.
|
||||||
|
func TestLaunchAndBreakpoint(t *testing.T) {
|
||||||
|
if runtime.GOARCH != "amd64" {
|
||||||
|
t.Skip("runs only on amd64 hosts")
|
||||||
|
}
|
||||||
|
// The tracer is the OS thread that forked the debuggee (PT_TRACE_ME
|
||||||
|
// binds the relation to that thread); every ptrace request must come
|
||||||
|
// from the same thread, so pin the test goroutine to one thread.
|
||||||
|
runtime.LockOSThread()
|
||||||
|
defer runtime.UnlockOSThread()
|
||||||
|
|
||||||
|
bin := filepath.Join(t.TempDir(), "gasm")
|
||||||
|
out, err := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm").CombinedOutput()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("build gasm: %v: %s", err, out)
|
||||||
|
}
|
||||||
|
|
||||||
|
const kernelPath = "../testdata/verify/basic_amd64.s"
|
||||||
|
k, err := verify.Load(kernelPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(k.Close)
|
||||||
|
fl, err := k.Func("wideCopy")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Func: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sess, err := Launch(bin, kernelPath, "wideCopy", make([]byte, fl.Args))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Launch: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(sess.Kill)
|
||||||
|
|
||||||
|
bm := NewBreakpoints(sess)
|
||||||
|
entry := sess.CodeBase() + uint64(fl.Offset)
|
||||||
|
if _, err := bm.Set(entry, "entry"); err != nil {
|
||||||
|
t.Fatalf("Set: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The INT3 must be visible in the debuggee's memory.
|
||||||
|
word, err := sess.Peek(entry)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Peek: %v", err)
|
||||||
|
}
|
||||||
|
if b := word & 0xFF; b != 0xCC {
|
||||||
|
t.Fatalf("int3 not patched: first byte %#02x at %#x", b, entry)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The debuggee raises a second SIGSTOP after the launch barrier (the
|
||||||
|
// child's RunTarget marks its entry), so like the REPL and the cover
|
||||||
|
// mode the test keeps resuming until the breakpoint trap arrives.
|
||||||
|
for range 10 {
|
||||||
|
if err := sess.Continue(); err != nil {
|
||||||
|
t.Fatalf("Continue: %v", err)
|
||||||
|
}
|
||||||
|
if sess.Exited() {
|
||||||
|
t.Fatal("debuggee exited instead of trapping on the breakpoint")
|
||||||
|
}
|
||||||
|
regs, err := sess.GetRegs()
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GetRegs: %v", err)
|
||||||
|
}
|
||||||
|
if bp := bm.HandleTrap(®s); bp != nil {
|
||||||
|
if bp.Addr != entry {
|
||||||
|
t.Fatalf("trap at %#x, want %#x", bp.Addr, entry)
|
||||||
|
}
|
||||||
|
return // trap on the entry breakpoint: the whole flow works
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatal("no breakpoint trap after 10 resumes")
|
||||||
|
}
|
||||||
@@ -14,7 +14,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
|
// buildGasm produces the gasm binary the debugger spawns as its debuggee.
|
||||||
@@ -24,7 +24,7 @@ func buildGasm(t *testing.T) string {
|
|||||||
return p
|
return p
|
||||||
}
|
}
|
||||||
bin := filepath.Join(t.TempDir(), "gasm")
|
bin := filepath.Join(t.TempDir(), "gasm")
|
||||||
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm")
|
cmd := exec.Command("go", "build", "-o", bin, "sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm")
|
||||||
out, err := cmd.CombinedOutput()
|
out, err := cmd.CombinedOutput()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("build gasm: %v: %s", err, out)
|
t.Fatalf("build gasm: %v: %s", err, out)
|
||||||
|
|||||||
@@ -0,0 +1,98 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && amd64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
// Regs holds the full general-purpose register set of a traced process
|
||||||
|
// (the FreeBSD amd64 struct reg layout, sys/x86/include/reg.h). FreeBSD
|
||||||
|
// reports segment selectors (FS/GS/ES/DS), not the bases the Linux ptrace
|
||||||
|
// surface carries, and has no ORIG_RAX slot.
|
||||||
|
type Regs struct {
|
||||||
|
R15 uint64
|
||||||
|
R14 uint64
|
||||||
|
R13 uint64
|
||||||
|
R12 uint64
|
||||||
|
RBP uint64
|
||||||
|
RBX uint64
|
||||||
|
R11 uint64
|
||||||
|
R10 uint64
|
||||||
|
R9 uint64
|
||||||
|
R8 uint64
|
||||||
|
RAX uint64
|
||||||
|
RCX uint64
|
||||||
|
RDX uint64
|
||||||
|
RSI uint64
|
||||||
|
RDI uint64
|
||||||
|
RIP uint64
|
||||||
|
CS uint64
|
||||||
|
RFLAGS uint64
|
||||||
|
RSP uint64
|
||||||
|
SS uint64
|
||||||
|
FS uint64
|
||||||
|
GS uint64
|
||||||
|
DS uint64
|
||||||
|
ES uint64
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetPC returns the program counter.
|
||||||
|
func (r *Regs) GetPC() uint64 { return r.RIP }
|
||||||
|
|
||||||
|
// SetPC sets the program counter.
|
||||||
|
func (r *Regs) SetPC(pc uint64) { r.RIP = pc }
|
||||||
|
|
||||||
|
// GetSP returns the stack pointer.
|
||||||
|
func (r *Regs) GetSP() uint64 { return r.RSP }
|
||||||
|
|
||||||
|
// RegValue returns the value of the named register, or false if unknown.
|
||||||
|
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||||
|
switch name {
|
||||||
|
case "rax", "eax", "ax", "al":
|
||||||
|
return r.RAX, true
|
||||||
|
case "rbx", "ebx", "bx", "bl":
|
||||||
|
return r.RBX, true
|
||||||
|
case "rcx", "ecx", "cx", "cl":
|
||||||
|
return r.RCX, true
|
||||||
|
case "rdx", "edx", "dx", "dl":
|
||||||
|
return r.RDX, true
|
||||||
|
case "rsi", "esi", "si":
|
||||||
|
return r.RSI, true
|
||||||
|
case "rdi", "edi", "di":
|
||||||
|
return r.RDI, true
|
||||||
|
case "rbp", "ebp", "bp":
|
||||||
|
return r.RBP, true
|
||||||
|
case "rsp", "esp", "sp":
|
||||||
|
return r.RSP, true
|
||||||
|
case "r8":
|
||||||
|
return r.R8, true
|
||||||
|
case "r9":
|
||||||
|
return r.R9, true
|
||||||
|
case "r10":
|
||||||
|
return r.R10, true
|
||||||
|
case "r11":
|
||||||
|
return r.R11, true
|
||||||
|
case "r12":
|
||||||
|
return r.R12, true
|
||||||
|
case "r13":
|
||||||
|
return r.R13, true
|
||||||
|
case "r14":
|
||||||
|
return r.R14, true
|
||||||
|
case "r15":
|
||||||
|
return r.R15, true
|
||||||
|
case "rip", "eip":
|
||||||
|
return r.RIP, true
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// breakpointInsn is the software breakpoint instruction.
|
||||||
|
var breakpointInsn = []byte{0xCC} // INT3
|
||||||
|
|
||||||
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
|
// a trap. INT3 leaves the hardware PC on the following instruction (Intel
|
||||||
|
// SDM vol 3, "Debug Exceptions") and the FreeBSD T_BPTFLT path delivers
|
||||||
|
// that frame unmodified (sys/amd64/amd64/trap.c), so the trap address is
|
||||||
|
// PC-1, the same correction the Linux side applies.
|
||||||
|
const breakpointPCAdjust = 1
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && arm64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
// Regs holds the full general-purpose register set of a traced process
|
||||||
|
// (the FreeBSD arm64 struct reg layout, sys/arm64/include/reg.h: x[30], lr,
|
||||||
|
// sp, elr, spsr).
|
||||||
|
type Regs struct {
|
||||||
|
X0 uint64
|
||||||
|
X1 uint64
|
||||||
|
X2 uint64
|
||||||
|
X3 uint64
|
||||||
|
X4 uint64
|
||||||
|
X5 uint64
|
||||||
|
X6 uint64
|
||||||
|
X7 uint64
|
||||||
|
X8 uint64
|
||||||
|
X9 uint64
|
||||||
|
X10 uint64
|
||||||
|
X11 uint64
|
||||||
|
X12 uint64
|
||||||
|
X13 uint64
|
||||||
|
X14 uint64
|
||||||
|
X15 uint64
|
||||||
|
X16 uint64
|
||||||
|
X17 uint64
|
||||||
|
X18 uint64
|
||||||
|
X19 uint64
|
||||||
|
X20 uint64
|
||||||
|
X21 uint64
|
||||||
|
X22 uint64
|
||||||
|
X23 uint64
|
||||||
|
X24 uint64
|
||||||
|
X25 uint64
|
||||||
|
X26 uint64
|
||||||
|
X27 uint64
|
||||||
|
X28 uint64
|
||||||
|
X29 uint64 // FP (frame pointer)
|
||||||
|
X30 uint64 // LR (link register)
|
||||||
|
SP uint64
|
||||||
|
PC uint64
|
||||||
|
PSTATE uint64
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetPC returns the program counter.
|
||||||
|
func (r *Regs) GetPC() uint64 { return r.PC }
|
||||||
|
|
||||||
|
// SetPC sets the program counter.
|
||||||
|
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
|
||||||
|
|
||||||
|
// GetSP returns the stack pointer.
|
||||||
|
func (r *Regs) GetSP() uint64 { return r.SP }
|
||||||
|
|
||||||
|
// RegValue returns the value of the named register, or false if unknown.
|
||||||
|
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||||
|
switch name {
|
||||||
|
case "x0":
|
||||||
|
return r.X0, true
|
||||||
|
case "x1":
|
||||||
|
return r.X1, true
|
||||||
|
case "x2":
|
||||||
|
return r.X2, true
|
||||||
|
case "x3":
|
||||||
|
return r.X3, true
|
||||||
|
case "x4":
|
||||||
|
return r.X4, true
|
||||||
|
case "x5":
|
||||||
|
return r.X5, true
|
||||||
|
case "x6":
|
||||||
|
return r.X6, true
|
||||||
|
case "x7":
|
||||||
|
return r.X7, true
|
||||||
|
case "x8":
|
||||||
|
return r.X8, true
|
||||||
|
case "x9":
|
||||||
|
return r.X9, true
|
||||||
|
case "x10":
|
||||||
|
return r.X10, true
|
||||||
|
case "x11":
|
||||||
|
return r.X11, true
|
||||||
|
case "x12":
|
||||||
|
return r.X12, true
|
||||||
|
case "x13":
|
||||||
|
return r.X13, true
|
||||||
|
case "x14":
|
||||||
|
return r.X14, true
|
||||||
|
case "x15":
|
||||||
|
return r.X15, true
|
||||||
|
case "x16":
|
||||||
|
return r.X16, true
|
||||||
|
case "x17":
|
||||||
|
return r.X17, true
|
||||||
|
case "x18":
|
||||||
|
return r.X18, true
|
||||||
|
case "x19":
|
||||||
|
return r.X19, true
|
||||||
|
case "x20":
|
||||||
|
return r.X20, true
|
||||||
|
case "x21":
|
||||||
|
return r.X21, true
|
||||||
|
case "x22":
|
||||||
|
return r.X22, true
|
||||||
|
case "x23":
|
||||||
|
return r.X23, true
|
||||||
|
case "x24":
|
||||||
|
return r.X24, true
|
||||||
|
case "x25":
|
||||||
|
return r.X25, true
|
||||||
|
case "x26":
|
||||||
|
return r.X26, true
|
||||||
|
case "x27":
|
||||||
|
return r.X27, true
|
||||||
|
case "x28":
|
||||||
|
return r.X28, true
|
||||||
|
case "x29", "fp":
|
||||||
|
return r.X29, true
|
||||||
|
case "x30", "lr":
|
||||||
|
return r.X30, true
|
||||||
|
case "sp":
|
||||||
|
return r.SP, true
|
||||||
|
case "pc":
|
||||||
|
return r.PC, true
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// breakpointInsn is the software breakpoint instruction (BRK #0).
|
||||||
|
var breakpointInsn = []byte{0x00, 0x00, 0x20, 0xD4} // BRK #0
|
||||||
|
|
||||||
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
|
// a trap: 0. The BRK synchronous exception leaves ELR_EL0 on the BRK
|
||||||
|
// itself (ARM DDI 0487), and the FreeBSD EXCP_BRKPT_EL0 handler delivers
|
||||||
|
// the frame's elr unmodified (sys/arm64/arm64/trap.c), so the trap address
|
||||||
|
// is the PC as reported.
|
||||||
|
const breakpointPCAdjust = 0
|
||||||
@@ -0,0 +1,134 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && riscv64
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
// Regs holds the full general-purpose register set of a traced process
|
||||||
|
// (the FreeBSD riscv64 struct reg layout: ra, sp, gp, tp, t0-t6, s0-s11,
|
||||||
|
// a0-a7, sepc, sstatus).
|
||||||
|
type Regs struct {
|
||||||
|
PC uint64 // sepc
|
||||||
|
Ra uint64 // x1 (return address)
|
||||||
|
Sp uint64 // x2
|
||||||
|
Gp uint64 // x3
|
||||||
|
Tp uint64 // x4
|
||||||
|
T0 uint64 // x5
|
||||||
|
T1 uint64 // x6
|
||||||
|
T2 uint64 // x7
|
||||||
|
S0 uint64 // x8 (frame pointer)
|
||||||
|
S1 uint64 // x9
|
||||||
|
A0 uint64 // x10
|
||||||
|
A1 uint64 // x11
|
||||||
|
A2 uint64 // x12
|
||||||
|
A3 uint64 // x13
|
||||||
|
A4 uint64 // x14
|
||||||
|
A5 uint64 // x15
|
||||||
|
A6 uint64 // x16
|
||||||
|
A7 uint64 // x17
|
||||||
|
S2 uint64 // x18
|
||||||
|
S3 uint64 // x19
|
||||||
|
S4 uint64 // x20
|
||||||
|
S5 uint64 // x21
|
||||||
|
S6 uint64 // x22
|
||||||
|
S7 uint64 // x23
|
||||||
|
S8 uint64 // x24
|
||||||
|
S9 uint64 // x25
|
||||||
|
S10 uint64 // x26
|
||||||
|
S11 uint64 // x27
|
||||||
|
T3 uint64 // x28
|
||||||
|
T4 uint64 // x29
|
||||||
|
T5 uint64 // x30
|
||||||
|
T6 uint64 // x31
|
||||||
|
}
|
||||||
|
|
||||||
|
// GetPC returns the program counter.
|
||||||
|
func (r *Regs) GetPC() uint64 { return r.PC }
|
||||||
|
|
||||||
|
// SetPC sets the program counter.
|
||||||
|
func (r *Regs) SetPC(pc uint64) { r.PC = pc }
|
||||||
|
|
||||||
|
// GetSP returns the stack pointer.
|
||||||
|
func (r *Regs) GetSP() uint64 { return r.Sp }
|
||||||
|
|
||||||
|
// RegValue returns the value of the named register, or false if unknown.
|
||||||
|
func (r *Regs) RegValue(name string) (uint64, bool) {
|
||||||
|
switch name {
|
||||||
|
case "pc":
|
||||||
|
return r.PC, true
|
||||||
|
case "ra", "x1":
|
||||||
|
return r.Ra, true
|
||||||
|
case "sp", "x2":
|
||||||
|
return r.Sp, true
|
||||||
|
case "gp", "x3":
|
||||||
|
return r.Gp, true
|
||||||
|
case "tp", "x4":
|
||||||
|
return r.Tp, true
|
||||||
|
case "t0", "x5":
|
||||||
|
return r.T0, true
|
||||||
|
case "t1", "x6":
|
||||||
|
return r.T1, true
|
||||||
|
case "t2", "x7":
|
||||||
|
return r.T2, true
|
||||||
|
case "s0", "fp", "x8":
|
||||||
|
return r.S0, true
|
||||||
|
case "s1", "x9":
|
||||||
|
return r.S1, true
|
||||||
|
case "a0", "x10":
|
||||||
|
return r.A0, true
|
||||||
|
case "a1", "x11":
|
||||||
|
return r.A1, true
|
||||||
|
case "a2", "x12":
|
||||||
|
return r.A2, true
|
||||||
|
case "a3", "x13":
|
||||||
|
return r.A3, true
|
||||||
|
case "a4", "x14":
|
||||||
|
return r.A4, true
|
||||||
|
case "a5", "x15":
|
||||||
|
return r.A5, true
|
||||||
|
case "a6", "x16":
|
||||||
|
return r.A6, true
|
||||||
|
case "a7", "x17":
|
||||||
|
return r.A7, true
|
||||||
|
case "s2", "x18":
|
||||||
|
return r.S2, true
|
||||||
|
case "s3", "x19":
|
||||||
|
return r.S3, true
|
||||||
|
case "s4", "x20":
|
||||||
|
return r.S4, true
|
||||||
|
case "s5", "x21":
|
||||||
|
return r.S5, true
|
||||||
|
case "s6", "x22":
|
||||||
|
return r.S6, true
|
||||||
|
case "s7", "x23":
|
||||||
|
return r.S7, true
|
||||||
|
case "s8", "x24":
|
||||||
|
return r.S8, true
|
||||||
|
case "s9", "x25":
|
||||||
|
return r.S9, true
|
||||||
|
case "s10", "x26":
|
||||||
|
return r.S10, true
|
||||||
|
case "s11", "x27":
|
||||||
|
return r.S11, true
|
||||||
|
case "t3", "x28":
|
||||||
|
return r.T3, true
|
||||||
|
case "t4", "x29":
|
||||||
|
return r.T4, true
|
||||||
|
case "t5", "x30":
|
||||||
|
return r.T5, true
|
||||||
|
case "t6", "x31":
|
||||||
|
return r.T6, true
|
||||||
|
default:
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// breakpointInsn is the software breakpoint instruction (EBREAK).
|
||||||
|
var breakpointInsn = []byte{0x73, 0x00, 0x10, 0x00} // ebreak
|
||||||
|
|
||||||
|
// breakpointPCAdjust is how far PC is past the breakpoint instruction after
|
||||||
|
// a trap: 0. The EBREAK synchronous exception leaves sepc on the ebreak
|
||||||
|
// itself (RISC-V privileged architecture), so the trap address is the PC as
|
||||||
|
// reported.
|
||||||
|
const breakpointPCAdjust = 0
|
||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
// SPDX-License-Identifier: BSD-3-Clause
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
//go:build linux
|
//go:build linux || (freebsd && (amd64 || arm64 || riscv64))
|
||||||
|
|
||||||
package debug
|
package debug
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,73 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && (amd64 || arm64 || riscv64)
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/binary"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
)
|
||||||
|
|
||||||
|
// FreeBSD TRAP_* si_code values (sys/signal.h). A breakpoint (INT3, BRK,
|
||||||
|
// EBREAK) arrives as TRAP_BRKPT on every supported architecture; TRAP_TRACE
|
||||||
|
// is shared by the completed single-step and the hardware watchpoint hit,
|
||||||
|
// so the watchpoint layer disambiguates from the debug registers.
|
||||||
|
const (
|
||||||
|
trapBRKPT = 1 // TRAP_BRKPT
|
||||||
|
trapTRACE = 2 // TRAP_TRACE
|
||||||
|
)
|
||||||
|
|
||||||
|
// StopReason describes why the debuggee stopped.
|
||||||
|
type StopReason int
|
||||||
|
|
||||||
|
const (
|
||||||
|
StopNone StopReason = iota
|
||||||
|
StopBreakpoint // software breakpoint hit
|
||||||
|
StopWatchpoint // hardware watchpoint triggered
|
||||||
|
StopSingleStep // single-step completed
|
||||||
|
StopSignal // stopped by a signal
|
||||||
|
StopExited // process exited
|
||||||
|
)
|
||||||
|
|
||||||
|
// StopInfo returns the reason the debuggee stopped and the faulting address
|
||||||
|
// (for watchpoints, the watched address that was accessed). FreeBSD has no
|
||||||
|
// PTRACE_GETSIGINFO; the stop's signal information comes from PT_LWPINFO,
|
||||||
|
// whose pl_siginfo carries the siginfo the kernel delivered. A ptrace stop
|
||||||
|
// with no signal behind it (a completed single-step, the initial attach)
|
||||||
|
// fills no siginfo at all.
|
||||||
|
func (s *Session) StopInfo() (StopReason, uint64) {
|
||||||
|
if s.exited {
|
||||||
|
return StopExited, 0
|
||||||
|
}
|
||||||
|
var info unix.PtraceLwpInfoStruct
|
||||||
|
if err := unix.PtraceLwpInfo(s.pid, &info); err != nil {
|
||||||
|
return StopNone, 0
|
||||||
|
}
|
||||||
|
// The siginfo layout is the FreeBSD siginfo_t: three leading ints
|
||||||
|
// (signo, errno, code), then the union, 8-byte aligned, whose _fault
|
||||||
|
// member puts the address at byte offset 16. The read is byte-wise
|
||||||
|
// because the blob's alignment is not guaranteed.
|
||||||
|
si := (*[64]byte)(unsafe.Pointer(&info.Siginfo))
|
||||||
|
signo := int32(binary.LittleEndian.Uint32(si[0:4]))
|
||||||
|
code := int32(binary.LittleEndian.Uint32(si[8:12]))
|
||||||
|
switch {
|
||||||
|
case signo == 0:
|
||||||
|
// A pure ptrace stop: single-step completion, attach, or the
|
||||||
|
// events the kernel resolves internally.
|
||||||
|
return StopSingleStep, 0
|
||||||
|
case signo != int32(syscall.SIGTRAP):
|
||||||
|
return StopSignal, uint64(code)
|
||||||
|
case code == trapBRKPT:
|
||||||
|
return StopBreakpoint, 0
|
||||||
|
case code == trapTRACE:
|
||||||
|
addr := binary.LittleEndian.Uint64(si[16:24])
|
||||||
|
return archStopTrace(s, addr)
|
||||||
|
default:
|
||||||
|
return StopSingleStep, 0
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,195 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
//go:build freebsd && (amd64 || arm64 || riscv64)
|
||||||
|
|
||||||
|
package debug
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/hex"
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
"runtime"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"syscall"
|
||||||
|
"unsafe"
|
||||||
|
|
||||||
|
"golang.org/x/sys/unix"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
|
)
|
||||||
|
|
||||||
|
// mapRWX maps code into a read-write-execute region.
|
||||||
|
func mapRWX(code []byte) ([]byte, error) {
|
||||||
|
const pageSize = 4096
|
||||||
|
size := (len(code) + pageSize - 1) &^ (pageSize - 1)
|
||||||
|
mem, err := syscall.Mmap(-1, 0, size,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE|syscall.PROT_EXEC,
|
||||||
|
syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
copy(mem, code)
|
||||||
|
return mem, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// setupBuffers allocates buffers in the debuggee's memory.
|
||||||
|
func setupBuffers(spec string, args []byte, tmpDir string) ([]byte, error) {
|
||||||
|
type bufSpec struct {
|
||||||
|
name string
|
||||||
|
size int
|
||||||
|
pattern string
|
||||||
|
}
|
||||||
|
var specs []bufSpec
|
||||||
|
for part := range strings.SplitSeq(spec, ",") {
|
||||||
|
fields := strings.SplitN(part, ":", 3)
|
||||||
|
if len(fields) != 3 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
size, err := strconv.Atoi(fields[1])
|
||||||
|
if err != nil || size <= 0 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
specs = append(specs, bufSpec{name: fields[0], size: size, pattern: fields[2]})
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(specs) == 0 {
|
||||||
|
return args, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
var bufAddrs []uint64
|
||||||
|
for _, s := range specs {
|
||||||
|
buf, err := syscall.Mmap(-1, 0, s.size,
|
||||||
|
syscall.PROT_READ|syscall.PROT_WRITE,
|
||||||
|
syscall.MAP_PRIVATE|syscall.MAP_ANON)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("mmap buffer %s: %w", s.name, err)
|
||||||
|
}
|
||||||
|
fillBuffer(buf, s.pattern)
|
||||||
|
bufAddrs = append(bufAddrs, uint64(uintptr(unsafe.Pointer(&buf[0]))))
|
||||||
|
}
|
||||||
|
|
||||||
|
addrFile, err := os.Create(tmpDir + "/bufaddrs")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
for _, addr := range bufAddrs {
|
||||||
|
fmt.Fprintf(addrFile, "%d\n", addr)
|
||||||
|
}
|
||||||
|
addrFile.Close()
|
||||||
|
|
||||||
|
return args, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// fillBuffer fills a buffer with the specified pattern.
|
||||||
|
func fillBuffer(buf []byte, pattern string) {
|
||||||
|
switch pattern {
|
||||||
|
case "zero":
|
||||||
|
case "ones":
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = 0xFF
|
||||||
|
}
|
||||||
|
case "seq":
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = byte(i)
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
if data, err := hex.DecodeString(pattern); err == nil && len(data) > 0 {
|
||||||
|
for i := range buf {
|
||||||
|
buf[i] = data[i%len(data)]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// RunTarget is the debuggee entry point (gasm debug --target).
|
||||||
|
func RunTarget(asmPath, funcName, argsFile, tmpDir string) error {
|
||||||
|
src, err := os.ReadFile(asmPath)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: %w", err)
|
||||||
|
}
|
||||||
|
file, errs := parser.Parse(asmPath, string(src))
|
||||||
|
if len(errs) > 0 {
|
||||||
|
return fmt.Errorf("debug target: parse: %v", errs[0])
|
||||||
|
}
|
||||||
|
img, err := asm.AssembleFile(file)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: assemble: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var fl *asm.FuncLayout
|
||||||
|
for i := range img.Funcs {
|
||||||
|
if img.Funcs[i].Name == funcName {
|
||||||
|
fl = &img.Funcs[i]
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if fl == nil {
|
||||||
|
return fmt.Errorf("debug target: function %q not found", funcName)
|
||||||
|
}
|
||||||
|
|
||||||
|
code := img.Bytes()
|
||||||
|
exec, err := mapRWX(code)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: mmap: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
codeBase := uintptr(unsafe.Pointer(&exec[0]))
|
||||||
|
if err := os.WriteFile(tmpDir+"/codebase", []byte(fmt.Sprintf("%d", codeBase)), 0o644); err != nil {
|
||||||
|
return fmt.Errorf("debug target: write codebase: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
meta := fmt.Sprintf("%d %d %d", fl.Offset, fl.Size, fl.Args)
|
||||||
|
os.WriteFile(tmpDir+"/funcmeta", []byte(meta), 0o644)
|
||||||
|
|
||||||
|
labelsFile, _ := os.Create(tmpDir + "/labels")
|
||||||
|
if labelsFile != nil {
|
||||||
|
for label, off := range fl.Labels {
|
||||||
|
fmt.Fprintf(labelsFile, "%s %d\n", label, off)
|
||||||
|
}
|
||||||
|
labelsFile.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
args, err := os.ReadFile(argsFile)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: read args: %w", err)
|
||||||
|
}
|
||||||
|
if len(args) < fl.Args {
|
||||||
|
padded := make([]byte, fl.Args)
|
||||||
|
copy(padded, args)
|
||||||
|
args = padded
|
||||||
|
}
|
||||||
|
|
||||||
|
bufSpecFile := tmpDir + "/bufspec"
|
||||||
|
if bufSpec, err := os.ReadFile(bufSpecFile); err == nil && len(bufSpec) > 0 {
|
||||||
|
args, err = setupBuffers(string(bufSpec), args, tmpDir)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("debug target: setup buffers: %w", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
runtime.LockOSThread()
|
||||||
|
|
||||||
|
if _, _, errno := unix.RawSyscall(unix.SYS_PTRACE, uintptr(unix.PT_TRACE_ME), 0, 0); errno != 0 {
|
||||||
|
return fmt.Errorf("debug target: PT_TRACE_ME: %v", errno)
|
||||||
|
}
|
||||||
|
os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644)
|
||||||
|
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
|
||||||
|
|
||||||
|
os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644)
|
||||||
|
syscall.Kill(syscall.Getpid(), syscall.SIGSTOP)
|
||||||
|
|
||||||
|
fnAddr := codeBase + uintptr(fl.Offset)
|
||||||
|
stackArgs := make([]byte, fl.Args)
|
||||||
|
copy(stackArgs, args)
|
||||||
|
|
||||||
|
if _, callErr := verify.Call(fnAddr, stackArgs); callErr != nil {
|
||||||
|
os.Exit(1)
|
||||||
|
}
|
||||||
|
// Success returns to the caller, which exits with status 0; the JIT
|
||||||
|
// code has already run to its own trampoline by the time Call returns.
|
||||||
|
return nil
|
||||||
|
}
|
||||||
@@ -12,9 +12,9 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// RunTarget is the debuggee entry point (gasm debug --target).
|
// RunTarget is the debuggee entry point (gasm debug --target).
|
||||||
|
|||||||
@@ -12,9 +12,9 @@ import (
|
|||||||
"syscall"
|
"syscall"
|
||||||
"unsafe"
|
"unsafe"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
|
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/verify"
|
"sourcedock.dev/petrbalvin/gasm-sdk/verify"
|
||||||
)
|
)
|
||||||
|
|
||||||
// RunTarget is the debuggee entry point (gasm debug --target).
|
// RunTarget is the debuggee entry point (gasm debug --target).
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user