Compare commits

..
218 Commits
Author SHA1 Message Date
petrbalvin a77dd12d2d chore: prepare release v0.36.0
Test / test (push) Successful in 32s
Release / build (amd64, freebsd) (push) Successful in 52s
Release / build (amd64, linux) (push) Successful in 13s
Release / build (arm64, freebsd) (push) Successful in 51s
Release / build (arm64, linux) (push) Successful in 29s
Release / build (loong64, linux) (push) Successful in 29s
Release / build (riscv64, linux) (push) Successful in 31s
Release / release (push) Successful in 10s
Assisted-by: GLM 5.3 Flash
2026-10-07 22:37:42 +02:00
petrbalvin e5df0a9d84 fix(asm): adapt the flag list integration test to the paired oracle
Test / test (push) Successful in 31s
Assisted-by: GLM 5.3 Flash
2026-10-07 21:52:27 +02:00
petrbalvin a9ee2bc702 test(parser): pin the trailing comment of a TEXT and GLOBL header
Assisted-by: GLM 5.3 Flash
2026-10-07 21:50:20 +02:00
petrbalvin 6f3e054bab test(asm): pin the flag-list TEXT shapes against the toolchain
Assisted-by: GLM 5.3 Flash
2026-10-07 21:50:20 +02:00
petrbalvin 9179d5cc7a test(format): pin the canonical rendering of the flags operand
Assisted-by: GLM 5.3 Flash
2026-10-07 21:50:20 +02:00
petrbalvin 431d0d9b0b feat(parser): read the TEXT and GLOBL operands the toolchain counts them
Assisted-by: GLM 5.3 Flash
2026-10-07 21:50:20 +02:00
petrbalvin cea5db6964 feat(parser): evaluate the TEXT and GLOBL flags operand as one expression
Assisted-by: GLM 5.3 Flash
2026-10-07 21:50:20 +02:00
petrbalvin 4eb9100def test(asm): walk the loong64 audit backlog under every operand shape
Assisted-by: GLM 5.3 Flash
2026-10-07 21:42:14 +02:00
petrbalvin 43494a30f0 test(disasm): pin the toolchain's own ADDU16I.D rendering
Assisted-by: GLM 5.3 Flash
2026-10-07 21:42:14 +02:00
petrbalvin f661c2fc78 test(asm): pin the ADDV16 immediate family the toolchain assembles
Assisted-by: GLM 5.3 Flash
2026-10-07 21:42:14 +02:00
petrbalvin a3eaa82f87 feat(asm): take the loong64 register-pair spellings the toolchain parses
Assisted-by: GLM 5.3 Flash
2026-10-07 21:42:14 +02:00
petrbalvin 1031cd9ae7 feat(arch): the same-size shift-immediate family
Assisted-by: GLM 5.3 Flash
2026-10-07 21:39:45 +02:00
petrbalvin eda8b16c5f feat(arch): the shift-immediate tsz:imm3 scheme
Assisted-by: GLM 5.3 Flash
2026-10-07 21:39:45 +02:00
petrbalvin 718181e7a7 feat(arch): the FP8-to-halfword conversion pairs
Assisted-by: GLM 5.3 Flash
2026-10-07 21:39:45 +02:00
petrbalvin 107ca512b2 feat(arch): the ZCNOT unary pair and the vector counter step
Assisted-by: GLM 5.3 Flash
2026-10-07 21:39:45 +02:00
petrbalvin 4066396226 feat(asm): encode the GETCALLERPC, REM, DWORD and half-FCVT shapes
Assisted-by: GLM 5.3 Flash
2026-10-07 21:34:30 +02:00
petrbalvin df7a5091e0 test(asm): pair the same-named GOROOT functions in definition order
Assisted-by: GLM 5.3 Flash
2026-10-07 21:34:30 +02:00
petrbalvin c27ba30862 feat(asm): encode the arm64 local-exec TLS load
Assisted-by: GLM 5.3 Flash
2026-10-07 21:34:30 +02:00
petrbalvin 5f6be4584d fix(asm): encode the flag-setting logicals to ZR and fold immediate expressions
Assisted-by: GLM 5.3 Flash
2026-10-07 21:34:30 +02:00
petrbalvin c1cef7b6e8 feat(asm): materialise frame-relative addresses the way the toolchain does
Assisted-by: GLM 5.3 Flash
2026-10-07 21:34:30 +02:00
petrbalvin 1120a52a54 test(asm): sweep every registered amd64 extension mnemonic from .s text
Assisted-by: GLM 5.3 Flash
2026-10-07 20:56:25 +02:00
petrbalvin fa50521619 feat(lint): surface the amd64 extension layer's refusals
Assisted-by: GLM 5.3 Flash
2026-10-07 20:56:25 +02:00
petrbalvin 5ee60c860b fix(cmd/gasm): probe the arm64 operand shapes the encoder accepts
Assisted-by: GLM 5.3 Flash
2026-10-07 20:39:44 +02:00
petrbalvin 76f8ba6403 feat(asm): give riscv64 the END and GETCALLERPC the toolchain accepts
Assisted-by: GLM 5.3 Flash
2026-10-07 20:39:44 +02:00
petrbalvin 0211d6672d feat(asm): expand riscv64 memory offsets beyond the 12-bit immediate
Assisted-by: GLM 5.3 Flash
2026-10-07 20:39:44 +02:00
petrbalvin 7aba29ac67 test(arch): pin the golden vectors of the last four families
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:42 +02:00
petrbalvin aeb109a64c feat(arch): the vector-length arithmetic pseudo group
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin bd85f8838d feat(arch): the SVE2.1 last-active vector and compare families
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin 12a5cfff52 feat(arch): the rest of the SVE2 BFloat16 wall
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin cdc3a75c88 feat(arch): the SVE multiple-structure loads and stores
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin 82dbf087a8 feat(arch): the SVE2 BFloat16 arithmetic core
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin 0760cc91af feat(arch): the SVE2 three-source and bitwise combine families
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin bd80499c74 feat(arch): the SVE2 shift-by-vector family
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin b93fccc075 feat(arch): the SVE2.1 pairwise and quadword-reduction families
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin 31ab584eee feat(arch): the SVE2.1 narrowing two-to-one family
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin ee2c6d51b3 fix(arch): keep Zdn out of the class bits of the predicated Z-alias source
Assisted-by: GLM 5.3 Flash
2026-10-07 20:35:41 +02:00
petrbalvin 6e49cbd099 feat(asm): assemble the extended instruction layer on amd64
Assisted-by: GLM 5.3 Flash
2026-10-07 20:24:51 +02:00
petrbalvin d4878524e8 test(asm): pin the clean GOROOT arm64 set against the toolchain
Assisted-by: GLM 5.3 Flash
2026-10-07 19:54:22 +02:00
petrbalvin 07f4622ffe style(asm): range over the kernel generator's statement count
Assisted-by: GLM 5.3 Flash
2026-10-07 19:54:22 +02:00
petrbalvin b3ede5f952 test(asm): pin the mid-function pool flush against the toolchain
Assisted-by: GLM 5.3 Flash
2026-10-07 19:54:22 +02:00
petrbalvin 5a5936d222 feat(asm): drain the arm64 literal pool mid-function at the distance bound
Assisted-by: GLM 5.3 Flash
2026-10-07 19:54:22 +02:00
petrbalvin 4632ac1bb9 feat(arch): add the AVX-VNNI-INT16 dot products to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 19:49:15 +02:00
petrbalvin 96000dd64d feat(arch): add the VEX encoder to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 19:49:15 +02:00
petrbalvin 47d561b229 refactor(arch): extract the displacement tail and SIB builders
Assisted-by: GLM 5.3 Flash
2026-10-07 19:49:15 +02:00
petrbalvin 27ef71859b feat(arch): add VMINMAXSH to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 19:49:15 +02:00
petrbalvin 03f9ef0ac6 feat(arch): add the FP16 complex fused multiply-add to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 19:49:15 +02:00
petrbalvin 86cb785e58 fix(asm): read the MOV family registers against their banks
Assisted-by: GLM 5.3 Flash
2026-10-07 19:40:32 +02:00
petrbalvin 4ff47fe17a fix(cmd/gasm): probe the riscv64 operand shapes the encoder accepts
Assisted-by: GLM 5.3 Flash
2026-10-07 19:40:32 +02:00
petrbalvin d18195f581 test(asm): pin the riscv64 error parity against the toolchain catalogues
Assisted-by: GLM 5.3 Flash
2026-10-07 19:40:32 +02:00
petrbalvin f0942ff7f3 fix(asm): reject the operand shapes the toolchain rejects on riscv64
Assisted-by: GLM 5.3 Flash
2026-10-07 19:40:32 +02:00
petrbalvin 6c66f5bbd9 test(asm): encode the new FP16 families through the registry
Assisted-by: GLM 5.3 Flash
2026-10-07 19:00:44 +02:00
petrbalvin 509afbb6c9 feat(arch): add the FP16 complex multiply and minimum-maximum to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 18:59:09 +02:00
petrbalvin 8080e0acef feat(arch): add the AVX512-FP16 FMA families to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 18:58:49 +02:00
petrbalvin 87493d9391 build(justfile): anchor the fuzz recipe to one exact target
Test / test (push) Successful in 1m2s
Assisted-by: GLM 5.3
2026-10-07 15:00:28 +02:00
petrbalvin 63ed141522 fix(parser): reject macro parameter lists the toolchain rejects
The C variadic spellings ("..." and the GNU "name..."), an empty or
trailing parameter, a missing comma, a stray token and an unterminated
list all parsed silently before: non-identifier tokens were skipped, so
"#define M(...)" defined a zero-parameter macro and invocations were
diagnosed only by argument count, if at all.  The toolchain rejects the
definition itself ("bad definition for macro"), and so does the
preprocessor now: the parameter list must be identifiers separated by
single commas and closed by the parenthesis, and a rejected definition
leaves the name unbound.  The spellings join the fuzz corpus.

Assisted-by: GLM 5.3
2026-10-07 14:04:15 +02:00
petrbalvin 373ec51061 fix(format): treat a selector-folded label as naming its macro
The lexer folds NAME.selector into one identifier, and an object macro
reached through the selector expands with it travelling along, so a
label spelled NAME.selector: restructures exactly like the bare name
would.  The macro-name checks now resolve the prefix before the first
period, and the crashing input joins the corpus.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin f932c5811c fix(format): never peel a stacked label that names a macro
Peeling a stacked label whose name is a macro moves it onto a line of
its own, where its expansion decides the line's shape: an empty body
leaves a bare colon behind, a line the parser rejects.  The peel loops
now hold such labels back with the rest of the line, and the crashing
input joins the corpus.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin b54d2b4520 fix(format): keep macro content behind a label on its line
The canonical form splits a label from the instruction that follows it,
but an identifier naming a macro may expand into any token at all: moved
to a line of its own it no longer parses, because the parser accepts a
non-mnemonic first token only behind a label.  The names #define'd in
the file now hold such content back, conservatively across the whole
file, and the crashing input joins the corpus.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin f452b8a995 fix(format): keep a macro-named label line whole
A label whose name is a macro expands into whatever the body is, so the
line's statement structure exists only after expansion; splitting the
label off changed clean input into a different statement sequence.  The
formatter now records the names #define'd above each line and renders
such a label line unsplit, and the crashing input joins the corpus.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 715298ed23 test(format): keep the expansion signature off the EOF comment newline
An unterminated block comment at the end of a file swallows the
formatter's mandatory final newline, an artifact the token view already
documents as layout; the preproc signature entry now trims it so the
invariant stays about tokens.  The triggering input joins the corpus.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin bbe14dd1ab fix(parser): diagnose GLOBL and DATA without a symbol name
A bare GLOBL or DATA parsed without a single diagnostic while leaving a
nil Name in the tree, a pointer every downstream tool dereferences; TEXT
keeps a placeholder beside its error for exactly that reason, and GLOBL
and DATA now do the same.  The crashing input enters the corpus.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 05de774c0a test(format): pin the macro-layout crashers as corpus seeds
The two inputs the expansion round-trip target was built for, the object
macro with a parenthesised body and the continuation-only macro body,
enter the seed corpus so every plain go test run replays them.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 6218023a63 style(parser): range over the include chain depth
Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 71cc65ea48 fix(format): separate an empty TEXT body from the next block
The one-blank rule before a new block skipped every line that followed a
TEXT directive, not just the function's first label, so a TEXT with an
empty body ran straight into the next declaration.  The exemption now
applies to labels only, and the directive shapes around GLOBL and DATA
ranges are pinned.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 21b6348dbe test(parser): cover self-includes, deep chains and diamonds
The include guard's scope is now pinned from three sides: a file
including itself is a cycle diagnostic, a chain of two thousand distinct
headers completes with the deepest content spliced, and a header reached
again through a separate branch splices a second time and is refused as
a redefinition.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 32c453d66e test(format): corpus invariants over GOROOT assembly
Three property tests walk every .s file under the installed GOROOT plus
the repository's kernels: formatting is idempotent and preserves the
token stream, a clean parse keeps its tree through a format pass, and a
cleanly expanding file expands to the same statements afterwards.  The
walks skip under -short so the push suite keeps its budget.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin 59637d35b1 test(format): fuzz the expansion round-trip
A second formatter target holds the assembly path's contract: beyond the
token view, a file the expander reads cleanly must expand to the same
statement sequence after formatting, because the macro language draws
distinctions from layout (the adjacency of a #define name and its '(',
continuation bodies) that a token count cannot see.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin e600fc00e7 fix(format): preserve macro adjacency and continuation bodies
The canonical spelling of a #define line glued a '(' to the macro name
whatever the input's spacing, turning an object macro whose body opens
with a parenthesis into a parameterised one, and it flattened the
backslash continuations of a multi-line body into one physical line,
fusing the statements the expansion splits at those boundaries.  Both
change what a valid file assembles to, so renderPreproc now keeps the
name's adjacency (the same column check the preprocessor applies) and
restores the continuation boundaries from the token positions.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00
petrbalvin acd30088af fix(asm): reject the operand-starved riscv64 spellings instead of panicking
Assisted-by: GLM 5.3
2026-10-07 13:53:37 +02:00
petrbalvin d03de62c07 feat(arch): add the amd64 fp16 packed imm8-control group
Assisted-by: GLM 5.3
2026-10-07 13:51:48 +02:00
petrbalvin 8de1b371da feat(arch): add the amd64 fp16 packed conversion family
Assisted-by: GLM 5.3
2026-10-07 13:51:48 +02:00
petrbalvin fb6d01a7d0 feat(arch): encode the amd64 embedded rounding and SAE decorations
Assisted-by: GLM 5.3
2026-10-07 13:51:48 +02:00
petrbalvin e56c04e9ee fix(asm): read three operands from the arm64 last-element form
Assisted-by: GLM 5.3
2026-10-07 13:51:10 +02:00
petrbalvin 104bea036b feat(asm): encode the arm64 SVE gather loads and scatter stores
Assisted-by: GLM 5.3
2026-10-07 13:51:10 +02:00
petrbalvin 6faf850793 feat(asm): encode the arm64 SVE2 crypto, counter and reduction families
Assisted-by: GLM 5.3
2026-10-07 13:51:10 +02:00
petrbalvin dac0a5b51b docs(asm): state the seed layout of the arch fuzz targets exactly
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin 965b33e5b0 test(asm): seed the riscv64 and loong64 assemble fuzz targets
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin b9dfb48d79 fix(asm): encode the loong64 64-bit-span 2RI14 offsets like the toolchain
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin 41aa8edfd2 fix(asm): match the toolchain's loong64 logical immediate expansion
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin 7f955f8bd1 fix(cmd/gasm): probe the loong64 sc.q operand order in the encodability battery
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin 8fdc511d0d test(asm): add the loong64 error-parity catalogue
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin e5035d92e9 fix(asm): reject the offset on the loong64 register-indexed memory form
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin b54547c08d fix(asm): reject the shifted-register compositions on loong64
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin e9510e8a68 test(asm): pin the riscv64 tail against the toolchain byte for byte
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin 9d50212a71 fix(asm): refuse the riscv64 width moves across register banks
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin f2892e4f59 feat(asm): emit the riscv64 local-exec TLS sequence
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin f8dbd4f017 fix(asm): lower the riscv64 immediate CSR pseudos onto their opcode forms
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin e5aabf9801 fix(asm): encode the riscv64 FENCE predecessor and successor flags
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin 36d1ac804a fix(asm): route the riscv64 register moves through the toolchain forms
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin da35883633 fix(asm): compress the riscv64 two-operand arithmetic and immediate tail
Assisted-by: GLM 5.3
2026-10-07 13:51:02 +02:00
petrbalvin f0318d2c99 test(disasm): pin the vector spaces the decoder refuses
The RISC-V vector extension and the LoongArch LSX/LASX families carry
no x/arch decode tables, and the loong64 WORD directive is no
instruction: all three stay on the placeholder.  The pins record
today's refusal with the corpus rows that assemble the same bytes, so
a decoder bump that learns one of these spaces flips a row here and
asks for the naming pass to cover it.

Assisted-by: GLM 5.3
2026-10-07 13:50:54 +02:00
petrbalvin a7f9d5eb67 feat(disasm): render the loong64 families x/arch names Unknown
x/arch's Plan 9 renderer decodes thirty-odd scalar loong64 operations
perfectly but prints them as "Unknown OP args", and names the
sign-extension pair EXT.W.B/EXT.W.H "?".  The supplementary naming
pass re-renders them with the toolchain's own spellings: the families
whose operand order x/arch already prints the Plan 9 way trade only
the mnemonic, and the pointer loads and stores, the acquire loads,
the release stores, PRELD and ALSL are rebuilt from the decoded
arguments with the toolchain's operand order and its raw displacement
reading.  ADDU16I.D, a macro helper the assembler never takes as
input, keeps the decoder's Unknown render.

The loong64 parity fixture grows from 73 to 385 rows, pinning every
unique four-byte corpus word the decoder accepts, and the LL/SC
displacement divergence in the encoder is documented for asm.

Assisted-by: GLM 5.3
2026-10-07 13:50:54 +02:00
petrbalvin 0f6e008c9a feat(disasm): name the arm64 words arm64asm refuses
arm64asm rejects three exception-space words the toolchain's corpus
assembles: the hypervisor call HVC, the secure monitor call SMC and
the speculation barrier SB.  The supplementary naming table reads
the raw word in the error branch and renders the corpus spellings;
the corpus rows are pinned in the parity fixture and the boundary
test holds the unallocated neighbours on the placeholder.

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00
petrbalvin 3f35b2a718 fix(disasm): drop the implicit GOARCH constraint the naming files carried
The names naming_amd64.go and naming_amd64_test.go carried the _amd64
filename suffix, which the go tool reads as an implicit GOARCH=amd64
build constraint: the tables disappeared from every non-amd64 build
and the package failed to compile for arm64, riscv64 and loong64, the
other three architectures the tool assembles.  Renaming to
amd64_naming.go removes the constraint; the module builds again for
all four GOARCH values.

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00
petrbalvin 2c70359ad0 feat(disasm): name the amd64 encodings x86asm refuses
The toolchain's assembler corpus carries 195 amd64 encodings the
x/arch decoder rejects or degenerates: the BMI1/BMI2 VEX families
(ANDN, BEXTR, BLSI, BLSMSK, BLSR, BZHI, MULX, PDEP, PEXT, RORX,
SARX, SHLX, SHRX), the 0F 01 quartet CLAC, STAC, RDPKRU and WRPKRU,
the bare and REX-only RDSEED forms, and UD1.  The supplementary
naming table decodes the VEX prefix and the ModR/M shape and renders
the toolchain's own spellings; every corpus row is pinned in the
unlisted fixture and round-trips byte for byte through the encoder,
and the boundary test pins the prefix shapes no family carries.

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00
petrbalvin d98aadbbbf test(disasm): drop the closed byte-width divergences from the parity map
The operand-width reconciliation closes the byte-register fixture lines
byte for byte: XADDL, XCHGL, CMPXCHGL and CRC32 with byte registers, the
ALU and TEST immediates against AL and DL, and the unlisted accumulator
short forms.  MOVL $0x7, DL stays mapped for the legal-encoding choice
alone: the toolchain's own table says "c6c207 or b207", go tool asm emits
b207, and the fixed-point invariant holds with it.

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00
petrbalvin 8bded3ea39 test(asm): pin the byte-form width reconciliation and its oracle
Table-driven rows for the renderer's spellings (the L suffix or none with
a byte register encodes the byte form, every register joining at its low
byte), the refusals (W, Q and the MOVD alias take no byte register) and a
differential kernel of the B-suffixed spellings assembled through both
gasm and go tool asm, byte for byte.

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00
petrbalvin 98a562d8b3 fix(asm): settle the byte-form width from the register operands
The suffixed scalar families derived the operand width from the mnemonic
alone, so a byte-spelled register under the L spelling or no suffix at all
encoded the widened form: XADDL DL, DL emitted 0F C1 where the byte form is
0F C0, CMPL AL, $7 emitted the 32-bit immediate form where the AL form is
3C 07, and CRC32 DL, R11 widened past the F0 byte opcode.  operandWidth now
reconciles the suffix with the operands: a byte register (AL, DL, R8B, ...)
forces the 8-bit form, which is the text the toolchain's own disassembly
prints for those encodings, while the W and Q spellings never ride a byte
register and are refused as go tool asm refuses them (MOVQ AL, AX).  The
shift count and the two- and three-operand IMUL forms stay out of the
reconciliation, and the byte accumulator short forms now belong to the AL
spelling alone, matching the toolchain's division (ADDB $3, AX is
80 c0 03, TESTB $7, AX is f6 c0 07).

Assisted-by: GLM 5.3
2026-10-07 13:49:58 +02:00
petrbalvin 33e7fdac98 fix(asm): enforce the arm64 TLBI, RPRFM, FCVT and integer-pair arities
Assisted-by: GLM 5.3
2026-10-07 13:49:49 +02:00
petrbalvin 7a69be8b59 fix(asm): reject the arm64 REGTMP spellings the toolchain refuses
Assisted-by: GLM 5.3
2026-10-07 13:49:49 +02:00
petrbalvin 2385bb7069 fix(asm): enforce the arm64 VLD/VST post-index contract
Assisted-by: GLM 5.3
2026-10-07 13:49:49 +02:00
petrbalvin 7bc80ccb54 fix(asm): emit nothing for the arm64 NOP pseudo-instruction
Assisted-by: GLM 5.3
2026-10-07 13:49:49 +02:00
petrbalvin 405e2427ed fix(asm): place the arm64 literal pool the way the toolchain flushes it
Assisted-by: GLM 5.3
2026-10-07 13:49:49 +02:00
petrbalvin 4fc96decc4 fix(asm): size the arm64 logical-immediate materialisation exactly
Assisted-by: GLM 5.3
2026-10-07 13:49:49 +02:00
petrbalvin 4ccd3bb4c6 test(debug): put the dead-debuggee and kill audits on deterministic ground
The launch-failure audit raced the clock: it asserted the dead debuggee
surfaced within 1.5 seconds, a bound the loaded machine behind a ten-way
test storm regularly starved past even though the poll detects the dead
notice within milliseconds of its appearance.  The audit now proves the
property itself: a stub debuggee that starts in single-digit milliseconds
marks itself dead, so the notice is always inside the poll's budget and
the error must come from the dead-file watch, while the real binary is
checked without any wall-clock bound.  A companion audit drives Kill
through a parked, a doubly killed and a run-to-exit session, the states
whose cleanup used to hang the package, under a watchdog.

Assisted-by: GLM 5.3
2026-10-07 13:46:29 +02:00
petrbalvin b5d6f1b46a fix(debug): target the traced thread and keep the kill from ever blocking
The Go runtime can migrate the debuggee's target-mode goroutine off the
process leader before PTRACE_TRACEME, which left the trace relation on a
thread the session never addressed: its stops starved the waits on the
leader, and a kill sequence that resumed nothing and then blocked in
Wait4 hung the whole package.  The debuggee now reports the traced thread
in the launch handshake and parks with a thread-directed stop, every
ptrace request and wait addresses that thread, a SIGURG arriving on a
single-step resumes it as a single-step again instead of letting the
tracee run uncontrolled, a resume rejected with ESRCH lifts a group-stop
with SIGCONT and retries once, and Kill resumes, kills and reaps through
non-blocking waits so it returns for a tracee in any state.

Assisted-by: GLM 5.3
2026-10-07 13:46:18 +02:00
petrbalvin 409c8b348d fix(asm): bound the arm64 VTBL table list before the destination read
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 2b2a72d54e fix(asm): encode the arm64 bitfield aliases with their own opc
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 8ab99c9b0c fix(asm): tighten the arm64 acceptance toward the toolchain
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 82d741514a fix(asm): key the arm64 immediate class order on the ZR spelling
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin f20aa156d0 fix(asm): treat the arm64 $-8 frame as frameless and encode the RET forms
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin d853432dba fix(asm): route the arm64 logical immediates to ZR through REGTMP
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin fbdad8424f feat(asm): encode the arm64 FP immediate moves
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin a9b54b6868 fix(asm): carry the arm64 immediate to ZR through MOVZ
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin d786b90fa1 feat(asm): lower the arm64 con(register) form to the ADD chain
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 9ef14bdb71 feat(asm): encode the arm64 SIMD arrangement bits
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 458cdd2066 feat(asm): encode the arm64 register-offset addressing forms
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin a344399b81 fix(asm): set the CASALH opcode bit fifteen
The CASALH entry carried the CASB/CASH opcode pattern where the acquire
forms take the full fixed field, so the word differed from the
toolchain's in one opcode bit.  Pinned against the oracle word.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin a0fa7e802c feat(asm): pool the arm64 offsets the split bands cannot carry
Offsets beyond the split bands ride a per-function literal pool the way
the toolchain lays one out: a PC-relative literal load into REGTMP, then
the register-offset access (the pair family adds the base addition), the
pooled words appended after the last instruction behind the UNDEF guard,
deduplicated by value with the sign- and width-aware load selection.

The same differential pass against the corpus exposed three wrong-code
bugs and fixes them: the logical-immediate period marker rode the wrong
position for every element below 64 bits, so the 32-bit forms encoded a
different constant than written; the plain register operand of an
ADD/SUB against SP took the shifted-register form where the toolchain
uses the extended one with the identity extend, silently truncating
through UXTB; and the AUTIA1716 and AUTIB1716 hint constants were the
PACIA and PACIB encodings.  An offset sweep across every band boundary
now pins all three against the live oracle.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 71e8dd550d feat(asm): split wide arm64 load and store offsets into REGTMP
Offsets the single-instruction forms cannot carry lower the way the
toolchain lowers them: ADD or SUB moves the whole distance into REGTMP
within the ±4095 band, and the 24-bit band above it splits into an ADD of
the high half and an access of the low half, with the pair family taking
the two-ADD sequence.  The split band follows loadStoreClass per width,
byte accesses taking the full 24 bits and the Q width the widest, so an
offset the toolchain pools is never split instead.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 2bd52eb7ad fix(asm): scale the FLDPQ and FSTPQ pair offsets by sixteen
The pair encoder derived the imm7 divisor from the width suffix alone, so
the 128-bit FP pairs divided their offsets by eight and encoded twice the
distance.  The Q spellings scale by sixteen like every other 128-bit
access; the differential kernel carries them now.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin daf7fad5b9 feat(asm): encode the arm64 Q-width FP load and store
FMOVQ routes through the MOV load/store machinery in the plain, post-index,
pre-index and static-symbol forms.  The Q width carries its size in the opc
field, so the store spelling is opc=10 and the access scales by sixteen;
both come from helpers now instead of the size exponent.  The static-symbol
form takes the toolchain's twelve-byte ADRP + ADD + access fallback with the
R_ADDRARM64 pair.  The register-to-register and immediate forms stay
rejected, matching the toolchain's own table.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin a6bd9c1ebe feat(asm): encode the LDEORAL acquire variants
Assisted-by: GLM 5.3 Flash
2026-10-07 02:36:24 +02:00
petrbalvin 458d981f31 feat(disasm): name the CLDEMOTE encoding the decoder refuses
The hint NOP opcode 0F 1C /r with a memory operand is CLDEMOTE, a
memory-only instruction the toolchain's own table carries; the decoder
rejects the encoding instead of naming it.  The rejected-encoding side
of the supplementary table names it from the bytes, and the corpus row
0f1c03 pins the text in the unlisted fixture, round trip byte exact.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:27:51 +02:00
petrbalvin dd356b9e6a fix(disasm): name the amd64 families x/arch decodes to the zero opcode
x/arch reports the ADCX, ADOX, RDSEED, RDPID, TPAUSE, UMONITOR, UMWAIT
and ENDBR families with no error but the degenerate zero instruction,
which GoSyntax renders as Op(0) under its prefix decoration and with a
length of one.  A supplementary naming table keyed by the opcode
pattern restores the toolchain's own spellings and lengths; the parity
fixtures pin all 41 corpus rows (ENDBR32 alone, which the toolchain
cannot spell, pins as bytes and text in the focused naming test).

Assisted-by: GLM 5.3 Flash
2026-10-07 02:27:51 +02:00
petrbalvin 6c4932c4ec feat(arch): the SVE2.1 Z-alias permutations and copies
Assisted-by: GLM 5.3 Flash
2026-10-07 02:27:35 +02:00
petrbalvin 8231302bca feat(arch): the SVE predicate family in the extended layer
Assisted-by: GLM 5.3 Flash
2026-10-07 02:26:19 +02:00
petrbalvin 83052ab466 feat(lint): accept the wired extension mnemonics
Assisted-by: GLM 5.3 Flash
2026-10-07 02:23:37 +02:00
petrbalvin 9c951c232e feat(asm): assemble the extended instruction layer on arm64
Assisted-by: GLM 5.3 Flash
2026-10-07 02:23:37 +02:00
petrbalvin c2adde948f feat(arch): add the write mask to the packed amd64 destinations
Assisted-by: GLM 5.3 Flash
2026-10-07 02:21:57 +02:00
petrbalvin 631fb8a8d7 chore: keep the fuzz cache and pending reproductions out of the tree
Assisted-by: GLM 5.3 Flash
2026-10-07 02:14:37 +02:00
petrbalvin 470639cd68 test(asm): carry the fuzz pipeline to arm64
Assisted-by: GLM 5.3 Flash
2026-10-07 02:14:37 +02:00
petrbalvin e7a1c20467 fix(asm): prefix MOVQ2DQ with F3
The two bank-crossing quadword moves take the mandatory prefix by
direction: the toolchain renders F3 0F D6 as MOVQ2DQ with the MMX
source and F2 0F D6 as MOVDQ2Q with the XMM source, and the encoder
emitted F2 for both, so MOVQ2DQ encoded MOVDQ2Q.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:12:21 +02:00
petrbalvin 0cfed5516c fix(asm): carry the riscv64 U-type immediate raw
The toolchain writes the source immediate straight into imm[31:12]
(riscv64.s: AUIPC 24287, X10 encodes 7ffff517), and rejects values
beyond the signed 20-bit span; the encoder divided by 4096 instead and
truncated silently, so the high bits of every large AUIPC and LUI were
lost.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:12:21 +02:00
petrbalvin a829b8f321 fix(parser): read the segment-absolute rendering FS:0
The toolchain's disassembler prints the segment-prefixed disp32
absolute as FS:0, but the bare-name branch read the segment register
alone and dropped the offset, so MOVQ FS:0, DX silently encoded a
register move.  The colon-offset form now lowers to the same
segment-absolute operand the 0(FS) spelling takes.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:12:21 +02:00
petrbalvin abc2d32b83 fix(asm): encode the CMOV condition the renderer prints
The renderer spells a conditional move CMOV plus the condition alone
(CMOVLE, CMOVG), the width carried by the operand registers, so
CMOVLE parsed as the size L and the condition E and encoded CMOVE.
A suffix that is itself a condition name now reads as that condition
with the width from the destination register, and the Plan 9
size-prefixed spellings keep their parse.

Assisted-by: GLM 5.3 Flash
2026-10-07 02:12:21 +02:00
petrbalvin 11cac26508 feat(arch): add the scaled index to the amd64 memory operands
Assisted-by: GLM 5.3 Flash
2026-10-07 02:06:11 +02:00
petrbalvin ccb155437e feat(arch): add the {1toN} broadcast to the packed amd64 memory sources
Assisted-by: GLM 5.3 Flash
2026-10-07 02:06:11 +02:00
petrbalvin 9c1392dfc3 test(disasm): pin the amd64 lines the objdump listing fragments
Assisted-by: GLM 5.3 Flash
2026-10-07 01:57:04 +02:00
petrbalvin 8d611bfdaf feat(arch): add the remaining scalar FP16 memory forms to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 01:35:48 +02:00
petrbalvin 454a21f5b7 feat(arch): add the packed FP16 and BF16 memory forms to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 01:35:48 +02:00
petrbalvin d275dee3ae feat(arch): add the scalar FP16 memory forms to the amd64 extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 01:35:48 +02:00
petrbalvin 7d69dda874 feat(arch): add the amd64 memory-operand mechanism to the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 01:35:48 +02:00
petrbalvin d1030a1788 test(verify): prove the arm64 lse atomics through the link-parity kernel
Assisted-by: GLM 5.3 Flash
2026-10-07 01:31:28 +02:00
petrbalvin 8e3b7f1caa test(verify): prove the amd64 sse families through the link-parity kernel
Assisted-by: GLM 5.3 Flash
2026-10-07 01:31:28 +02:00
petrbalvin b22abf512f test(verify): prove the riscv64 vector families through the link-parity kernel
Assisted-by: GLM 5.3 Flash
2026-10-07 01:31:28 +02:00
petrbalvin ffc23929c6 refactor(verify): extract the shared goobj link-parity harness
Assisted-by: GLM 5.3 Flash
2026-10-07 01:31:28 +02:00
petrbalvin a4a77b3168 build(justfile): lower the fuzz worker default to the fence the assembler target holds
Assisted-by: GLM 5.3 Flash
2026-10-07 01:15:42 +02:00
petrbalvin 380314a05f test(asm): keep the data-section ceiling out of the seed corpus
Assisted-by: GLM 5.3 Flash
2026-10-07 01:14:58 +02:00
petrbalvin de2f28504a test(asm): parse once and assemble twice in the fuzz body
Assisted-by: GLM 5.3 Flash
2026-10-07 01:14:58 +02:00
petrbalvin 7878112b93 fix(asm): bound the data section to what the image can materialise
Assisted-by: GLM 5.3 Flash
2026-10-07 01:14:58 +02:00
petrbalvin dd074dbafd fix(asm): reject an out-of-range GLOBL size with a diagnostic
Assisted-by: GLM 5.3 Flash
2026-10-07 01:13:40 +02:00
petrbalvin 14de1dd287 test(asm): fuzz the parse-and-assemble pipeline for amd64
Assisted-by: GLM 5.3 Flash
2026-10-07 01:12:27 +02:00
petrbalvin ea8b02b190 test(disasm): pin decode parity with the toolchain and the round-trip invariant
Assisted-by: GLM 5.3 Flash
2026-10-07 01:00:45 +02:00
petrbalvin b394551af8 fix(disasm): render amd64 in the toolchain's Go syntax
Assisted-by: GLM 5.3 Flash
2026-10-07 01:00:45 +02:00
petrbalvin 9b751f6e58 feat(asm): encode the riscv64 quad-precision family and fix the fp cvt paths
Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin 955bc6643e fix(asm): carry the riscv64 scalar pseudos and the swapped branches
Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin 82e8919208 feat(asm): encode the riscv64 privileged instructions
Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin 778c297214 feat(asm): encode the riscv64 vector arithmetic families
Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin 5de9e985f8 feat(asm): encode the riscv64 vector load and store families
The vector memory section stopped at the three hand-written shapes the
GOROOT kernels use: every other spelling the toolchain accepts, the
width variants, the constant-stride and indexed accesses, the segment
families, the fault-only-first loads, the whole-register moves and the
bit-mask pair were names without an encoder.  The mnemonic now parses
into its own fields (direction, segment count, addressing mode, width,
fault-only-first and whole-register markers) and one encoder lays the
word down, with the optional V0 mask operand and the toolchain's
operand shapes.  VSETVL joins the configuration settings.  The
toolchain's whole vector memory section, six hundred and twenty-eight
statements of masked and unmasked forms, is a differential test against
the oracle, word for word.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin e43fa39dc9 feat(asm): encode the explicit riscv64 compressed instructions
The C extension's own spellings were names the table carried and the
encoder refused: CLWSP stopped the corpus audit's riscv64 file first.
Thirty-eight mnemonics now encode directly to their halfword, with the
toolchain's operand spellings and validation: the stack loads and stores
pin their base to SP, the register-based loads, stores and arithmetic
carry prime registers, CLUI refuses zero and SP, CADDI4SPN scales by
four, CADDI16SP by sixteen, and CJ, CBEQZ and CBNEZ resolve their N(PC)
targets against the final layout, taking a two-byte placeholder in the
early passes so the offsets stay honest.  CAND with an immediate is the
toolchain's C.ANDI spelling.  The toolchain's whole C extension testdata
block is a differential test, halfword for halfword, beside a range test
at the toolchain's own boundaries.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin ddfa33ccc1 feat(asm): encode the riscv64 bit-manipulation families
The Zba address generation, Zbb unary bit operations, Zbc carry-less
multiplication and Zbs single-bit families were names the table carried
and the encoder refused: thirty spellings plus RORI and XNOR fell over.
The register and immediate forms now encode as the toolchain does, the
unary operations carry their fixed rs2 constant, RORI lowers to ROR's
expansion (its reverse shift compressing like ROR's), XNOR XORs and
inverts in place, and ROL/ROLW rotate left through the same temporary
the toolchain uses, taking a register amount only as its own expansion
requires.  The toolchain's whole testdata block for these families is
now a differential test: every word must agree byte for byte.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin d82ef33fa9 fix(asm): bound the riscv64 shift immediate at the instruction width
SLLI $64 assembled with the amount silently masked into the six-bit
field where the toolchain rejects it, and the word forms took 0-63 where
they take 0-31.  Both families now validate against their own width and
the check reads the immediate at full width, so a value the source
spelled beyond int32 cannot wrap into the range; the boundary is pinned
in a test.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin c3540f0549 fix(asm): read the riscv64 raw-data immediates at full width
WORD and BYTE read their immediate through the truncating helper, so the
int32 wrap turned WORD $0xffffffff into -1 and rejected it, while WORD
$0x100000000 and BYTE $0x100000001 arrived pre-truncated and slipped
past the range check as small values.  Both statements now read the
immediate the source wrote and bound it at the toolchain's own limits:
[0, 0xffffffff] for WORD, [0, 0xff] for BYTE, with the bounds pinned in
a test.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin bffe408afa feat(asm): resolve every CSR name the toolchain knows
The riscv64 assembler carried fourteen hand-picked CSR names where the
toolchain resolves three hundred and twenty-nine: a CSRR/CSRW family
instruction naming any privileged register beyond the few base ones came
out as unknown CSR.  The table now carries the RISC-V privileged
specification's register set exactly as go tool asm spells it, and a
differential test assembles every name through both assemblers and
requires the words to agree byte for byte.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:47:27 +02:00
petrbalvin 103864e8b2 feat(arch): add the VL packed FP16 forms to the amd64 extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 00:37:41 +02:00
petrbalvin aa9c7ca030 feat(arch): add the imm8 scalar FP16 controls to the amd64 extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 00:33:30 +02:00
petrbalvin 0354a1f4c1 feat(arch): scale and exponent-extract the scalar FP16 in the extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 00:07:58 +02:00
petrbalvin 127e52de69 test(verify): accumulate the BF16 dot product on the metal
Assisted-by: GLM 5.3 Flash
2026-10-07 00:07:58 +02:00
petrbalvin 22055b9bf3 feat(arch): add the packed FP16 arithmetic to the amd64 extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 00:07:58 +02:00
petrbalvin a2301b52de test(verify): run the amd64 extension encodings on the metal
Assisted-by: GLM 5.3 Flash
2026-10-07 00:07:58 +02:00
petrbalvin bc37ea5b79 feat(arch): add the AVX512-FP16 scalar family to the amd64 extension layer
Assisted-by: GLM 5.3 Flash
2026-10-07 00:07:58 +02:00
petrbalvin 28bea95128 feat(arch): add the amd64 extended-instruction layer with BF16 and VP2INTERSECT
Assisted-by: GLM 5.3 Flash
2026-10-07 00:07:58 +02:00
petrbalvin 2d803e38d8 feat(lsp): leave the extension verdicts to the registry
Diagnostics no longer repeat the generated table's ignorance of a
registered mnemonic: unknown-instruction and unencodable-instruction
against a statement the registry encodes are filtered from the server's
own presentation, driven by asm.LookupExtension directly.  The filter is
a no-op once lint learns the registry, so the two compose unchanged.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:06:05 +02:00
petrbalvin dadeda144a feat(lsp): offer and document the registered extended mnemonics
Completion merges the extension registry's mnemonics beside the toolchain
entries, with operand shapes measured against the layer's own encoder, and
hover documents a registered mnemonic from its metadata: the extension
notice, the forms, the features, the fixed encodings and the manual
references.

Assisted-by: GLM 5.3 Flash
2026-10-07 00:06:05 +02:00
petrbalvin a90ec84bee fix(arch): narrow file names by go/build's suffix rule
Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin c107b45933 fix(asm): reject duplicate symbol declarations like the toolchain
Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin a69f8cf4a8 fix(asm): match the toolchain's bytes across the corpus sweep
A line-for-line byte comparison of the whole amd64enc.s corpus against
go tool asm surfaced divergences the pass-only accounting never showed:
PEXTRW's GPR form swapped its fields, PUSHW took an imm32 where the
toolchain bounds the immediate to 16 bits, the double shift wrote the
unmasked register number into the reg field, VCOMISS carried a 0x66
prefix, RORX dropped the destination's R bit, and the variable bit
shifts used the manual's per-width opcodes where the toolchain
consolidates each row on one opcode with the W bit.  The VEX forms the
toolchain prefers for plain vector registers (the SSE2/SSSE3/SSE4.1
AVX twins, the compare-with-predicate family, VMOVUPS, VSHUFPS, the
variable shifts) now encode under VEX, with EVEX left to the ZMM,
opmask and index-16+ spellings, and the mnemonics whose rows never
offer the 2-byte prefix force it.  Every line is pinned through the new
corpus parity test (793 lines); the whole corpus file now assembles to
the toolchain's bytes at every commented line (10022 of 10022).

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin cfc3abb752 feat(asm): encode the remaining amd64 VEX families
The SSE3 horizontal and add-subtract pairs, the SSSE3 sign and
horizontal integers, the masked moves in both directions, the
reciprocity and test pairs, the AVX imm8 tail (blends, dot products,
inserts, rounds, MPSADBW, the string compares), the four-operand
variable blends with their /is4 mask byte, the scalar three-operand
moves, the MXCSR accessors, the VPERMIL register controls and the
variable word shifts, plus the BMI2 count forms over memory.  Every
encoding is pinned byte for byte against go tool asm through every
corpus line the toolchain's own amd64enc.s carries for the families
(852 lines); the /is4 byte carries the mask register number in its high
nibble, the layout the toolchain emits.

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin 6af3fd60d5 feat(asm): encode the legacy amd64 SSE and MMX families
The packed integer and float binaries, the imm8-controlled SSE4.1 forms,
the variable blends with their X0 mask, the high/low half moves, the
sign-mask extractions, the non-temporal stores, the MOVQ bank crossings
and their odd spellings, the MMX shifts and shuffle and the cache-line
mask stores, plus the scalar leaves LEAVE, INVPCID and the RTM controls.
Every encoding is pinned byte for byte against go tool asm through every
corpus line the toolchain's own amd64enc.s carries for the families
(1465 lines).  Two corpus-wide gaps fell out of the comparison: the
64-bit MOV immediate uses the zero-extending form across the unsigned
32-bit span, and the MMX-to-GPR MOVQ puts the bank register in reg.

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin 257feace6e feat(asm): encode the amd64 XSAVE family
XSAVE, XSAVEOPT, XSAVEC and XSAVES with their restore twins, plain and
64, each pinned byte for byte against go tool asm through every corpus
line the toolchain's own amd64enc.s carries for the family (24 lines).
The toolchain emits XSAVEOPT without the manual's 0x66 prefix; the bytes
are the oracle, so the family carries none.

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin f83bc8ddef feat(asm): encode the amd64 system, string and segment families
The no-operand flag and system controls, the sign-extension pair, the
string primitives, the multi-byte no-ops, the cache controls, MOVBE, the
compare-exchange doubles, the random source and FS/GS base pairs, the
descriptor-table accesses, the 0F 00/01 register controls and the
LAR/LSL selector reads and far-segment loads, each pinned byte for byte
against go tool asm through every corpus line the toolchain's own
amd64enc.s carries for the families (279 lines).

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin f0b08ccea0 feat(asm): encode the amd64 x87 family
The x87 stack controls, the D8/DC arithmetic pair, the conditional moves,
the register compares, FADDDP, the memory loads and the FXSAVE pair, each
pinned byte for byte against go tool asm through every corpus line the
toolchain's own amd64enc.s carries for the family (78 lines).

Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:47 +02:00
petrbalvin 1040739fbb fix(asm): validate the loong64 ll/sc offset span like the toolchain
Assisted-by: GLM 5.3 Flash
2026-10-06 23:59:27 +02:00
petrbalvin 6cc6165c2b build(justfile): cap the fuzz recipe's workers
Assisted-by: GLM 5.3 Flash
2026-10-06 21:13:31 +02:00
petrbalvin 0768b4dc84 fix(cmd): locate GOROOT when the build is trimmed
Assisted-by: GLM 5.3 Flash
2026-10-06 20:24:39 +02:00
petrbalvin ae1baa0e61 ci: run the affordable gate set on push and publish at the tag
Test / test (push) Successful in 1m45s
Assisted-by: DeepSeek V4.1 Flash
2026-10-04 21:17:40 +02:00
petrbalvin 5a8e9acbf3 feat(arch): add the extended-instruction layer with SVE arithmetic
Test / test (push) Successful in 3m38s
2026-10-02 20:39:33 +02:00
petrbalvin 2747fce7d3 feat(asm): encode the arm64 system registers and structure loads 2026-10-02 20:39:26 +02:00
petrbalvin e02918c17b fix(ci): keep the push suite inside the runner's memory and time budget
Test / test (push) Successful in 2m55s
Assisted-by: GLM 5.3 Flash
2026-10-02 17:20:31 +02:00
petrbalvin 02a6359c1f ci: shrink the push pipeline to the affordable gate set
Test / test (push) Failing after 5m26s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:47:52 +02:00
petrbalvin b4c1e133c0 ci: keep the GOOBJ link parity gate off the push pipeline
Test / test (push) Failing after 12m18s
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin 1107928870 build(justfile): run the test recipes under the memory fence
Assisted-by: GLM 5.3 Flash
2026-10-02 16:15:02 +02:00
petrbalvin bc4ac93fd9 style(asm): reindent the evex comment gofmt asks for
Test / test (push) Failing after 21m29s
Assisted-by: GLM 5.3
2026-10-02 00:41:46 +02:00
petrbalvin c1bca7ce7e docs: record the development deltas in the changelog
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin fefb76beb9 docs(asm): correct the reference against the assemblers' behaviour
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 42bc1669d7 feat(lsp): document directives on hover and widen completion
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 69dcbec8ef feat(lint): eleven new rules over directives, data and addressing
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f405cea5bc fix(cmd): stop the coverage run on a stray in-place trap
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin f57377abb9 fix(debug): handle mapping edges, stray traps and dying debuggees
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin dd1782c538 test(verify): gate GOOBJ link parity with cmd/link
Assisted-by: GLM 5.3
2026-10-02 00:40:54 +02:00
petrbalvin 17cc49fee4 fix(asm): emit NOPTR data as its own symbol kind
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin bafb2fd130 feat(asm): encode the amd64 and loong64 tails of the corpus testdata
Assisted-by: GLM 5.3
2026-10-02 00:40:43 +02:00
petrbalvin 2f679326c2 style(testdata): canonicalise forms_amd64.s
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 96f2dd65b4 test(lexer): fuzz the token stream invariants
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin ca887d3927 fix(parser): bound folding depth and macro expansion work
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 6570709226 fix(parser): peel stacked labels the way the formatter renders them
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 9a5217d9c1 fix(format): keep every token of a line in the canonical output
Assisted-by: GLM 5.3
2026-10-02 00:40:20 +02:00
petrbalvin 4d01bb3ecf build: rename the module to sourcedock.dev/petrbalvin/gasm-sdk
Test / test (push) Successful in 4m18s
2026-09-26 11:08:43 +02:00
petrbalvin 332c63e440 ci: compile-gate FreeBSD in the test pipeline
Test / test (push) Successful in 4m10s
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:49 +02:00
petrbalvin b306c210c6 feat(debug): port the debugger to FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:40 +02:00
petrbalvin b9015e1c2e fix(verify): make the executable mapping build on FreeBSD
Assisted-by: GLM 5.3 Flash
2026-09-25 21:46:31 +02:00
petrbalvin 26c5008136 fix(cmd): honour //go:build in the corpus audit
Test / test (push) Successful in 2m32s
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:16 +02:00
petrbalvin 74d6b90d69 fix(asm): read the arm64 move-wide immediate as an unsigned pattern
Assisted-by: GLM 5.3 Flash
2026-09-23 21:03:03 +02:00
petrbalvin 7b11c62f53 fix(asm): resolve negative numeric PC-relative jumps
Assisted-by: GLM 5.3 Flash
2026-09-23 21:02:50 +02:00
petrbalvin 8eed54b3da feat(lsp): quick fixes for the textflag include and the argument area
Test / test (push) Successful in 2m56s
Assisted-by: GLM 5.3 Flash
2026-09-23 20:23:35 +02:00
petrbalvin 4be16dcdf5 feat(lsp): resolve symbols across workspace files
Assisted-by: GLM 5.3 Flash
2026-09-23 20:21:58 +02:00
petrbalvin ded9cabdf4 fix(lsp): apply each rename edit to its own document
Assisted-by: GLM 5.3 Flash
2026-09-23 20:19:21 +02:00
295 changed files with 54959 additions and 2321 deletions

No files matched your search

-37
View File
@@ -1,37 +0,0 @@
# Race, Go. Dispatched by hand, and never a gate on a push or a tag: the release tag is
# cut only after `just gates` has already raced the tree, so this workflow is the
# explicit second opinion, not a step of the release.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is a decision rather than a routine.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
race:
runs-on: fedora
timeout-minutes: 20
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install gcc
# The race detector needs cgo and the runner image carries no C compiler.
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 10m ./...
+26 -116
View File
@@ -1,20 +1,28 @@
# Release, Go binaries. Runs on version tags (v1.2.3) pushed to main.
#
# The module sits at the repository root: the toolchain records a version only for a root
# module, measured on go1.27.1, so a build of a module in a subdirectory reports (devel)
# even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never pass for
# it. A Go repository is one module at the root.
# No gate runs at the tag: the tagged tree was tested on every push to development,
# the full suite is the suite workflow's business, and race never runs in CI at all.
# This pipeline publishes and nothing else. The version contract has no injection
# step: the toolchain records the tag into the binary's build information, so the
# build simply has to happen at the tag, which the trigger guarantees.
#
# The version contract these steps implement: nothing is injected. The toolchain records
# the tag into the binary's build information, so the build simply has to happen at the
# tag, which the trigger guarantees.
# The module sits at the repository root: the toolchain records a version only for a
# root module, measured on go1.27.1, so a build of a module in a subdirectory reports
# (devel) even at its own <module>/vX.Y.Z tag and this workflow's smoke test can never
# pass for it. A Go repository is one module at the root.
#
# The gates run in their own job, once, before the matrix, minus the race detector: race
# never runs on a push path or a tag, and the local gate raced this tree before the tag
# was cut. Putting the gates inside the matrix would run the whole suite once per target
# on the box that also hosts the forge. Each job validates the tag for itself rather than
# passing a value between jobs, so no workflow feature has to be trusted for the version
# to reach the file name.
# The matrix carries the platforms the project ships: Linux on amd64, arm64,
# loong64 and riscv64, and FreeBSD on amd64 and arm64. Go cross-compiles
# FreeBSD natively and the tree builds under CGO_ENABLED=0, which the
# suite's cross-build gates prove for the whole module; the ptrace suite
# itself still needs a FreeBSD machine, so nothing FreeBSD runs here.
# Nothing is installed: the fedora job image carries git, perl and node
# (verified on the runner, 2026-10-04).
#
# Each job validates the tag for itself rather than passing a value between jobs, so
# no workflow feature has to be trusted for the version to reach the file name. Every
# step is one command, and the scripted steps are Perl with builtins only: Perl drives
# curl through a list, so no argument is ever word-split, globbed or quoted wrong.
name: Release
on:
@@ -22,111 +30,16 @@ on:
tags: ["v*"]
env:
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job
# needs it most; the build jobs inherit it for their parallel compilation.
# The box is shared with the forge, so parallelism is bounded on purpose.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
gates:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Install Perl
# Perl for the steps below. The install is a no-op where the package
# is already present.
run: dnf install -y perl
- name: Validate the tag
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
$v =~ m{^v[0-9]+(\.[0-9]+){0,2}([-+].*)?$}
or die qq{ERROR: expected a semver tag like v1.2.3, got: $v\n};
print qq{tag $v\n};
'
- name: Security policy names this release
# The supported-versions table is the one part of SECURITY.md that
# carries a version, so it goes stale the moment a tag is cut. Fail
# here rather than publish a policy naming the previous release.
env:
VERSION: ${{ gitea.ref_name }}
run: |
perl -e '
my $v = $ENV{VERSION} // q{};
(my $nv = $v) =~ s/^v//;
open(my $f, q{<}, q{SECURITY.md}) or die qq{SECURITY.md: $!\n};
local $/;
my $t = <$f>;
close $f;
$t =~ m{^\|\s*\Q$nv\E\s*\|\s*yes\s*\|}m
or die qq{ERROR: SECURITY.md does not name $nv as supported; update the table before releasing.\n};
print qq{SECURITY.md names $nv\n};
'
- name: Build
run: go build ./...
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
run: go fix -diff ./...
- name: Tests
# The same command as in test.yml, so the floor is the same number everywhere.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The same command as in test.yml: the CLI's exit codes and manual-page guard,
# and the debugger's architecture-neutral units, run outside the floor.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
build:
runs-on: fedora
timeout-minutes: 25
needs: gates
strategy:
fail-fast: false
matrix:
# Portable targets: amd64, arm64, loong64 and riscv64 on Linux, at the toolchain
# default level. No 32-bit, no wasm, no macOS, no Windows. FreeBSD stays out until
# verify/jit.go ports off syscall.Mprotect: the Go syscall package defines no
# Mprotect for freebsd, and verify/jit.go:50 calls it to drop the write bit from
# the JIT mapping, so every freebsd target fails to build with "undefined:
# syscall.Mprotect" (verified for amd64, arm64 and riscv64 on go1.27.1).
include:
- goos: linux
goarch: amd64
@@ -136,6 +49,10 @@ jobs:
goarch: loong64
- goos: linux
goarch: riscv64
- goos: freebsd
goarch: amd64
- goos: freebsd
goarch: arm64
steps:
- uses: actions/checkout@v7
@@ -144,9 +61,6 @@ jobs:
go-version-file: go.mod
cache: true
- name: Install Perl
run: dnf install -y perl
- name: Validate the tag
id: version
env:
@@ -209,7 +123,6 @@ jobs:
release:
runs-on: fedora
timeout-minutes: 15
needs: build
permissions:
# contents: read is required for the checkout: a job that declares any
@@ -226,9 +139,6 @@ jobs:
with:
path: dist
- name: Install Perl
run: dnf install -y perl
- name: Extract the CHANGELOG section
env:
VERSION: ${{ gitea.ref_name }}
+86
View File
@@ -0,0 +1,86 @@
# Suite, Go. Dispatched by hand, on development. Never a push gate.
#
# The complete gate set minus race: the build, the FreeBSD cross-builds the
# port's compile proof needs (amd64 and arm64, the same two the release
# matrix ships), both static gates, the full suite with every short-layer
# skip unskipped, and the coverage floor. It is the pipeline form of the
# local `just test`, for the moments when the tree must be proven end to end
# and nobody is at the keyboard.
#
# Race never runs in CI. It roughly doubles the time and the memory on a box shared
# with the forge, and the local `just gates` races the tree on the machine at the
# keyboard, which is where that gate belongs.
name: Suite
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
suite:
runs-on: fedora
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
cache: true
- name: Build
run: go build ./...
- name: FreeBSD build (amd64)
# The port is pure Go: the cross-build needs no C toolchain and the
# release matrix ships the same two binaries.
run: CGO_ENABLED=0 GOOS=freebsd GOARCH=amd64 go build ./...
- name: FreeBSD build (arm64)
run: CGO_ENABLED=0 GOOS=freebsd GOARCH=arm64 go build ./...
- name: Format
run: |
perl -e '
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
my @bad = <$g>;
close($g);
print @bad;
exit(@bad ? 1 : 0);
'
- name: Vet
run: go vet ./...
- name: Modernise
# Exits non-zero when it has something to rewrite, so it needs no output capture.
run: go fix -diff ./...
- name: Tests
# The full suite over the logic packages (`packages` in the justfile), every
# short-layer skip lifted, with the coverage profile. No timeout: a run that
# does not finish is a defect to find.
run: go test -count=1 -timeout 0 -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The same second invocation as the local gate: the CLI's exit codes and
# manual-page guard, and the debugger's units, live ptrace sessions included.
# They run without a profile, because a thin main and a ptrace-bound package
# would drag the floor down rather than measure the product.
run: go test -count=1 -timeout 0 ./cmd/... ./debug/...
- name: Coverage floor
run: |
perl -e '
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
my $total;
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 80 ? 1 : 0);
'
+24 -46
View File
@@ -1,17 +1,20 @@
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so it lives in race.yml.
# The box is one core and 2 GB beside Gitea, so parallelism is bounded on purpose and
# everything runs in one job. Extra jobs would duplicate the checkout, the Go setup and
# the dependency download three times without buying any parallelism.
# The push path owns a two-minute budget end to end, so it carries exactly the gates
# that fit it: the format check, go vet, the suite's short layer and the coverage
# floor. There is no build step (go test compiles what it runs) and the modernisation
# gate (`go fix -diff`) belongs to the suite workflow, which is dispatched by hand.
# Nothing is installed: the fedora job image carries git, perl and node (verified on
# the runner, 2026-10-04), and no pipeline ever installs gcc or runs the race detector.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
# not Python: Perl behaves the same on both runner images, there is no bashism to trip over
# on ash, and it is one language instead of two. The Perl uses builtins only, because
# Fedora packages the Perl modules separately and nothing beyond `perl` itself may be
# assumed present.
# The live ptrace sessions and the other deliberate-run categories skip under
# testing.Short: they need the machine to themselves and belong to the suite workflow
# and to the local `just test`, never to every push.
#
# Every step is one command, so the step that fails is the gate that failed, and no
# shell option has to be trusted for the run to stop. The scripted steps are Perl with
# builtins only: Fedora packages the Perl modules separately, so nothing beyond `perl`
# itself may be assumed present, and there is no bashism to trip over.
name: Test
on:
@@ -36,7 +39,6 @@ concurrency:
jobs:
test:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
@@ -46,16 +48,6 @@ jobs:
go-version-file: go.mod
cache: true
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
# no-op where it is already present; drop this step once verified on the box.
run: dnf install -y perl
# The steps follow the `gates` order of the justfile contract: build, format,
# vet, test. The vet gate is go vet and go fix -diff, two steps here.
- name: Build
run: go build ./...
- name: Format
run: |
perl -e '
@@ -69,33 +61,19 @@ jobs:
- name: Vet
run: go vet ./...
- name: Modernise
# Exits non-zero when it has something to rewrite, so it needs no output capture.
run: go fix -diff ./...
- name: Tests
# The suite must be fast: a push pipeline that cannot finish in a few minutes moves
# its heavy part behind a dispatch. The inner timeout matches the job's, so a
# hanging test reports its own goroutine dump rather than a silent job kill.
# The pattern is `packages` in the project's justfile: the logic packages, since a
# thin cmd/ would drag the total under the floor. release.yml runs the same
# command, so the floor is the same number everywhere. ./verify/... carries the
# live oracle-parity comparison against `go tool asm` (the TestGroundTruth
# suites); the runner's Go setup provides both the tool and GOROOT.
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
# The pattern is `packages` in the project's justfile, so the floor is the
# same number the local gate reports: the logic packages, since a thin cmd/
# and a ptrace-bound debug package would drag the total under it. No timeout
# anywhere: `-timeout 0` disables go test's own ten-minute default, because a
# run that does not finish is a defect to find and a timeout only hides it.
run: go test -short -count=1 -timeout 0 -coverprofile=coverage.out ./arch/... ./asm/... ./ast/... ./disasm/... ./format/... ./lexer/... ./lint/... ./lsp/... ./parser/... ./token/... ./verify/...
- name: Tests outside the coverage set
# The CLI and the debugger sit outside `packages` because a thin main and a
# ptrace-bound package pull the total under the floor, but their tests guard
# shipped surfaces: the command exit codes, the manual pages against the
# binary's own help, and the debugger's architecture-neutral units. They run
# here so the floor stays a product measure and nothing is left untested.
run: go test -count=1 -timeout 10m ./cmd/... ./debug/...
- name: Oracle parity
# Re-run the live go-tool-asm comparison as its own step so that a parity
# regression names the gate that failed instead of hiding inside the suite.
run: go test -count=1 -timeout 10m -run 'TestGroundTruth' ./verify/...
# The CLI's exit codes and manual-page guard and the debugger's
# architecture-neutral units, still tested but outside the profile the floor
# is computed from. `-short` skips the debugger's live ptrace sessions.
run: go test -short -count=1 -timeout 0 ./cmd/... ./debug/...
- name: Coverage floor
run: |
+5
View File
@@ -7,6 +7,11 @@
coverage.out
*.test
# The fuzz campaigns' cache: committed seeds live in the test file, crashes are
# minimised into it, so everything go test -fuzz writes under testdata/fuzz is
# transient. pending-* holds minimised reproductions awaiting their owner's fix.
/asm/testdata/fuzz/
# Crash dumps from the emulator runs
core
core.*
+255 -2
View File
@@ -1,15 +1,268 @@
# Changelog
All notable changes to gasm-devkit are documented here.
All notable changes to gasm-sdk are documented here.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
## [development]
## [0.36.0] - 2026-10-07
### Added
-
- **The SVE2 and SVE2.1 instruction sets on arm64.** Over 350 SVE
mnemonics across 537 encoded forms join the extension layer: the
narrowing two-to-one arithmetic family, the SVE2 cryptographic set
including ZADCLB, BFloat16 arithmetic, predicate counters and
reductions, the pairwise and quadword forms, multiple-structure loads
and stores (LD2/ST2 through LD4/ST4), shift by vector and the
shift-immediate scheme with its element-size encoding, CLASTA and
CLASTB, the last-active and compare predicate families, and the
vector-length arithmetic ADDVL, ADDPL and RDVL. Every form derives
from the toolchain's own encoder tables and is pinned against the
GOROOT corpus words.
- **The complete AVX512-FP16 set and AVX-VNNI-INT16 on amd64.** The
extension layer reaches 191 entries: the packed and scalar FMA
families (VFMADD, VFMSUB, VFMADDSUB and VFMSUBADD in the PH widths
with their SH mirrors), the complex multiply and complex FMA pairs
(VFMULC, VFCMULC, VFMADDC, VFCMADDC, packed and scalar, each
conjugating the source its prefix names), VMINMAXPH and VMINMAXSH
under their imm8 control, and the AVX-VNNI-INT16 dot products
(VPDPWSUD, VPDPWSUDS, VPDPWUSD, VPDPWUSDS) through a new VEX
encoding path beside the EVEX one. Every entry carries golden
vectors byte-checked against GNU as.
- **Extension-layer instructions assemble straight from `.s` source on
amd64.** The registered mnemonics dispatch from the assembly front
end with their operand grammar: `k0` through `k7` write masks with
merging and zeroing, `{1toN}` broadcast, embedded rounding and
`{sae}`, memory and scaled-index operands, and the imm8-control
forms, the VEX forms beside the EVEX ones. `gasm lint` surfaces the
layer's refusals as `extension-form` errors instead of letting a bad
shape reach the encoder, and every distinct mnemonic is pinned from
source text against the registry's bytes.
- **arm64 output byte-identical with the toolchain across the GOROOT
tree.** 31 of the 36 files the differential harness walks (90
functions, 27408 bytes) now assemble without a differing byte: the
missing shapes are in (GETCALLERPC, the REM, REMW, UREM and UREMW
family, DWORD, the FCVTHS, FCVTSH, FCVTDH and FCVTHD conversions, TLS
local-exec loads as a single MOVZ with the TLS relocation), the
literal pool drains mid-function when a function's literals would
overflow the ±512 KiB load-literal range, and frame-relative
addresses and flag-setting logicals materialise the way
cmd/internal/obj/arm64 writes them.
- **riscv64 END and GETCALLERPC.** END is accepted anywhere and emits
nothing, and GETCALLERPC reads the return address the toolchain's
rewrite describes: a leaf function reads LR, a framed body reads the
prologue's save slot, with the compressed spellings where they fit.
- **The loong64 register-pair spelling.** `MULV R4:R5, R6` parses the
way the toolchain's pair sugar does: the colon splits the operand and
the halves swap, any instruction shape takes the spelling, and the
malformed forms fail with the toolchain's own wording.
- **Fuzz targets across the tool.** The assembler carries
per-architecture FuzzAssemble targets and the front end carries
targets over the lexer, parser, formatter and the extension
registry; the seed corpora run as ordinary tests, and `just fuzz`
drives the campaigns locally. Hardening the targets found caps GLOBL
and DATA at 64 MiB per symbol and 128 MiB per section, so a crafted
input cannot materialise unbounded memory.
- **The FreeBSD port of the debugger.** `gasm debug` runs on FreeBSD on
amd64 and arm64 with the same interactive surface as on Linux:
breakpoints, hardware watchpoints (x86 debug registers, the arm64 debug
register file), single-stepping, register and memory access, all behind
the kernel's own ptrace requests, with tracee memory through `PT_IO` and
stop reports through `PT_LWPINFO`. The JIT substrate maps executable
memory through `golang.org/x/sys/unix`, so `verify` builds on FreeBSD
too. The dispatched suite compile-gates both architectures, the
release pipeline ships the two binaries, and live validation awaits a
FreeBSD machine.
- **Workspace-wide navigation in the language server.** `gasm lsp` indexes
the `.s` files under the workspace root beyond the documents the editor
has open, so go-to-definition, find references and workspace symbol search
reach files that were never opened. An open buffer always shadows its
disk copy, and watched-file events together with a per-query freshness
check keep the index current.
- **Quick fixes for the textflag include and the argument area.** The
`missing-textflag-include` warning offers to add the include after the
last one in the file, and the `abi-argsize` warning offers to set the
TEXT argument area to the size the `// func` signature implies, computed
by the new `lint.ExpectedArgSize`.
- **Eleven new lint rules over the directives, the data section and the
sharpest addressing edges.** `missing-argsize` flags a TEXT that
declares no argument area its `// func` signature implies;
`noframe-frame-size` a NOFRAME with a positive frame;
`unnamed-fp-reference` a nameless `0(FP)`, which both assemblers
reject; `hardware-sp-addressing` a negative offset off the hardware SP
rather than the virtual frame; `vex-sse-mixing` a kernel that mixes VEX
and legacy SSE spellings and pays the transition penalty;
`unnamed-result` a `ret+N(FP)` the signature names; `data-width`,
`data-value-overflow`, `data-string-width`, `data-without-globl` and
`data-exceeds-globl` police the DATA width against its value type and
the GLOBL size behind it. `missing-ret` now also flags a function
whose tail can fall off its end even though a RET sits somewhere in the
body, and `invalid-textflag` also reports a flag misplaced between TEXT
and GLOBL.
- **Hover documentation and wider completions in the language server.**
Hovering a directive or pseudo-operation (TEXT, DATA, GLOBL, PCALIGN,
FUNCDATA, PCDATA, the BYTE family) shows its grammar and rules in
preference to the empty instruction-table entry, and completion offers
those names beside the instruction set. The `missing-argsize` warning
carries a quick fix that declares the argument area the signature
implies (`$0` becomes `$0-16`).
- **The GOOBJ link parity gate.** A regression test assembles a kernel
per architecture through `gasm asm --format goobj`, substitutes the
object into a real `go build`'s package archive, proves the archive
carries it byte for byte and re-links with cmd/link on amd64, arm64,
riscv64 and loong64. The binaries run, natively on amd64 and under
qemu-user on arm64 and riscv64, and their output must match the
toolchain-built baseline; loong64 is link-only. The pipeline installs
qemu-user and runs the gate on every push.
- **The amd64 and loong64 encoders close four more corpus files.** The
whole-tree measure moves to 272 of 322 (84.5 %): amd64 gains the
one-operand IMUL, the SSE compare family (CMPPD, CMPPS, CMPSS), RETFL,
the LOOP family, the MMX register bank with its bank-crossing moves,
MOVNTDQ, the CR and DR register moves, PUSH and POP of FS and GS, the
`(TLS)` pseudo-base, the wait and cache controls (CLWB, CLDEMOTE,
TPAUSE, UMONITOR, UMWAIT, RDPID, ENDBR64), indirect branches with the
star spelling (`JMP *(R12)(R13*4)`), `RET sym(SB)` as the tail jump,
the colon shift spelling (`SHLL CX, R11:AX`) and the EVEX and VEX forms
of the rounds, AES key assist, string compares, extracts, blends and
permutes; loong64 gains the acquire and release pair (LLACQ, SCREL,
with the vector widths), the VMOVQ and XVMOVQ lane forms and the BYTE
literal-data escape hatch the other architectures already take.
### Changed
- **TEXT flag operands behave like the toolchain's.** A numeric or
parenthesised flag list (`TEXT ·f(SB), 4, $4096-0`,
`(NOSPLIT|NOFRAME)`) now suppresses the prologue and stack guard the
names suppress, where the numbers were silently ignored and the
prologue was emitted anyway; unknown flag names are rejected with the
toolchain's wording instead of passing unnoticed; and the toolchain's
own TEXT diagnostics fire, `ABIInternal requires NOSPLIT` and
`NOFRAME functions must have a frame size of 0`.
- **The disassembler names what it used to print raw.** The 195 amd64
encodings that decoded to anonymous renders now decode to their
instructions and round-trip, the BMI1 and BMI2 VEX set, RDSEED, RDPID,
the WAITPKG family, ENDBR, CLDEMOTE, UD1 and RORX among them, with the
arm64 HVC, SMC and SB forms and 74 loong64 rendering corrections
beside.
- **Variadic macro parameters are rejected like the toolchain rejects
them.** A macro declaration whose parameter list the toolchain
refuses no longer parses into a broken definition.
- **The corpus audit assembles like the build.** A file's `//go:build`
constraint decides which target architectures attempt it: cpu_x86.s is
an x86 build alone, and the msan and goexperiment.runtimesecret trees
are compiled by no supported build, so they leave the measured set
instead of failing it. The headline now reads "assemble for every
applicable target": every real-code GOROOT assembly file, the tree
without testdata, assembles for all four architectures (250 of 250,
100 %); over the whole tree including testdata the measure is 272 of
322 (84.5 %).
- **The module moves to `sourcedock.dev/petrbalvin/gasm-sdk`.** The
repository and the module rename together with the product, now the
GAsm Software Development Kit. Fresh installs become
`go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest`, and
installs pinned to the old `gasm-sdk` path stop resolving once the
repository takes the new name: reinstall from the new path. The
binary stays `gasm`.
### Fixed
- **riscv64 memory offsets beyond the 12-bit immediate no longer
truncate silently.** `LD 4096(X6), X5` encoded as `LD X5, 0(X6)` and
read the wrong address; the expansion the toolchain performs (the
upper bits into its temporary register, then the access) is emitted
now, byte-identically, and offsets the toolchain refuses are refused
with its wording.
- **riscv64 operand acceptance matches the toolchain's.** 398 operand
shapes the toolchain rejects assembled without complaint (a float
register where an integer one is required, out-of-range CSRs, vector
shapes against the wrong bank, MOV family widths), and 488 more
produced the wrong diagnostic; all now fail with the toolchain's
texts, and a 1030-case error-parity catalogue pins them against the
toolchain's own negative files.
- **The debugger survives its failure paths.** Killing a debuggee that
had already died or sat ptrace-stopped could block the session
forever; Kill now wakes the tracee, signals and reaps it without
waiting on a parked child, so every launch-failure path returns
instead of hanging the tool.
- **Encoder defects found by fuzzing and the differential audits.** A
GLOBL with a negative size panicked the assembler and a small crafted
input materialised 3.92 GiB of DATA; certain SVE VTBL operands
panicked; an over-strict guard refused BIC into RSP where the
toolchain assembles it; byte-register operands now select the 8-bit
forms the toolchain picks (`XADDL DL, DL` and its companions); and
the loong64 shift compositions and indexed-offset forms that dropped
fields re-encode faithfully.
- **Formatter edges around labels and macros.** A stacked label that
names a macro stays attached to it, macro content behind a label
stays on its line, a selector-folded label resolves to its macro, an
empty TEXT body formats, and a file ending in a comment no longer
loses the comment.
- **Rename edits land in their own documents.** A rename collected the
ranges of every reference across the open documents but applied them all
to the document that started it, so renaming a symbol used in a second
file moved that file's text into the first. Each edit now applies to the
document it was collected in.
- **Negative numeric PC-relative jumps.** `JMP -3(PC)`, the shape the
runtime's exit loops write (sys_linux_amd64.s, sys_netbsd_amd64.s),
resolved to nothing: only the forward forms counted. A negative count
now walks the same instruction statements backwards, labels excluded,
byte-identical with the toolchain.
- **The arm64 move-wide family reads its immediate as an unsigned
pattern.** `MOVK $(40000<<48)` folds to a negative int64 and was
rejected; the toolchain picks the 16-bit lane from the 64-bit bit
pattern, so the encoder now does the same, and a zero immediate is
rejected where the toolchain rejects it.
- **NOPTR data emits its own symbol kind.** A GLOBL with NOPTR was
emitted as plain SDATA, the kind the linker holds to its Go type
information requirement, so every gasm object carrying runtime-shaped
data (`GLOBL ·x(SB), NOPTR, ...`) died in cmd/link with "missing Go
type information". NOPTR data is now SNOPTRDATA, the toolchain's kind
for pointer-free globals, and RODATA still wins where both flags
appear, exactly as the toolchain chooses. The three end-to-end link
tests that should have caught this substituted the gasm object after
the package archive was already packed, so they passed vacuously; they
now substitute inside the archive, prove the substitution byte for
byte and re-link.
- **Latent encoder divergences against the toolchain, found by the
whole-file differential harness.** On loong64, `MOVx $off(reg)` lost
its base register, SCQ swapped its operand fields, the vector lane
inserts did not scale offsets by the element width and two immediate
opcodes were mistyped; on amd64, NOP with operands encoded 0x90 where
the toolchain emits nothing at all, the VEX gather length bit ignored
the VSIB index width, four permute and extract families were pushed
into EVEX where the toolchain stays VEX, and a reference to a static
symbol no GLOBL defines failed where the toolchain defers it to the
linker as an external relocation.
- **Seven front-end defects found by fuzzing.** The formatter swallowed
the statement after a leading block comment (`/* head */ MOVQ AX, BX`
formatted to the comment alone), dropped trailing tokens after an
`#include` header, and trimmed the trailing whitespace inside a string
literal; the parser peeled only one label of a stacked pair (`a: b:`
parsed differently than it formatted); deeply nested parentheses in an
immediate overflowed the stack through the constant folder, which now
stops at a bounded depth and falls back to the ordinary operand paths;
macro expansion is linear in the invocation count instead of
quadratic; and amplifying macros (the billion-laughs shape) stop at a
work budget sized by the line and the macro table, reported as a
diagnostic instead of running for hours.
- **The debugger handles the edges its tests now reach.** Memory reads
and writes cover exactly the requested bytes, so a request ending in
the last page of a mapping no longer fails on the unmapped page behind
it; disassembly shrinks its instruction window at a mapping's end
instead of failing; a debuggee that dies before signalling readiness
writes a failure notice the launcher reads, so launch fails fast with
the reason instead of after the whole poll (and the poll no longer
waits on the child, which could consume the SIGSTOP park and hang the
launch); amd64 watchpoints acknowledge the sticky DR6 hit bits and
clear the address register on release; the breakpoint listing is
ordered by address so its numbering is stable; the REPL rejects bad
counts, sizes and watchpoint types instead of silently guessing,
reports stops on signals during stepping, and a breakpoint-class trap
that matches no breakpoint and leaves the PC in place surfaces instead
of spinning the continue loop and the coverage run forever.
## [0.35.0] - 2026-09-22
+10 -9
View File
@@ -1,6 +1,6 @@
# Contributing
Contributions to **gasm-devkit** are governed by the Contributor terms
Contributions to **gasm-sdk** are governed by the Contributor terms
below; submitting one means you accept them.
## Contributor terms
@@ -29,8 +29,8 @@ compiler (gcc), because `just gates` includes `just race` and the race
detector needs cgo.
```sh
git clone https://sourcedock.dev/petrbalvin/gasm-devkit.git
cd gasm-devkit
git clone https://sourcedock.dev/petrbalvin/gasm-sdk.git
cd gasm-sdk
just build
just gates
```
@@ -117,16 +117,17 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
| Workflow | Trigger | What it does |
|---|---|---|
| Test | push or pull request to `development` | build, format check, vet, modernisation, the test suite with the coverage floor, the CLI and debugger tests outside the profile, then the oracle-parity rerun against `go tool asm` |
| Release | a `v*` tag | the same gates as Test minus the oracle-parity step, then the matrix build, the version smoke test and the release itself; the race detector runs locally in `just gates` before the tag is cut |
| Test | push or pull request to `development` | the format check, `go vet`, and the short layer of the test suite with the coverage profile and the 80 % floor, inside the two-minute budget |
| Suite | dispatched by hand | the complete gate set minus race: the build, the two FreeBSD cross-build gates (amd64, arm64), both static gates, the full suite and the coverage floor |
| Release | a `v*` tag | the matrix build (Linux on four architectures, FreeBSD on two), the version smoke test and the release with its assets; no gate runs at the tag |
The local equivalent is `just gates`, which is the same set plus the race detector. The
race detector also has its own workflow, dispatched by hand; it never runs on a push or a
tag, where it would double the time and the memory a shared runner cannot spare.
The local equivalent is `just gates`, which is the same set plus the race detector. Race
never runs in CI, on a push or a tag: it would double the time and the memory a shared
runner cannot spare, so the local `just gates` runs it before the tag is cut.
## Reporting bugs
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-devkit/issues` with the
Open an issue at `https://sourcedock.dev/petrbalvin/gasm-sdk/issues` with the
version, the operating system and architecture, the exact command, the full output,
and the expected against the actual behaviour.
+30 -21
View File
@@ -1,6 +1,6 @@
# Plan 9 assembly tooling, inside and outside Go
# GAsm: Software Development Kit for Plan 9 Assembly
> **Warning: this is an experiment.** gasm-devkit is under active
> **Warning: this is an experiment.** gasm-sdk is under active
> development and is not stable. The version is 0.x.x: commands, flags,
> output formats and behaviour can change without warning at any time.
> A 1.0.0 release is light years away. Nothing in this document is a
@@ -13,7 +13,7 @@
there is no formatter, no linter and no debugger for `.s` files, and no
assembler that works without a Go installation. Developers write
assembly blind, validate it by benchmark, and debug it by print
statement. gasm-devkit is the missing toolkit: a single, self-contained
statement. gasm-sdk is the missing toolkit: a single, self-contained
binary, `gasm`, that serves both purposes.
- **Help develop Plan 9 assembly.** Formatting, linting, disassembly,
@@ -58,7 +58,7 @@ Plan 9 (Go): MOVQ AX, total-16(SP)
The same lines, but only one of them tells you what the number is for.
The syntax is uppercase, regular and boring, which is the highest
compliment a language for machine code can earn. gasm-devkit exists
compliment a language for machine code can earn. gasm-sdk exists
to give that syntax the tooling it deserves.
## Features
@@ -94,13 +94,16 @@ to give that syntax the tooling it deserves.
- **Debugger.** `gasm debug` is a source-level ptrace debugger with
breakpoints (optionally conditional), hardware watchpoints, register and
memory inspection, and headless script runs that report instruction and
label coverage.
label coverage; it runs on Linux (all four architectures) and FreeBSD
(amd64, arm64).
- **Language server.** `gasm lsp` serves completion, hover, document symbols,
push and pull diagnostics, semantic-token highlighting, go-to-definition,
find references, rename, formatting, inlay hints, code actions, signature
help, document highlights, workspace symbol search, #include document
links and folding ranges over stdio; definition, references and rename
work across every open document.
work across every open document and the indexed workspace files beyond
them, and the quick fixes add a missing textflag.h include and set the
argument area from the // func signature.
- **Comparators and audits.** `gasm diff` compares the machine code of two
assembly files byte-for-byte, `gasm profile` shows basic-block structure,
`gasm audit-instructions` diffs the encoder against the installed toolchain,
@@ -127,10 +130,13 @@ can emit today is narrower, and a recognised but unencodable instruction is
reported as an explicit error, never as a wrong byte.
The same measurement runs over GOROOT's whole assembly corpus:
`gasm audit-instructions --corpus` reports 291 of 353 attemptable files
(82.4 %) assembling for every target architecture today (files named for
other Go ports are counted but never attempted), with the top failure
reasons per architecture; the number moves with every release.
`gasm audit-instructions --corpus` reports every real-code GOROOT assembly
file (the tree without testdata) assembling for every target its build
admits: 250 of 250, 100 %. Over the whole tree including testdata the
measure is 271 of 322 attemptable (84.2 %); files named for other Go ports
are counted but never attempted, and `//go:build` constraints decide which
targets attempt a file at all, exactly as the build does. The number moves
with every release.
### Validation status
@@ -145,7 +151,7 @@ actually been executed.
|---|---|---|
| Encoding: byte-for-byte against `go tool asm` | native hardware | native hardware (the toolchain cross-assembles any GOARCH on any host) |
| Execution: JIT calls, ABI checks, differential fuzzing | native hardware | qemu-user emulation |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more |
| Debugger: ptrace tracing, breakpoints, watchpoints, coverage | native hardware | emulation cannot run ptrace; the layer compiles and its architecture-neutral units run under `go test ./...`, nothing more. FreeBSD (amd64, arm64) is in the same position: the port compiles behind the cross-build gate and its integration test is ready, but no FreeBSD machine has executed it |
Consequences, stated plainly. An emulator is a model of a CPU, not the
CPU: instruction semantics are implemented in software and can differ
@@ -156,9 +162,10 @@ other. Encoding parity is the exception: the byte comparison against the
toolchain runs on the host for every architecture, so no emulator stands
between the claim and the evidence. The debugger is the weakest case: on
the three emulated architectures its per-architecture ptrace code has
been compiled and read, never executed. Its architecture-neutral units
run under `go test ./...`, which the race workflow and a manual run
perform; the default `just test` gate does not sweep `./debug/...`.
been compiled and read, never executed. Its units run under
`go test ./...`, which `just test` sweeps in a second invocation beside
the coverage profile; the live ptrace sessions are skipped in short
mode, so they run locally or in the dispatched suite, never on a push.
## The documentation goal
@@ -215,21 +222,23 @@ The plan, in the order it is being worked:
toolchain itself does not support; through ELF, Plan 9 assembly becomes
usable outside Go entirely.
- **Platforms: Linux and FreeBSD.** Linux is supported today on all four
architectures and is where the binary builds. FreeBSD follows: the
JIT's executable-memory mapping and the ptrace debugger layer are the
two pieces of porting work. Other unix systems may follow those two.
architectures and is where the binary builds. FreeBSD follows on amd64
and arm64: the JIT's executable-memory mapping and the ptrace debugger
layer are ported, the release carries the two FreeBSD binaries, and the
debugger's live validation awaits a FreeBSD machine (the validation
status states it). Other unix systems may follow those two.
- **Four architectures, no more.** amd64, arm64, riscv64 and loong64.
No others are planned.
## Install
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64 and
linux/loong64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-devkit/releases).
Prebuilt binaries for linux/amd64, linux/arm64, linux/riscv64,
linux/loong64, freebsd/amd64 and freebsd/arm64 are on the
[releases page](https://sourcedock.dev/petrbalvin/gasm-sdk/releases).
From source (Go 1.27.1):
```sh
go install sourcedock.dev/petrbalvin/gasm-devkit/cmd/gasm@latest
go install sourcedock.dev/petrbalvin/gasm-sdk/cmd/gasm@latest
```
Or from a repository checkout:
+7 -7
View File
@@ -5,7 +5,7 @@
// toolchain's own assembler source. Go's Plan 9 assembler defines the exact,
// complete set of mnemonics it accepts for each architecture in
// $GOROOT/src/cmd/internal/obj/<arch>/anames.go; this tool extracts those
// names so gasm-devkit supports every instruction the real assembler does,
// names so gasm-sdk supports every instruction the real assembler does,
// with no hand-maintained (and therefore inevitably incomplete) lists.
//
// The same data feeds the generated instruction appendices of the assembly
@@ -31,11 +31,11 @@ import (
"sort"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/arch"
"sourcedock.dev/petrbalvin/gasm-devkit/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
)
// archDirs maps a gasm-devkit architecture name to its obj sub-directory.
// archDirs maps a gasm-sdk architecture name to its obj sub-directory.
var archDirs = []struct {
arch string
sub string
@@ -119,7 +119,7 @@ func filterCommon(names []string) []string {
// writeCommon emits arch/common_gen.go.
func writeCommon(names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/util.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -190,7 +190,7 @@ func stringLit(elt ast.Expr) string {
// writeGen emits arch/<arch>_gen.go.
func writeGen(arch, sub string, names []string) error {
var b strings.Builder
b.WriteString("// Code generated by gasm-devkit _gen; DO NOT EDIT.\n")
b.WriteString("// Code generated by gasm-sdk _gen; DO NOT EDIT.\n")
b.WriteString("// Source: cmd/internal/obj/" + sub + "/anames.go from the Go toolchain.\n")
b.WriteString("//\n")
b.WriteString("// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)\n")
@@ -216,7 +216,7 @@ func writeDocPage(a arch.Arch, title, file, anames, version string, encodable bo
var b strings.Builder
b.WriteString("# " + title + ": instruction inventory\n\n")
b.WriteString("Generated by gasm-devkit's `_gen` from the Go toolchain's instruction table\n")
b.WriteString("Generated by gasm-sdk's `_gen` from the Go toolchain's instruction table\n")
b.WriteString("(`" + anames + "`, " + version + "); DO NOT EDIT. This page lists every mnemonic\n")
b.WriteString("`go tool asm` accepts on this target, which is the upper bound of the\n")
b.WriteString("language on it: a name absent here is not an instruction of the target,\n")
+1835
View File
File diff suppressed because it is too large. Load diff
+201
View File
@@ -0,0 +1,201 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package arch
import (
"encoding/hex"
"strings"
"testing"
)
// The memory-operand mechanism: the base-plus-displacement validation and
// the ModR/M, SIB and displacement byte choices. The expected words are
// built on the VMOVSH memory-load template, whose rows the binutils-gdb
// assembler testsuite quotes byte for byte; the rows marked GNU match the
// listing bytes. Note on the quoted disp8 rows: binutils mainline encodes
// EVEX displacements with the APX disp8*N scaling, so its source spellings
// (254 for the m16 rows, 8128 for the m512 ones) are N times the plain SDM
// displacement the disp8 bytes carry; the words here pin the bytes with the
// plain displacement those bytes encode.
func TestAmd64ExtMemoryEncoding(t *testing.T) {
load := []byte{0x62, 0x05, 0x06, 0x00, 0x10, 0xC0}
for _, tt := range []struct {
name string
base int
disp int64
want string
gnuSource string // the binutils source line the bytes serve, empty for a derived row
}{
{"zero displacement drops the disp bytes", 9, 0,
"62457e081031", "vmovsh (%r9),%xmm30"},
{"positive disp8", 1, 127,
"62657e0810717f", "vmovsh 254(%rcx),%xmm30 (Disp8(7f) under the disp8*N scaling)"},
{"negative disp8", 2, -128,
"62657e08107280", "vmovsh -256(%rdx),%xmm30 (Disp8(80) under the disp8*N scaling)"},
{"disp32 past the disp8 range", 2, 8128,
"62657e0810b2c01f0000", ""},
{"negative disp32", 13, -200,
"62457e0810b538ffffff", ""},
{"RSP base takes the SIB byte", 12, 0,
"62457e08103424", ""},
{"RSP base with a disp8", 12, 4,
"62457e0810742404", ""},
{"RBP base keeps a zero displacement explicit", 5, 0,
"62657e08107500", ""},
} {
got := amd64EncodeMemory(load, 30, -1, tt.base, tt.disp)
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestAmd64ExtMemoryRejects checks the validation around the memory operand:
// the kinds and ranges the layer refuses before a byte is laid down.
func TestAmd64ExtMemoryRejects(t *testing.T) {
in := ExtInstr{Name: "TEST"}
for _, tt := range []struct {
name string
op ExtOperand
quote string
}{
{"a register where the memory operand belongs", ExtXmm(3),
"wants a memory operand"},
{"base beyond r15", ExtMemory(16, 0),
"outside 0-15"},
{"base under r0", ExtMemory(-1, 0),
"outside 0-15"},
{"displacement past the signed 32-bit ceiling", ExtMemory(8, 1<<31),
"outside the signed 32-bit range"},
{"displacement past the signed 32-bit floor", ExtMemory(8, -1<<31-1),
"outside the signed 32-bit range"},
{"shift on the memory operand", ExtOperand{Kind: ExtMem, Reg: 8, Imm: 4, Shift: 2, HasShift: true},
"take none"},
{"arrangement suffix on the memory operand", ExtOperand{Kind: ExtMem, Reg: 8, Arr: ExtArrH},
"arrangement"},
{"predicate qualifier on the memory operand", ExtOperand{Kind: ExtMem, Reg: 8, Qual: ExtQualZeroing},
"predicate qualifier"},
{"broadcast on an entry that takes none", ExtBroadcast(8, 0),
"takes none"},
} {
_, _, err := in.amd64Memory(tt.op, 1)
if err == nil {
t.Errorf("%s: validation succeeded, want an error", tt.name)
continue
}
if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
// The broadcast spelling passes the flag gate on an entry that carries
// Bcast and then meets the same base and displacement checks.
bcast := ExtInstr{Name: "TEST", Bcast: true}
for _, tt := range []struct {
name string
op ExtOperand
quote string
}{
{"broadcast base beyond r15", ExtOperand{Kind: ExtMem, Reg: 16, Imm: 0, Broadcast: true},
"outside 0-15"},
{"broadcast displacement past the signed 32-bit ceiling", ExtBroadcast(8, 1<<31),
"outside the signed 32-bit range"},
{"broadcast base under r0", ExtBroadcast(-1, 0),
"outside 0-15"},
} {
if _, _, err := bcast.amd64Memory(tt.op, 1); err == nil {
t.Errorf("%s: validation succeeded, want an error", tt.name)
} else if !strings.Contains(err.Error(), tt.quote) {
t.Errorf("%s: error %q lacks %q", tt.name, err, tt.quote)
}
}
}
// TestAmd64ExtBroadcastEncoding pins the broadcast layer over the memory
// encoding: EVEX.b, bit 4 of byte three, set on the VADDPH 512-bit template
// while the ModR/M mod bits, the SIB byte and the displacement choices keep
// the plain semantics amd64EncodeMemory chooses.
func TestAmd64ExtBroadcastEncoding(t *testing.T) {
add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}
for _, tt := range []struct {
name string
base int
disp int64
want string
}{
{"zero displacement keeps the mod-00 form under the broadcast bit", 9, 0,
"624514505831"},
{"disp8 semantics unchanged", 1, 127,
"6265145058717f"},
{"disp32 semantics unchanged", 2, 8128,
"6265145058b2c01f0000"},
{"RBP keeps the forced displacement", 5, 0,
"62651450587500"},
{"RSP keeps the SIB byte", 12, 0,
"62451450583424"},
} {
got := amd64EncodeBroadcast(add, 30, 29, tt.base, tt.disp)
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestAmd64ExtScaledMemoryEncoding pins the SIB layer over the memory
// encoding: the scale field, the index and the base in one byte, the r/m
// field 100, EVEX.X clearing on an index above 7, and the ModR/M and
// displacement choices keeping the plain semantics, RBP's forced
// displacement included.
func TestAmd64ExtScaledMemoryEncoding(t *testing.T) {
add := []byte{0x62, 0x05, 0x04, 0x40, 0x58, 0xC0}
for _, tt := range []struct {
name string
base int
index int
scale int
disp int64
want string
}{
{"scale 1 encodes the scale field zero", 1, 2, 1, 0,
"62651440583411"},
{"scale 2", 1, 2, 2, 0,
"62651440583451"},
{"scale 4", 1, 2, 4, 0,
"62651440583491"},
{"scale 8", 1, 2, 8, 0,
"626514405834d1"},
{"an index above 7 clears EVEX.X", 1, 12, 2, 0,
"62251440583461"},
{"RBP base keeps the forced displacement", 5, 14, 8, 0,
"622514405874f500"},
{"RSP base takes the SIB byte with the index", 12, 3, 4, 0,
"6245144058349c"},
} {
got := amd64EncodeScaledMemory(add, 30, 29, tt.base, tt.index, tt.scale, tt.disp)
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestAmd64ExtMemoryVocabulary pins the names the shared layer gives the
// memory operand and its two forms.
func TestAmd64ExtMemoryVocabulary(t *testing.T) {
if got := ExtMem.String(); got != "memory operand" {
t.Errorf("ExtMem = %q, want %q", got, "memory operand")
}
if got := ExtFormAmdMemVec.String(); got != "memory into a vector" {
t.Errorf("ExtFormAmdMemVec = %q, want %q", got, "memory into a vector")
}
if got := ExtFormAmdVecMem.String(); got != "a vector into memory" {
t.Errorf("ExtFormAmdVecMem = %q, want %q", got, "a vector into memory")
}
for _, f := range []ExtForm{ExtFormAmdMemVec, ExtFormAmdVecMem} {
if got := f.Arity(); got != 2 {
t.Errorf("%s takes %d operands, want 2", f, got)
}
}
if got := ExtMemory(9, 4096); got.Kind != ExtMem || got.Reg != 9 || got.Imm != 4096 {
t.Errorf("ExtMemory(9, 4096) = %+v, want base 9 with displacement 4096", got)
}
}
File diff suppressed because it is too large. Load diff
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/x86/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+31 -10
View File
@@ -24,20 +24,41 @@ const (
Unknown Arch = ""
)
// FromFilename guesses the target architecture from a source file name. Go
// assembly files conventionally carry a GOARCH suffix such as "_amd64.s",
// "_arm64.s", "_riscv64.s" or "_loong64.s". It returns Unknown when no suffix
// matches.
// FromFilename guesses the target architecture from a source file name, by
// go/build's goodOSArchFile rule: directories are stripped, the name is cut
// at its first dot, everything before the first underscore is ignored and a
// trailing _test segment is dropped; the name is per-architecture only when
// its final underscore segment is an architecture name (foo_amd64.s,
// sys_darwin_arm64.s, foo_amd64_test.s). Any other name narrows nothing:
// not one whose last segment merely contains an architecture name
// (x_loong64y.s), not one with no underscore at all (amd64.s, which
// go/build also keeps generic), and not one differing in case
// (foo_AMD64.s), because the segment match is case-sensitive as go/build's
// is. A name carrying an operating system alone (foo_linux.s) narrows by
// GOOS rather than architecture, and one carrying an architecture gasm
// does not assemble for (vlop_arm.s) reports Unknown so the caller's own
// port filter decides. It returns Unknown when the name narrows nothing.
func FromFilename(name string) Arch {
lower := strings.ToLower(name)
switch {
case strings.Contains(lower, "_amd64"):
if i := strings.LastIndexByte(name, '/'); i >= 0 {
name = name[i+1:]
}
name, _, _ = strings.Cut(name, ".")
i := strings.Index(name, "_")
if i < 0 {
return Unknown
}
segs := strings.Split(name[i:], "_")
if last := len(segs) - 1; segs[last] == "test" {
segs = segs[:last]
}
switch segs[len(segs)-1] {
case "amd64":
return AMD64
case strings.Contains(lower, "_arm64"):
case "arm64":
return ARM64
case strings.Contains(lower, "_riscv64"), strings.Contains(lower, "_riscv"):
case "riscv64":
return RISCV
case strings.Contains(lower, "_loong64"), strings.Contains(lower, "_loong"):
case "loong64":
return LOONG64
default:
return Unknown
+29 -2
View File
@@ -7,13 +7,40 @@ import "testing"
func TestFromFilename(t *testing.T) {
cases := map[string]Arch{
// Positive: the final underscore segment is the architecture.
"avx2_amd64.s": AMD64,
"foo_arm64.s": ARM64,
"portable.s": Unknown,
"decode_ARM64.S": ARM64,
"kernels_amd64.s": AMD64,
"kernel_riscv64.s": RISCV,
"kernel_loong64.s": LOONG64,
"decode_arm64.S": ARM64, // the extension is cut with the first dot
// OS before arch: the _<os>_<arch> pair form.
"sys_darwin_arm64.s": ARM64,
"rt0_linux_amd64.s": AMD64,
// A trailing _test segment is dropped before the suffix rule.
"foo_amd64_test.s": AMD64,
// Trailing junk: a final segment that merely contains an
// architecture name narrows nothing, exactly as go/build's
// segment rule says.
"x_loong64y.s": Unknown,
"asm_amd64x.s": Unknown, // a real GOROOT name
"asm_darwin_arm64_gc.s": Unknown, // the trailing _gc segment is junk
"x_loong.s": Unknown,
"foo_amd64x_test.s": Unknown,
// Case sensitivity: the segment match is exact, as go/build's is.
"foo_AMD64.s": Unknown,
"decode_ARM64.S": Unknown,
// No underscore at all: generic whatever the stem says.
"amd64.s": Unknown,
"portable.s": Unknown,
// A GOOS-only name narrows by operating system, not architecture.
"foo_linux.s": Unknown,
// An architecture gasm does not assemble for: named, but Unknown.
"vlop_arm.s": Unknown,
"foo_riscv.s": Unknown,
// Directories are stripped first, whatever dots they carry.
"some/dir/kernel_loong64.s": LOONG64,
"/a.b/x_amd64.s": AMD64,
}
for name, want := range cases {
if got := FromFilename(name); got != want {
+5294
View File
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/arm64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/util.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/loong64/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+1 -1
View File
@@ -1,4 +1,4 @@
// Code generated by gasm-devkit _gen; DO NOT EDIT.
// Code generated by gasm-sdk _gen; DO NOT EDIT.
// Source: cmd/internal/obj/riscv/anames.go from the Go toolchain.
//
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
+15 -66
View File
@@ -11,7 +11,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectAARCH64Structure checks the basic structure of the emitted
@@ -178,26 +178,10 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q)", work, asmObj)
}
defer os.RemoveAll(work)
st := parseBuildLog(t, buildLog, "main_arm64.s")
defer os.RemoveAll(st.work)
// Expand $WORK in the object path.
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
// Read the toolchain-produced object and assemble the same source with gasm.
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
t.Fatal(err)
@@ -210,30 +194,16 @@ func main() {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64link", "main_arm64.s")
// The package path is "main", the prefix the Go code's references carry.
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
// Replace the toolchain-produced object with gasm's.
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
// Re-link.
if linkLine == "" {
t.Skip("could not find link command in build log")
}
// Expand $WORK in the link command.
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
gasmObj, "GOARCH=arm64")
// Verify the binary exists and contains the symbol.
binPath := filepath.Join(dir, "prog")
binPath := filepath.Join(dir, "prog2")
if _, err := os.Stat(binPath); err != nil {
t.Fatalf("binary not found: %v", err)
}
@@ -296,23 +266,8 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_arm64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
st := parseBuildLog(t, buildLog, "main_arm64.s")
defer os.RemoveAll(st.work)
src, err := os.ReadFile(filepath.Join(dir, "main_arm64.s"))
if err != nil {
@@ -326,19 +281,13 @@ func main() {
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gasmObj, err := img.GOObjectAARCH64("a64dlink", "main_arm64.s")
gasmObj, err := img.GOObjectAARCH64("main", "main_arm64.s")
if err != nil {
t.Fatalf("GOObjectAARCH64: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
linkCmd.Env = append(os.Environ(), "GOARCH=arm64")
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
binData, err := os.ReadFile(filepath.Join(dir, "prog"))
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"),
gasmObj, "GOARCH=arm64")
binData, err := os.ReadFile(filepath.Join(dir, "prog2"))
if err != nil {
t.Fatal(err)
}
+320
View File
@@ -0,0 +1,320 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The assembler's side of the amd64 extension layer: this file turns a parsed
// amd64 statement into the operand form arch.ExtInstr.Encode consumes and
// routes statements only the layer can encode through the registry. It sits
// beside the main amd64 encoders, never inside them: the generated table, the
// legacy SSE paths and the VEX and EVEX mechanisms are untouched, and a
// statement reaches this file only when the mnemonic is registered in the
// extension layer and the scalar paths cannot encode it.
//
// The spellings are the layer's own Plan 9 forms, the ones its metadata
// documents: the vector registers carry the house names X0, Y0 and Z0 (the
// EVEX 128, 256 and 512-bit classes, registers 16 to 31 included), the general
// registers the width their spelling fixes (RAX through R15, EAX through EDI
// and R8D through R15D), the opmask registers K0 through K7, and the memory
// operand the base-relative form off(base) with the optional scaled index
// off(base)(index*scale) the SIB byte carries. The decorations ride the
// operand in braces: the write mask {k1} through {k7} and zeroing {z} on the
// destination, the {1toN} broadcast on the memory source, and {sae} and
// {rn-sae} through {rz-sae} beside the rounding-capable destinations. The
// imm8-control forms take their control byte as the leading $ immediate the
// reference listings write first.
//
// The Feature field stays metadata at assembly time: the assembler has no CPU,
// the toolchain does not gate assembly on CPU features, and every registered
// feature assembles, the behaviour the arm64 wiring established
// (asm/arm64_ext.go). The field remains for the linter and the listing.
package asm
import (
"fmt"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// amd64ExtStatement converts one instruction's operands into the extended
// layer's operand form. pinned reports that the statement belongs to the
// layer: the mnemonic is registered in the amd64 registry and the scalar
// paths cannot encode it. A pinned statement can only encode through the
// layer, so every operand is read here and its diagnostic replaces whatever
// the scalar paths would have said about operands they cannot read; err is
// non-nil for a pinned statement whose operands the layer refuses, and
// extops is complete only when err is nil. Unpinned means the statement is
// nobody's: the caller falls through to the ordinary amd64 encoders, which
// keep their exact behaviour for every statement they knew before.
func amd64ExtStatement(mnem string, ops []*ast.Operand) (extops []arch.ExtOperand, pinned bool, err error) {
if _, ok := LookupExtension(arch.AMD64, mnem); !ok {
return nil, false, nil
}
if Encodable(mnem) {
// A mnemonic the main encoder knows is never the layer's, whatever
// the registry carries: the scalar paths keep the statement. No
// registered mnemonic trips this today (the layer is sealed by
// test), but the guard keeps the fall-through promise exact should
// the toolchain ever learn one of these names.
return nil, false, nil
}
out := make([]arch.ExtOperand, 0, len(ops))
for i, op := range ops {
ext, convErr := amd64ExtOperand(mnem, op, i+1)
if convErr != nil {
return nil, true, convErr
}
out = append(out, ext)
}
return out, true, nil
}
// EncodeAmd64Statement runs one parsed amd64 statement through the layer
// exactly as the assembler does: the operands convert the amd64ExtOperand way
// and the mnemonic resolves and encodes through the registry. pinned reports
// that the statement belongs to the layer alone (the mnemonic is registered
// and the scalar paths cannot encode it); err is the layer's own refusal of
// the operands, the same text the assembler prints, so the linter surfaces
// one diagnostic where assembly would fail. A statement the scalar paths own
// returns pinned false, with nothing to report.
func EncodeAmd64Statement(mnem string, ops []*ast.Operand) (code []byte, pinned bool, err error) {
extops, pinned, err := amd64ExtStatement(mnem, ops)
if !pinned || err != nil {
return nil, pinned, err
}
code, err = EncodeExtension(arch.AMD64, mnem, extops...)
if err != nil {
return nil, true, err
}
return code, true, nil
}
// amd64ExtOperand converts one parsed operand into the layer's form: a $ immediate,
// a vector, general or opmask register, or a base-relative memory operand, each
// with the brace decorations the spelling carries.
func amd64ExtOperand(mnem string, op *ast.Operand, pos int) (arch.ExtOperand, error) {
if op.Kind == ast.OpImmediate {
return amd64ExtImmediate(mnem, op, pos)
}
body, dec, err := amd64ExtDecorations(mnem, op, pos)
if err != nil {
return arch.ExtOperand{}, err
}
if strings.ContainsRune(body, '(') {
ext, ok := amd64ExtMemory(mnem, op, pos)
if !ok {
return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an extended-layer operand: want a base-relative memory operand, off(base)(index*scale) shape", mnem, pos, op.Raw)
}
ext.Broadcast = dec.broadcast
if dec.hasMask {
ext.Mask, ext.HasMask = dec.mask, true
}
ext.Zeroing = dec.zeroing
ext.Round = dec.round
return ext, nil
}
ext, ok := amd64ExtRegister(mnem, op, pos, body)
if !ok {
return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an extended-layer operand: want a vector, general or opmask register, a base-relative memory operand or an immediate", mnem, pos, op.Raw)
}
if dec.hasMask {
ext.Mask, ext.HasMask = dec.mask, true
}
ext.Zeroing = dec.zeroing
ext.Round = dec.round
return ext, nil
}
// amd64ExtImmediate converts a $ immediate into the layer's form. The
// parser folds a parenthesised constant expression in full and reads a bare
// literal greedily, dropping any trailing operator tokens: $255<<8 parses as
// 255 with the shift silently gone. Encoding that silent prefix would
// assemble what the text did not say, so an unparenthesised immediate is
// accepted only when its whole text reads back as one integer carrying the
// parser's value.
func amd64ExtImmediate(mnem string, op *ast.Operand, pos int) (arch.ExtOperand, error) {
if !op.Imm.HasVal {
return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an immediate the layer can read", mnem, pos, op.Raw)
}
text := strings.Join(strings.Fields(strings.TrimPrefix(op.Raw, "$")), "")
if !strings.HasPrefix(text, "(") {
if _, parseErr := strconv.ParseInt(text, 0, 64); parseErr != nil {
return arch.ExtOperand{}, fmt.Errorf("%s: operand %d (%s) is not an immediate the layer can read", mnem, pos, op.Raw)
}
}
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return arch.ExtOperand{Kind: arch.ExtImm, Imm: v}, nil
}
// amd64ExtRegister parses a register operand off a normalised operand body:
// the vector classes X0-X31, Y0-Y31 and Z0-Z31, the width-fixed general
// spellings RAX through R15 and EAX through R15D, and the opmask registers
// K0-K7. The register ranges are left to the encoding, whose diagnostics
// name them.
func amd64ExtRegister(mnem string, op *ast.Operand, pos int, body string) (arch.ExtOperand, bool) {
if body == "" {
return arch.ExtOperand{}, false
}
r, ok := ParseReg(body)
if !ok {
return arch.ExtOperand{}, false
}
switch {
case r.mask:
return arch.ExtOperand{Kind: arch.ExtKReg, Reg: r.idx}, true
case r.isVec():
kind := arch.ExtXMM
switch r.size {
case 32:
kind = arch.ExtYMM
case 64:
kind = arch.ExtZMM
}
return arch.ExtOperand{Kind: kind, Reg: r.idx}, true
case r.size == 8:
return arch.ExtOperand{Kind: arch.ExtR64, Reg: r.idx}, true
case r.size == 4:
return arch.ExtOperand{Kind: arch.ExtR32, Reg: r.idx}, true
}
return arch.ExtOperand{}, false
}
// amd64ExtMemory parses a base-relative memory operand off the parsed
// address: off(base) and off(base)(index*scale), the SIB shapes the layer's
// entries carry. The base and the index are general registers spelled in any
// width the house names offer, the displacement the leading signed term, and
// a group whose scale is not written scales by one, the choice the main
// amd64 paths make for the same spelling. Vector, opmask and segment
// registers are refused as base and index, and so is every frame form: the
// layer's memory operand is hardware addressing alone.
func amd64ExtMemory(mnem string, op *ast.Operand, pos int) (arch.ExtOperand, bool) {
a := op.Addr
if a.Range != nil || a.Base == "" {
return arch.ExtOperand{}, false
}
if a.Sym != nil && a.Sym.Pseudo != "" {
return arch.ExtOperand{}, false
}
base, ok := amd64ExtGprNumber(a.Base)
if !ok {
return arch.ExtOperand{}, false
}
ext := arch.ExtOperand{Kind: arch.ExtMem, Reg: base, Imm: a.Offset}
if a.Index != "" {
index, ok := amd64ExtGprNumber(a.Index)
if !ok {
return arch.ExtOperand{}, false
}
scale := a.Scale
if scale == 0 {
scale = 1
}
ext.Index, ext.Scale, ext.HasIndex = index, scale, true
}
return ext, true
}
// amd64ExtGprNumber resolves one general-register spelling to its number:
// whatever the register table carries for indices 0-15, the vector, opmask,
// x87, MMX, segment and control-debug classes refused, so a vector register
// in a base or index position names itself rather than encoding as its
// same-numbered general register.
func amd64ExtGprNumber(name string) (int, bool) {
r, ok := ParseReg(name)
if !ok || r.mask || r.fp || r.mmx || r.seg != 0 || r.ctl != 0 || r.size > 8 {
return 0, false
}
return r.idx, true
}
// amd64ExtDecorations splits the brace decorations off a normalised operand
// text and returns the body before the first brace and the decorations they
// spell: the write mask {k1} through {k7}, zeroing {z}, the {1toN} broadcast
// and the rounding controls {sae} and {rn-sae} through {rz-sae}, matched
// case-insensitively the way the register spellings are. The mask, zeroing
// and rounding fields land on the operand the conversion builds; whether the
// position takes them is the encoding's judgement, whose diagnostics name the
// entry. The broadcast factor N is checked as a number and otherwise left to
// the entry: the layer's model carries the spelling, not the lane count.
func amd64ExtDecorations(mnem string, op *ast.Operand, pos int) (body string, dec amd64ExtDecor, err error) {
compact := strings.Join(strings.Fields(op.Raw), "")
i := strings.IndexByte(compact, '{')
if i < 0 {
return compact, dec, nil
}
body = compact[:i]
for i < len(compact) {
if compact[i] != '{' {
return "", dec, fmt.Errorf("%s: operand %d (%s): text between brace decorations", mnem, pos, op.Raw)
}
end := strings.IndexByte(compact[i:], '}')
if end < 0 {
return "", dec, fmt.Errorf("%s: operand %d (%s): brace decoration without a closing brace", mnem, pos, op.Raw)
}
content := strings.ToUpper(compact[i+1 : i+end])
switch {
case content == "Z":
if dec.zeroing {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two zeroing decorations", mnem, pos, op.Raw)
}
dec.zeroing = true
case content == "SAE":
if dec.round != arch.ExtRoundNone {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw)
}
dec.round = arch.ExtRoundSAE
case content == "RN-SAE":
if dec.round != arch.ExtRoundNone {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw)
}
dec.round = arch.ExtRoundNearest
case content == "RD-SAE":
if dec.round != arch.ExtRoundNone {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw)
}
dec.round = arch.ExtRoundDown
case content == "RU-SAE":
if dec.round != arch.ExtRoundNone {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw)
}
dec.round = arch.ExtRoundUp
case content == "RZ-SAE":
if dec.round != arch.ExtRoundNone {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two rounding controls", mnem, pos, op.Raw)
}
dec.round = arch.ExtRoundTruncate
case strings.HasPrefix(content, "K") && content != "K":
n, convErr := strconv.Atoi(content[1:])
if convErr != nil || n < 0 {
return "", dec, fmt.Errorf("%s: operand %d (%s): %q is not a mask decoration, want {k1} through {k7}", mnem, pos, op.Raw, content)
}
if dec.hasMask {
return "", dec, fmt.Errorf("%s: operand %d (%s) carries two write masks", mnem, pos, op.Raw)
}
dec.mask, dec.hasMask = n, true
case strings.HasPrefix(content, "1TO"):
if _, convErr := strconv.Atoi(content[3:]); convErr != nil {
return "", dec, fmt.Errorf("%s: operand %d (%s): %q is not a broadcast decoration, want {1toN}", mnem, pos, op.Raw, content)
}
dec.broadcast = true
default:
return "", dec, fmt.Errorf("%s: operand %d (%s): {%s} is not a decoration the layer reads: want {k1} through {k7}, {z}, {1toN}, {sae} or {rn-sae} through {rz-sae}", mnem, pos, op.Raw, content)
}
i += end + 1
}
return body, dec, nil
}
// amd64ExtDecor carries the brace decorations one operand's spelling names.
type amd64ExtDecor struct {
mask int
hasMask bool
zeroing bool
broadcast bool
round arch.ExtRounding
}
+509
View File
@@ -0,0 +1,509 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"fmt"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// assembleAmd64ExtBody parses src, assembles it for amd64 and returns the
// first function's body. Every statement must encode: a failure is the
// test's.
func assembleAmd64ExtBody(t *testing.T, src string) []byte {
t.Helper()
f, errs := parser.Parse("ext_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d functions, want 1", len(img.Funcs))
}
return img.Code[img.Funcs[0].Offset:][:img.Funcs[0].Size]
}
// assembleAmd64ExtError parses and assembles src and returns the assembler's
// error text.
func assembleAmd64ExtError(t *testing.T, src string) string {
t.Helper()
f, errs := parser.Parse("ext_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
_, err := AssembleFile(f)
if err == nil {
t.Fatal("assembled, want an error")
}
return err.Error()
}
const amd64ExtProbeHead = "#include \"textflag.h\"\nTEXT ·t(SB), NOSPLIT, $0\n"
// amd64ExtProbe assembles one statement alone and returns the function body:
// exactly the statement's bytes, no trailing RET.
func amd64ExtProbe(t *testing.T, stmt string) []byte {
t.Helper()
return assembleAmd64ExtBody(t, amd64ExtProbeHead+"\t"+stmt+"\n")
}
// TestAmd64AssembleExtensionGolden drives the wired layer through the full
// assembler: text in, machine bytes out. One statement per family, the
// decorations the layer spells beside them, and the memory mechanism's
// canonical choices; each want is the byte string the registry's golden
// vectors in arch/amd64_ext_test.go and arch/amd64_ext_mem_test.go already
// pin, so these prove the text-to-bytes path lands on the same encoding the
// metadata layer produces.
func TestAmd64AssembleExtensionGolden(t *testing.T) {
tests := []struct {
stmt string
want string
}{
// AVX512-BF16: the two converts and the dot product, the write mask
// riding the destination in braces.
{"VCVTNE2PS2BF16 Z5, Z4, Z6", "62f2574872f4"},
{"VCVTNE2PS2BF16 Z21, Z20, Z23", "62a2574072fc"},
{"VCVTNEPS2BF16 Z5, Y6", "62f27e4872f5"},
{"VCVTNEPS2BF16 Y5, X6{K6}", "62f27e2e72f5"},
{"VDPBF16PS Z5, Z4, Z6{K5}", "62f2564d52f4"},
// AVX512-VP2INTERSECT: sources first, the opmask destination last.
{"VP2INTERSECTD Y2, Y1, K2", "62f26f2868d1"},
// AVX512-FP16, the scalar core: high registers, memory and the
// general-register pairs in both directions.
{"VADDSH X29, X28, X30", "6205160058f4"},
{"VMINSH X5, X4, X6{SAE}", "62f556185df4"},
{"VMOVSH (R9), X30", "62457e081031"},
{"VCVTSH2SI (R9), R12", "6255fe082d21"},
{"VCVTSH2SI X30, EDX", "62957e082dd6"},
{"VCVTSI2SH X29, R12, X30", "624596002af4"},
{"VMOVW R12, X30", "62457d086ef4"},
{"VCMPSH $0x7b, X29, X28, K5", "62931600c2ec7b"},
{"VGETMANTSH $0x0b, X29, X28, X30", "6203140027f40b"},
// The packed FP16 arithmetic and the embedded rounding: {sae} and the
// four rounding modes compose with the write mask and zeroing.
{"VADDPH Z5, Z4, Z6{RN-SAE}", "62f5541858f4"},
{"VADDPH Z5, Z4, Z6{RD-SAE}", "62f5543858f4"},
{"VADDPH Z5, Z4, Z6{RU-SAE}", "62f5545858f4"},
{"VADDPH Z5, Z4, Z6{RZ-SAE}", "62f5547858f4"},
{"VADDPH Z29, Z28, Z30{K7}{Z}", "620514c758f4"},
{"VADDPH Z5, Z4, Z6{K7}{RZ-SAE}", "62f5547f58f4"},
{"VADDPH Z28, (R9), Z30{K7}{Z}", "62451cc75831"},
{"VSQRTPH Z29, Z30{K3}{Z}", "62057ccb51f5"},
{"VFMADD132PH Z29, Z28, Z30", "6206154098f4"},
// The packed conversions: full-width sources and the {1toN} broadcast
// over the integer sources.
{"VCVTPH2W Z5, Z6", "62f57d487df5"},
{"VCVTPH2QQ X5, Z6{RZ-SAE}", "62f57d787bf5"},
{"VCVTPH2PD X5, Z6", "62f57c485af5"},
{"VCVTDQ2PH (R9){1TO8}, Y30", "62457c585b31"},
{"VRNDSCALEPH $0x7b, Z5, Z6", "62f37c4808f57b"},
// The complex families and the imm8 minimum-or-maximum pair.
{"VFCMULCPH Z29, Z28, Z30", "62061740d6f4"},
{"VFMADDCPH Z29, Z28, Z30", "6206164056f4"},
{"VFCMADDCPH Z5, Z4, Z6{RN-SAE}", "62f6571856f4"},
{"VFMADDCSH X29, (R9), X30", "624616005731"},
{"VMINMAXPH $0x88, Z29, (R9), Z30", "62431440523188"},
{"VMINMAXSH $0x88, X28, (R9), X29", "62431c00532988"},
// AVX-VNNI-INT16: the VEX word, its 256-bit length and the memory
// source.
{"VPDPWSUD X2, X1, X3", "c4e26ad2d9"},
{"VPDPWUSDS Y10, Y15, Y8", "c4422dd3c7"},
{"VPDPWSUD X2, 127(RCX), X1", "c4e26ad2497f"},
// The memory mechanism through the ext statements: the disp8 and
// disp32 choices, the RSP-base SIB byte, RBP's forced displacement
// and the scaled index with its EVEX.X handling.
{"VMOVSH 127(RCX), X30", "62657e0810717f"},
{"VMOVSH 8128(RDX), X30", "62657e0810b2c01f0000"},
{"VMOVSH (R12), X30", "62457e08103424"},
{"VMOVSH (RBP), X30", "62657e08107500"},
{"VADDPH Z29, (RCX)(DX*1), Z30", "62651440583411"},
{"VADDPH Z29, (RCX)(R12*2), Z30", "62251440583461"},
{"VADDPH Z29, (RBP)(R14*8), Z30", "622514405874f500"},
}
for _, tt := range tests {
body := amd64ExtProbe(t, tt.stmt)
if got := hex.EncodeToString(body); got != tt.want {
t.Errorf("%s:\n got %s\n want %s", tt.stmt, got, tt.want)
}
}
}
// TestAmd64AssembleExtensionRegistryParity pins the layer's contract over a
// wider slice: every statement here assembles to exactly the bytes
// EncodeExtension produces for the model operands the statement spells, so
// the front end and the registry cannot drift apart unnoticed.
func TestAmd64AssembleExtensionRegistryParity(t *testing.T) {
tests := []struct {
stmt string
mnem string
ops []arch.ExtOperand
}{
{"VCVTNE2PS2BF16 Z5, Z4, Z6", "VCVTNE2PS2BF16",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtZmm(6)}},
{"VCVTNEPS2BF16 Y5, X6{K6}", "VCVTNEPS2BF16",
[]arch.ExtOperand{arch.ExtYmm(5), arch.ExtWriteMasked(arch.ExtXmm(6), 6, false)}},
{"VDPBF16PS Z5, Z4, Z6{K5}", "VDPBF16PS",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtWriteMasked(arch.ExtZmm(6), 5, false)}},
{"VP2INTERSECTD Y2, Y1, K2", "VP2INTERSECTD",
[]arch.ExtOperand{arch.ExtYmm(2), arch.ExtYmm(1), arch.ExtMask(2)}},
{"VADDSH X29, X28, X30", "VADDSH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}},
{"VMINSH X5, X4, X6{SAE}", "VMINSH",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)}},
{"VCVTSI2SH X29, R12, X30", "VCVTSI2SH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtGpr64(12), arch.ExtXmm(30)}},
{"VCVTSH2SI X30, EDX", "VCVTSH2SI",
[]arch.ExtOperand{arch.ExtXmm(30), arch.ExtGpr32(2)}},
{"VCMPSH $0x7b, X29, X28, K5", "VCMPSH",
[]arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtMask(5)}},
{"VGETMANTSH $0x0b, X29, X28, X30", "VGETMANTSH",
[]arch.ExtOperand{arch.ExtImmediate(0x0b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)}},
{"VADDPH Z5, Z4, Z6{K7}{RZ-SAE}", "VADDPH",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4),
arch.ExtRounded(arch.ExtWriteMasked(arch.ExtZmm(6), 7, false), arch.ExtRoundTruncate)}},
{"VADDPH Z28, (R9), Z30{K7}{Z}", "VADDPH",
[]arch.ExtOperand{arch.ExtZmm(28), arch.ExtMemory(9, 0),
arch.ExtWriteMasked(arch.ExtZmm(30), 7, true)}},
{"VCVTDQ2PH (R9){1TO8}, Y30", "VCVTDQ2PH",
[]arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)}},
{"VFMADD132PH Z29, Z28, Z30", "VFMADD132PH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)}},
{"VFMADD231PH Y5, (RCX){1TO8}, Y6", "VFMADD231PH",
[]arch.ExtOperand{arch.ExtYmm(5), arch.ExtBroadcast(1, 0), arch.ExtYmm(6)}},
{"VFCMADDCPH Z5, Z4, Z6{RN-SAE}", "VFCMADDCPH",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundNearest)}},
{"VFMADDCSH X29, (R9), X30", "VFMADDCSH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtMemory(9, 0), arch.ExtXmm(30)}},
{"VMINMAXPH $0x88, Z29, (R9), Z30", "VMINMAXPH",
[]arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtZmm(29), arch.ExtMemory(9, 0), arch.ExtZmm(30)}},
{"VPDPWSUD X2, 127(RCX), X1", "VPDPWSUD",
[]arch.ExtOperand{arch.ExtXmm(2), arch.ExtMemory(1, 127), arch.ExtXmm(1)}},
{"VPDPWSUD X2, (RCX)(R12*2), X1", "VPDPWSUD",
[]arch.ExtOperand{arch.ExtXmm(2), arch.ExtScaledMemory(1, 12, 2, 0), arch.ExtXmm(1)}},
{"VMOVSH X30, (R9)", "VMOVSH",
[]arch.ExtOperand{arch.ExtXmm(30), arch.ExtMemory(9, 0)}},
{"VPDPWUSDS Y10, Y15, Y8", "VPDPWUSDS",
[]arch.ExtOperand{arch.ExtYmm(10), arch.ExtYmm(15), arch.ExtYmm(8)}},
{"VMOVSH 8128(RDX), X30", "VMOVSH",
[]arch.ExtOperand{arch.ExtMemory(2, 8128), arch.ExtXmm(30)}},
{"VADDPH Z29, (RCX)(R12*2), Z30", "VADDPH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtScaledMemory(1, 12, 2, 0), arch.ExtZmm(30)}},
}
for _, tt := range tests {
body := amd64ExtProbe(t, tt.stmt)
want, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...)
if err != nil {
t.Fatalf("%s: registry encode: %v", tt.stmt, err)
}
if !bytes.Equal(body, want) {
t.Errorf("%s:\n got %x\n want %x (the registry encoding)", tt.stmt, body, want)
}
}
}
// TestAmd64AssembleExtensionRefusals pins the diagnostics a pinned statement
// gets from the layer instead of a scalar path's complaint, and the
// conversion's own diagnostics for spellings the layer cannot read. Where
// the Go toolchain knows a family member the shape of the message is its
// rejection bar; the FP16, BF16, VP2INTERSECT and VNNI-INT16 families take
// the GNU assembler's.
func TestAmd64AssembleExtensionRefusals(t *testing.T) {
tests := []struct {
stmt string
want string
}{
{"VADDPH Z33, Z1, Z2", "not an extended-layer operand"},
{"VADDPH Z5, Z4, Z6{K0}", "outside the masking registers k1-k7"},
{"VADDPH Z5, Z4, Z6{Z}", "zeroing without a write mask"},
{"VADDPH Y5, Y4, Y6{RZ-SAE}", "wants a ZMM register"},
{"VADDPH Z5, Z4, Z6{SAE}", "spells {sae} without a mode"},
{"VADDPH Z29, (RCX){RZ-SAE}, Z30", "the memory operand takes none"},
{"VFMULCPH Z5, (RCX){1TO8}, Z6", "the entry's memory operand takes none"},
{"VADDPH Z29, Z28, Z30{BOGUS}", "is not a decoration the layer reads"},
{"VADDPH Z29, Z28, Z30{K7}{K3}", "carries two write masks"},
{"VADDSH X5, X4, X6{K3}", "the entry's destination takes none"},
{"VCMPSH $300, X29, X28, K5", "outside the unsigned byte range"},
{"VGETMANTSH $0x20, X29, X28, X30", "the upper nibble of the mantissa control is reserved"},
{"VCVTSI2SH X29, X12, X30", "general register"},
{"VADDPH Z5, Z4", "got 2 operands"},
{"VMOVSH (Z4), X30", "not an extended-layer operand"},
{"VMOVSH foo+4(SB), X30", "not an extended-layer operand"},
{"VMOVSH 8(RCX)(DX*3), X30", "outside the byte multipliers"},
{"VMOVSH (K1), X30", "not an extended-layer operand"},
}
for _, tt := range tests {
got := assembleAmd64ExtError(t, amd64ExtProbeHead+"\t"+tt.stmt+"\n")
if !strings.Contains(got, tt.want) {
t.Errorf("%s: error %q does not name %q", tt.stmt, got, tt.want)
}
}
}
// TestAmd64AssembleExtensionLabelOffsets proves pass 1 and pass 2 agree on a
// function mixing two ext statements with a backward jump: the label sits
// exactly where the laid-down bytes put it, so the JMP's rel8 reaches it.
func TestAmd64AssembleExtensionLabelOffsets(t *testing.T) {
body := assembleAmd64ExtBody(t, amd64ExtProbeHead+`
VADDPH Z1, Z2, Z3
loop:
VFMADD132PH Z1, Z2, Z3
JMP loop
`)
// Two six-byte EVEX words, then the short JMP whose displacement
// measures from its own end (14) back to the label (6).
if len(body) != 14 {
t.Fatalf("body is %d bytes, want 14", len(body))
}
if body[12] != 0xEB || body[13] != 0xF8 {
t.Errorf("JMP encoded % x, want ebf8", body[12:14])
}
}
// TestAmd64AssembleExtensionLeavesTheMainEncoderAlone pins the non-invasion
// promise on the amd64 side: VPOPCNTD, an AVX-512 instruction the toolchain
// knows and the layer deliberately does not carry, encodes through the main
// EVEX path, byte for byte what that path produces on its own.
func TestAmd64AssembleExtensionLeavesTheMainEncoderAlone(t *testing.T) {
body := amd64ExtProbe(t, "VPOPCNTD Z1, Z2")
e := &enc{}
if err := e.encode("VPOPCNTD", []Operand{Reg{idx: 1, size: 64}, Reg{idx: 2, size: 64}}); err != nil {
t.Fatalf("main encoder: %v", err)
}
if !bytes.Equal(body, e.out) {
t.Errorf("VPOPCNTD Z1, Z2: got %x, want the main encoder's %x", body, e.out)
}
}
// amd64SweepClass recomputes the vector class an entry encodes, the read of
// the template's length field the arch package keeps private: EVEX.L'L in
// byte three, VEX.L in byte two.
func amd64SweepClass(in arch.ExtInstr) arch.ExtOperandKind {
if in.Vex {
if in.Bytes[2]&0x04 != 0 {
return arch.ExtYMM
}
return arch.ExtXMM
}
switch (in.Bytes[3] >> 5) & 3 {
case 0:
return arch.ExtXMM
case 1:
return arch.ExtYMM
default:
return arch.ExtZMM
}
}
// amd64SweepVec spells and models one vector register of the class, the
// house names X, Y and Z the layer's text forms carry.
func amd64SweepVec(kind arch.ExtOperandKind, n int) (string, arch.ExtOperand) {
switch kind {
case arch.ExtXMM:
return fmt.Sprintf("X%d", n), arch.ExtXmm(n)
case arch.ExtYMM:
return fmt.Sprintf("Y%d", n), arch.ExtYmm(n)
default:
return fmt.Sprintf("Z%d", n), arch.ExtZmm(n)
}
}
// amd64SweepGpr spells and models the general register of an entry: the W bit
// picks the width, and an entry that ignores W takes the 64-bit spelling.
func amd64SweepGpr(in arch.ExtInstr) (string, arch.ExtOperand) {
if in.Wig || in.Bytes[2]&0x80 != 0 {
return "R12", arch.ExtGpr64(12)
}
return "R12D", arch.ExtGpr32(12)
}
// amd64SweepImm spells and models the control immediate of an entry: the
// mantissa control keeps its reserved upper nibble at zero, every other
// layout takes a whole byte.
func amd64SweepImm(in arch.ExtInstr) (string, arch.ExtOperand) {
if in.Imm8 == arch.ExtImm8GetMant {
return "$0x0b", arch.ExtImmediate(0x0b)
}
return "$0x7b", arch.ExtImmediate(0x7b)
}
// amd64SweepDest spells and models the register destination with the
// decorations the entry carries on its register form: the write mask beside
// every masked entry, the rounding or the exception suppression beside every
// entry that takes one.
func amd64SweepDest(in arch.ExtInstr, kind arch.ExtOperandKind, n int) (string, arch.ExtOperand) {
text, model := amd64SweepVec(kind, n)
dec := ""
if in.Mask {
dec += "{K5}"
}
switch {
case in.Er:
dec += "{RZ-SAE}"
case in.Sae:
dec += "{SAE}"
}
if dec == "" {
return text, model
}
if in.Mask {
model = arch.ExtWriteMasked(model, 5, false)
}
switch {
case in.Er:
model = arch.ExtRounded(model, arch.ExtRoundTruncate)
case in.Sae:
model = arch.ExtRounded(model, arch.ExtRoundSAE)
}
return text + dec, model
}
// amd64SweepMaskedDest spells and models the destination with the write mask
// alone, the one decoration the memory shape keeps.
func amd64SweepMaskedDest(in arch.ExtInstr, kind arch.ExtOperandKind, n int) (string, arch.ExtOperand) {
text, model := amd64SweepVec(kind, n)
if in.Mask {
return text + "{K5}", arch.ExtWriteMasked(model, 5, false)
}
return text, model
}
// amd64ExtSweepStatements builds, for the first registered entry of every
// distinct amd64 mnemonic, the register-form statement the sweep drives and,
// where the entry carries a memory position, the memory-form statement beside
// it. Each statement comes back with its mnemonic and the model operands the
// text spells, so the sweep can pin the assembled bytes against
// EncodeExtension. The statements follow the registry: a mnemonic registered
// on a form this builder knows lands in the sweep in the same change.
func amd64ExtSweepStatements() (stmts, mnems []string, models [][]arch.ExtOperand, distinct int) {
memText, memModel := "(R9)", arch.ExtMemory(9, 0)
seen := make(map[string]bool)
for _, in := range arch.Extensions(arch.AMD64) {
if seen[in.Name] {
continue
}
seen[in.Name] = true
distinct++
class := amd64SweepClass(in)
// The two-vector forms narrow one side: the half form the
// destination, the wide form the source, and the quarter forms pin
// one side to the XMM class.
srcClass, destClass := class, class
switch in.Form {
case arch.ExtFormAmdVec2Half:
destClass = arch.ExtXMM
if class == arch.ExtZMM {
destClass = arch.ExtYMM
}
case arch.ExtFormAmdVec2Wide:
srcClass = arch.ExtXMM
if class == arch.ExtZMM {
srcClass = arch.ExtYMM
}
case arch.ExtFormAmdVec2Quarter:
srcClass = arch.ExtXMM
case arch.ExtFormAmdVec2ToQuarter:
destClass = arch.ExtXMM
}
srcText, srcModel := amd64SweepVec(srcClass, 1)
src2Text, src2Model := amd64SweepVec(class, 2)
gprText, gprModel := amd64SweepGpr(in)
immText, immModel := amd64SweepImm(in)
add := func(stmt, mnem string, ops ...arch.ExtOperand) {
stmts = append(stmts, stmt)
mnems = append(mnems, mnem)
models = append(models, ops)
}
switch in.Form {
case arch.ExtFormAmdVec3:
dText, dModel := amd64SweepDest(in, class, 3)
add(fmt.Sprintf("%s %s, %s, %s", in.Name, srcText, src2Text, dText), in.Name, srcModel, src2Model, dModel)
if in.Mem == 2 {
dText, dModel = amd64SweepMaskedDest(in, class, 3)
add(fmt.Sprintf("%s %s, %s, %s", in.Name, srcText, memText, dText), in.Name, srcModel, memModel, dModel)
}
case arch.ExtFormAmdVec2, arch.ExtFormAmdVec2Half, arch.ExtFormAmdVec2Wide,
arch.ExtFormAmdVec2Quarter, arch.ExtFormAmdVec2ToQuarter:
dText, dModel := amd64SweepDest(in, destClass, 2)
add(fmt.Sprintf("%s %s, %s", in.Name, srcText, dText), in.Name, srcModel, dModel)
if in.Mem == 1 {
dText, dModel = amd64SweepMaskedDest(in, destClass, 2)
add(fmt.Sprintf("%s %s, %s", in.Name, memText, dText), in.Name, memModel, dModel)
}
case arch.ExtFormAmdMask2:
add(fmt.Sprintf("%s %s, %s, K3", in.Name, srcText, src2Text), in.Name, srcModel, src2Model, arch.ExtMask(3))
case arch.ExtFormAmdVecGprVec:
dText, dModel := amd64SweepVec(class, 2)
add(fmt.Sprintf("%s %s, %s, %s", in.Name, srcText, gprText, dText), in.Name, srcModel, gprModel, dModel)
case arch.ExtFormAmdGprVec:
dText, dModel := amd64SweepVec(class, 1)
add(fmt.Sprintf("%s %s, %s", in.Name, gprText, dText), in.Name, gprModel, dModel)
case arch.ExtFormAmdVecGpr:
add(fmt.Sprintf("%s %s, %s", in.Name, srcText, gprText), in.Name, srcModel, gprModel)
case arch.ExtFormAmdVec3Imm:
dText, dModel := amd64SweepMaskedDest(in, class, 4)
add(fmt.Sprintf("%s %s, %s, %s, %s", in.Name, immText, srcText, src2Text, dText), in.Name, immModel, srcModel, src2Model, dModel)
if in.Mem == 3 {
add(fmt.Sprintf("%s %s, %s, %s, %s", in.Name, immText, srcText, memText, dText), in.Name, immModel, srcModel, memModel, dModel)
}
case arch.ExtFormAmdMask2Imm:
add(fmt.Sprintf("%s %s, %s, %s, K3", in.Name, immText, srcText, src2Text), in.Name, immModel, srcModel, src2Model, arch.ExtMask(3))
if in.Mem == 3 {
add(fmt.Sprintf("%s %s, %s, %s, K3", in.Name, immText, srcText, memText), in.Name, immModel, srcModel, memModel, arch.ExtMask(3))
}
case arch.ExtFormAmdVec2Imm:
dText, dModel := amd64SweepMaskedDest(in, class, 3)
add(fmt.Sprintf("%s %s, %s, %s", in.Name, immText, srcText, dText), in.Name, immModel, srcModel, dModel)
if in.Mem == 2 {
add(fmt.Sprintf("%s %s, %s, %s", in.Name, immText, memText, dText), in.Name, immModel, memModel, dModel)
}
case arch.ExtFormAmdMemVec:
dText, dModel := amd64SweepVec(class, 1)
add(fmt.Sprintf("%s %s, %s", in.Name, memText, dText), in.Name, memModel, dModel)
case arch.ExtFormAmdVecMem:
add(fmt.Sprintf("%s %s, %s", in.Name, srcText, memText), in.Name, srcModel, memModel)
default:
stmts = append(stmts, "")
mnems = append(mnems, in.Name)
models = append(models, nil)
}
}
return stmts, mnems, models, distinct
}
// TestAmd64AssembleExtensionSweep drives every distinct registered mnemonic
// through the full assembler from .s text: the register-form statement of the
// mnemonic's first entry, and the memory-form statement beside it where the
// entry carries a memory position. Every statement must assemble, and every
// body must equal EncodeExtension's encoding of the model operands the text
// spells, so the front end and the registry cannot drift apart on any family.
func TestAmd64AssembleExtensionSweep(t *testing.T) {
stmts, mnems, models, distinct := amd64ExtSweepStatements()
if distinct != 86 {
t.Errorf("the amd64 layer registers %d distinct mnemonics, want 86", distinct)
}
for i, stmt := range stmts {
if stmt == "" {
t.Errorf("%s registers a form the sweep builder does not spell", mnems[i])
continue
}
body := amd64ExtProbe(t, stmt)
want, err := EncodeExtension(arch.AMD64, mnems[i], models[i]...)
if err != nil {
t.Errorf("%s: registry encode: %v", stmt, err)
continue
}
if !bytes.Equal(body, want) {
t.Errorf("%s:\n got %x\n want %x (the registry encoding)", stmt, body, want)
}
}
}
+122
View File
@@ -0,0 +1,122 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "strings"
// encodeAmd64Family routes a mnemonic through the amd64 corpus families the
// dedicated amd64_*.go files implement. The first family that owns the name
// decides the outcome: its bytes, or its error. A name no family claims
// falls through to the scalar dispatch in (*enc).encode untouched.
func (e *enc) encodeAmd64Family(upper string, ops []Operand) (bool, error) {
if ok, err := e.encodeX87(upper, ops); ok {
return true, err
}
if ok, err := e.encodeSystem(upper, ops); ok {
return true, err
}
if ok, err := e.encodeXsave(upper, ops); ok {
return true, err
}
if ok, err := e.encodeSSEMore(upper, ops); ok {
return true, err
}
return false, nil
}
// amd64FamilyEncodable mirrors encodeAmd64Family for the lint-time
// predicate: true when some family owns the mnemonic, whatever the operand
// shapes. It must claim exactly the names encodeAmd64Family does.
func amd64FamilyEncodable(upper string) bool {
if _, ok := x87NoOperand[upper]; ok {
return true
}
if _, ok := x87Arith[upper]; ok {
return true
}
if _, ok := x87FCmov[upper]; ok {
return true
}
if _, ok := x87Compare[upper]; ok {
return true
}
if _, ok := x87MemUnary[upper]; ok {
return true
}
if _, ok := x87Fxsav[upper]; ok {
return true
}
if upper == "FADDDP" {
return true
}
if _, ok := systemNoOperand[upper]; ok {
return true
}
if b, size := splitSize(upper); size != 0 {
switch upper[len(upper)-1] {
case 'B', 'W', 'L', 'Q':
if _, ok := stringOp[b]; ok {
return true
}
}
if m, ok := sysRm[b]; ok && m.sized {
return true
}
}
if _, ok := nopWidth[upper]; ok {
return true
}
if _, ok := cacheControl[upper]; ok {
return true
}
if _, ok := movbeSize[upper]; ok {
return true
}
if _, ok := randSource[base(upper, 6)]; ok {
return true
}
if _, ok := fsGsBase[base(upper, 8)]; ok {
return true
}
if _, ok := descTable[upper]; ok {
return true
}
if _, ok := sysRm[upper]; ok {
return true
}
if _, ok := selectorRead[base(upper, 3)]; ok {
return true
}
if _, ok := farSegLoad[base(upper, 3)]; ok {
return true
}
if upper == "CMPXCHG8B" || upper == "CMPXCHG16B" {
return true
}
if upper == "INVPCID" || upper == "XABORT" {
return true
}
if _, ok := xsaveTable[upper]; ok {
return true
}
if before, ok := strings.CutSuffix(upper, "64"); ok {
if _, ok := xsaveTable[before]; ok {
return true
}
}
switch upper {
case "EMMS", "PSHUFW", "PSLLO", "PSRLO", "PMOVMSKB", "MOVQOZX", "MOVZWW", "MOVSWW":
return true
}
if _, ok := sseMaskmov[upper]; ok {
return true
}
if _, ok := mmxShiftImm[upper]; ok {
return true
}
if _, ok := mmxShiftVar[upper]; ok {
return true
}
return false
}
+566
View File
@@ -0,0 +1,566 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements the amd64 SIMD pieces the main tables lack: the MMX
// bank glue (EMMS, the masked stores, the MMX shifts and shuffle, the
// byte-mask extract), the MOVQ bank crossings' odd spellings and the leaf
// aliases. Every encoding here is pinned byte for byte against go tool asm
// through the corpus lines in amd64_sse_test.go.
// sseMaskmov maps the masked cache-line stores to their prefix and opcode:
// OP src, dst with the second operand in the reg field, no memory operand.
var sseMaskmov = map[string]sseBin{
"MASKMOVQ": {0, 0xF7, false},
"MASKMOVOU": {0x66, 0xF7, false},
}
// sseMoreBin maps the packed and scalar legacy binaries the main table
// lacks, reg = destination and r/m = source: the float comparisons, square
// roots and reciprocal estimates, the SSE3 horizontal arithmetic, the SSE4.1
// packed integers and the SSSE3 sign and horizontal ops. The MMX twins of
// the 0x66-prefixed members drop the prefix in the shared encoder.
var sseMoreBin = map[string]sseBin{
"COMISS": {0, 0x2F, false},
"UCOMISS": {0, 0x2E, false},
"UCOMISD": {0x66, 0x2E, false},
"SQRTPS": {0, 0x51, false},
"SQRTPD": {0x66, 0x51, false},
"SQRTSS": {0xF3, 0x51, false},
"RCPPS": {0, 0x53, false},
"RCPSS": {0xF3, 0x53, false},
"RSQRTPS": {0, 0x52, false},
"RSQRTSS": {0xF3, 0x52, false},
"ADDSUBPD": {0x66, 0xD0, false},
"ADDSUBPS": {0xF2, 0xD0, false},
"HADDPD": {0x66, 0x7C, false},
"HADDPS": {0xF2, 0x7C, false},
"HSUBPD": {0x66, 0x7D, false},
"HSUBPS": {0xF2, 0x7D, false},
"MOVDDUP": {0xF2, 0x12, false},
"MOVSHDUP": {0xF3, 0x16, false},
"MOVSLDUP": {0xF3, 0x12, false},
"LDDQU": {0xF2, 0xF0, false},
"MOVNTDQA": {0x66, 0x2A, true},
"PTEST": {0x66, 0x17, true},
"PABSB": {0x66, 0x1C, true},
"PABSW": {0x66, 0x1D, true},
"PABSD": {0x66, 0x1E, true},
"PACKSSWB": {0x66, 0x63, false},
"PACKUSWB": {0x66, 0x67, false},
"PACKSSLW": {0x66, 0x6B, false},
"PACKUSDW": {0x66, 0x2B, true},
"PADDSB": {0x66, 0xEC, false},
"PADDSW": {0x66, 0xED, false},
"PADDUSB": {0x66, 0xDC, false},
"PADDUSW": {0x66, 0xDD, false},
"PAVGB": {0x66, 0xE0, false},
"PAVGW": {0x66, 0xE3, false},
"PCMPEQQ": {0x66, 0x29, true},
"PCMPGTQ": {0x66, 0x37, true},
"PHADDW": {0x66, 0x01, true},
"PHADDD": {0x66, 0x02, true},
"PHADDSW": {0x66, 0x03, true},
"PHSUBW": {0x66, 0x05, true},
"PHSUBD": {0x66, 0x06, true},
"PHSUBSW": {0x66, 0x07, true},
"PHMINPOSUW": {0x66, 0x41, true},
"PMADDUBSW": {0x66, 0x04, true},
"PMADDWL": {0x66, 0xF5, false},
"PMAXSB": {0x66, 0x3C, true},
"PMAXSD": {0x66, 0x3D, true},
"PMAXSW": {0x66, 0xEE, false},
"PMAXUB": {0x66, 0xDE, false},
"PMAXUD": {0x66, 0x3F, true},
"PMAXUW": {0x66, 0x3E, true},
"PMINSB": {0x66, 0x38, true},
"PMINSD": {0x66, 0x39, true},
"PMINSW": {0x66, 0xEA, false},
"PMINUB": {0x66, 0xDA, false},
"PMINUD": {0x66, 0x3B, true},
"PMINUW": {0x66, 0x3A, true},
"PMULDQ": {0x66, 0x28, true},
"PMULLD": {0x66, 0x40, true},
"PMULHRSW": {0x66, 0x0B, true},
"PMULHUW": {0x66, 0xE4, false},
"PMULHW": {0x66, 0xE5, false},
"PMULLW": {0x66, 0xD5, false},
"PMULULQ": {0x66, 0xF4, false},
"PCMPGTL": {0x66, 0x66, false},
"PMOVSXBD": {0x66, 0x21, true},
"PMOVSXBQ": {0x66, 0x22, true},
"PMOVSXBW": {0x66, 0x20, true},
"PMOVSXDQ": {0x66, 0x25, true},
"PMOVSXWD": {0x66, 0x23, true},
"PMOVSXWQ": {0x66, 0x24, true},
"PMOVZXBD": {0x66, 0x31, true},
"PMOVZXBQ": {0x66, 0x32, true},
"PMOVZXBW": {0x66, 0x30, true},
"PMOVZXDQ": {0x66, 0x35, true},
"PMOVZXWD": {0x66, 0x33, true},
"PMOVZXWQ": {0x66, 0x34, true},
// The conversion aliases the Plan 9 table spells with an L: the dword
// sources and destinations of the packed integer/float converts.
"CVTPL2PD": {0xF3, 0xE6, false},
"CVTPL2PS": {0, 0x5B, false},
"CVTPD2PL": {0xF2, 0xE6, false},
"CVTPS2PL": {0x66, 0x5B, false},
"CVTTPD2PL": {0x66, 0xE6, false},
"CVTTPS2PL": {0xF3, 0x5B, false},
"PSADBW": {0x66, 0xF6, false},
"PSUBSB": {0x66, 0xE8, false},
"PSUBSW": {0x66, 0xE9, false},
"PSUBUSB": {0x66, 0xD8, false},
"PSUBUSW": {0x66, 0xD9, false},
"PSIGNB": {0x66, 0x08, true},
"PSIGNW": {0x66, 0x09, true},
"PSIGND": {0x66, 0x0A, true},
"PUNPCKHBW": {0x66, 0x68, false},
"PUNPCKHLQ": {0x66, 0x6A, false},
"PUNPCKHQDQ": {0x66, 0x6D, false},
"PUNPCKHWL": {0x66, 0x69, false},
"PUNPCKLLQ": {0x66, 0x62, false},
"PUNPCKLQDQ": {0x66, 0x6C, false},
"PUNPCKLWL": {0x66, 0x61, false},
}
// sseMoreImm3 maps the imm8-controlled three-operand instructions the main
// table lacks: OP $imm, src, dst.
var sseMoreImm3 = map[string]sseImm3{
"ROUNDPS": {0x66, 0x08, true},
"ROUNDPD": {0x66, 0x09, true},
"ROUNDSS": {0x66, 0x0A, true},
"ROUNDSD": {0x66, 0x0B, true},
"DPPS": {0x66, 0x40, true},
"DPPD": {0x66, 0x41, true},
"BLENDPS": {0x66, 0x0C, true},
"BLENDPD": {0x66, 0x0D, true},
"INSERTPS": {0x66, 0x21, true},
"MPSADBW": {0x66, 0x42, true},
"PCMPESTRM": {0x66, 0x60, true},
"PCMPESTRI": {0x66, 0x61, true},
"PCMPISTRM": {0x66, 0x62, true},
"PCMPISTRI": {0x66, 0x63, true},
}
// sseBlendv maps the variable blends whose implicit mask is X0: the first
// operand must be the literal X0, the register the hardware reads.
var sseBlendv = map[string]sseBin{
"BLENDVPS": {0x66, 0x14, true},
"BLENDVPD": {0x66, 0x15, true},
"PBLENDVB": {0x66, 0x10, true},
}
// sseHighLow maps the high/low half moves to their load/store opcode pair.
// A memory source loads (reg = destination), a memory destination stores
// (reg = the register source).
var sseHighLow = map[string]sseMove{
"MOVHPD": {0x66, 0x16, 0x17},
"MOVHPS": {0, 0x16, 0x17},
"MOVLPD": {0x66, 0x12, 0x13},
"MOVLPS": {0, 0x12, 0x13},
}
// sseRegReg maps the register-to-register half moves, register destination
// and register source alone: MOVHLPS and MOVLHPS.
var sseRegReg = map[string]sseBin{
"MOVHLPS": {0, 0x12, false},
"MOVLHPS": {0, 0x16, false},
}
// sseMovmsk maps the sign-mask extractions to a GPR: OP vec, gpr.
var sseMovmsk = map[string]sseBin{
"MOVMSKPS": {0, 0x50, false},
"MOVMSKPD": {0x66, 0x50, false},
}
// sseMovnt maps the non-temporal stores, OP reg, mem, plus MOVNTDQA's load
// (which rides sseMoreBin).
var sseMovnt = map[string]sseBin{
"MOVNTPS": {0, 0x2B, false},
"MOVNTPD": {0x66, 0x2B, false},
"MOVNTQ": {0, 0xE7, false},
"MOVNTO": {0x66, 0xE7, false},
"MOVNTIL": {0, 0xC3, false},
"MOVNTIQ": {0, 0xC3, false},
}
// mmxShiftImm lists the packed integer shifts whose immediate form the MMX
// bank spells without the 0x66 prefix; the digit rides the 0F 71/72/73
// group, the same /digits the XMM forms carry.
var mmxShiftImm = map[string]sseShift{
"PSLLW": {0x71, 6},
"PSRLW": {0x71, 2},
"PSRAW": {0x71, 4},
"PSLLL": {0x72, 6},
"PSRLL": {0x72, 2},
"PSRAL": {0x72, 4},
"PSLLQ": {0x73, 6},
"PSRLQ": {0x73, 2},
}
// mmxShiftVar lists the variable-count forms over the MMX bank, the same
// opcodes the XMM variable shifts ride, prefix dropped.
var mmxShiftVar = map[string]byte{
"PSLLW": 0xF1,
"PSRLW": 0xD1,
"PSRAW": 0xE1,
"PSLLL": 0xF2,
"PSRLL": 0xD2,
"PSRAL": 0xE2,
"PSLLQ": 0xF3,
"PSRLQ": 0xD3,
}
// sseOctaShift lists the octa byte shifts' extra spellings: PSLLO and PSRLO
// are the Plan 9 names of PSLLDQ/PSRLDQ, XMM only.
var sseOctaShift = map[string]sseShift{
"PSLLO": {0x73, 7},
"PSRLO": {0x73, 3},
}
// isMmx reports whether the operand is an MMX register.
func isMmx(op Operand) bool {
r, ok := op.(Reg)
return ok && r.mmx
}
// encodeSSEMore encodes the MMX glue and the leaf SIMD spellings. It
// reports whether the mnemonic belongs to the layer; a false result hands
// the mnemonic back to the caller.
func (e *enc) encodeSSEMore(upper string, ops []Operand) (bool, error) {
if upper == "EMMS" {
if len(ops) != 0 {
return true, fmt.Errorf("EMMS takes no operands, got %d", len(ops))
}
return true, e.emit(&instr{opcode: []byte{0x0F, 0x77}, modrm: -1, sib: -1})
}
if m, ok := sseMaskmov[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 || !srcReg.isVec() && !srcReg.mmx || !dstReg.isVec() && !dstReg.mmx {
return true, fmt.Errorf("%s takes two vector or MMX registers", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// The packed shifts over the MMX bank drop the 0x66 prefix the XMM forms
// carry; only a shift name enters the MMX path, the XMM spellings of the
// shifts and every other packed binary fall through to the main tables.
if _, isShift := mmxShiftImm[upper]; isShift {
if e.encodeMmxShiftGate(ops) {
return e.encodeMmxShift(upper, ops)
}
} else if _, isVar := mmxShiftVar[upper]; isVar {
if e.encodeMmxShiftGate(ops) {
return e.encodeMmxShift(upper, ops)
}
}
// PSHUFW is the MMX word shuffle, 0F 70 with no prefix: the XMM twins
// (PSHUFD and friends) dispatch through the main shuffle table.
if upper == "PSHUFW" {
if len(ops) != 3 {
return true, fmt.Errorf("PSHUFW expects 3 operands ($imm, src, dst), got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return true, fmt.Errorf("PSHUFW needs an imm8 first operand")
}
immByte, err := imm8(int64(imm))
if err != nil {
return true, err
}
dstReg, ok2 := ops[2].(Reg)
if !ok2 || !dstReg.mmx {
return true, fmt.Errorf("PSHUFW destination must be an MMX register")
}
i := &instr{opcode: []byte{0x0F, 0x70}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[1], 8); err != nil {
return true, err
}
i.imm = []byte{immByte}
return true, e.emit(i)
}
if spec, ok := sseOctaShift[upper]; ok {
return e.encodeMmxShiftForm(upper, spec, 0x66, ops)
}
// The byte-mask extract over the MMX bank rides the same 0F D7 opcode
// without the prefix; the XMM spelling falls through.
if upper == "PMOVMSKB" && len(ops) == 2 {
if srcReg, ok := ops[0].(Reg); ok && srcReg.mmx {
dstReg, ok2 := ops[1].(Reg)
if !ok2 || dstReg.isVec() || dstReg.mmx {
return true, fmt.Errorf("%s destination must be a general register", upper)
}
i := newInstr(4, []byte{0x0F, 0xD7})
if err := setRM(i, dstReg, srcReg, 4); err != nil {
return true, err
}
return true, e.emit(i)
}
}
// MOVQOZX is the octa-to-quad zero-extend load, F3 0F D6: an MMX or
// memory source into an XMM destination, the MOVQ2DQ opcode.
if upper == "MOVQOZX" {
if len(ops) != 2 {
return true, fmt.Errorf("MOVQOZX expects 2 operands, got %d", len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return true, fmt.Errorf("MOVQOZX destination must be an XMM register")
}
switch ops[0].(type) {
case Reg, Mem, sbMem:
default:
return true, fmt.Errorf("MOVQOZX source must be an MMX register or memory")
}
i := &instr{prefix: 0xF3, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// MOVZWW and MOVSWW are the word zero/sign-extend moves under aliases:
// the MOVWLZX and MOVWLSX opcodes carrying the word width's 0x66 prefix.
switch upper {
case "MOVZWW", "MOVSWW":
op := []byte{0x0F, 0xB7}
if upper == "MOVSWW" {
op[1] = 0xBF
}
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok {
return true, fmt.Errorf("%s destination must be a register", upper)
}
i := newInstr(2, op)
if err := setRM(i, dstReg, ops[0], 2); err != nil {
return true, err
}
return true, e.emit(i)
}
// The packed and scalar binaries: reg = destination, r/m = source, the
// shared encoder carrying the MMX prefix drop.
if m, ok := sseMoreBin[upper]; ok {
return true, e.encodeSSEBin(m, ops)
}
// The imm8-controlled instructions: OP $imm, src, dst.
if m, ok := sseMoreImm3[upper]; ok {
return true, e.encodeSSEImm3(m, ops)
}
// The variable blends with their implicit X0 mask: the first operand is
// the literal X0, the register the encoding leaves out.
if m, ok := sseBlendv[upper]; ok {
if len(ops) != 3 {
return true, fmt.Errorf("%s expects 3 operands (X0, src, dst), got %d", upper, len(ops))
}
x0, ok := ops[0].(Reg)
if !ok || !x0.isVec() || x0.idx != 0 || x0.size != 16 {
return true, fmt.Errorf("%s first operand must be X0", upper)
}
return true, e.encodeSSEBin(m, ops[1:])
}
// The high/low half moves split by direction: a memory source loads, a
// memory destination stores, both two-operand.
if m, ok := sseHighLow[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
dstReg, dstVec := vecReg(ops[1])
var op byte
var reg Reg
var rm Operand
switch {
case srcVec && isX86Mem(ops[1]):
op, reg, rm = m.store, srcReg, ops[1]
case dstVec && isX86Mem(ops[0]):
op, reg, rm = m.load, dstReg, ops[0]
default:
return true, fmt.Errorf("%s takes one vector register and one memory operand", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, reg, rm, 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// The register-to-register half moves.
if m, ok := sseRegReg[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
dstReg, dstVec := vecReg(ops[1])
if !srcVec || !dstVec {
return true, fmt.Errorf("%s takes two XMM registers", upper)
}
i := &instr{opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// The sign-mask extractions: the vector source's sign bits pack into a
// general register.
if m, ok := sseMovmsk[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, srcVec := vecReg(ops[0])
dstReg, ok := ops[1].(Reg)
if !srcVec || !ok || dstReg.isVec() || dstReg.mmx {
return true, fmt.Errorf("%s takes a vector register and a general register", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 4); err != nil {
return true, err
}
return true, e.emit(i)
}
// The non-temporal stores: the register source rides reg, the memory
// destination r/m; MOVNTIL/IQ store from a general register and MOVNTIQ
// carries REX.W.
if m, ok := sseMovnt[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
var srcReg Reg
switch r := ops[0].(type) {
case Reg:
if upper == "MOVNTIL" || upper == "MOVNTIQ" {
if r.isVec() || r.mmx {
return true, fmt.Errorf("%s source must be a general register", upper)
}
srcReg = r
} else {
if !r.isVec() && !r.mmx {
return true, fmt.Errorf("%s source must be a vector or MMX register", upper)
}
srcReg = r
}
default:
return true, fmt.Errorf("%s source must be a register", upper)
}
if !isX86Mem(ops[1]) {
return true, fmt.Errorf("%s destination must be memory", upper)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1, rexW: upper == "MOVNTIQ"}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// EXTRACTPS is the lane extract to a GPR or memory, the PEXTR layout.
if upper == "EXTRACTPS" {
return true, e.encodeSSEExtract(sseExtract{op: []byte{0x0F, 0x3A, 0x17}}, ops)
}
// CVTSL2SS and CVTSQ2SS are the integer-to-scalar-single converts, the
// CVTSL2SD pair's F3 twin: F3 0F 2A with reg = XMM destination, REX.W
// on the quad source spelling.
if upper == "CVTSL2SS" || upper == "CVTSQ2SS" {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.isVec() {
return true, fmt.Errorf("%s destination must be a vector register", upper)
}
i := newInstr(0, []byte{0x0F, 0x2A})
i.rexW = upper == "CVTSQ2SS"
i.prefix = 0xF3
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
return false, nil
}
// encodeMmxShiftGate reports whether the shift's destination operand is an
// MMX register, the case the bank's own prefix-free forms cover.
func (e *enc) encodeMmxShiftGate(ops []Operand) bool {
if len(ops) != 2 {
return false
}
dstReg, ok := ops[1].(Reg)
return ok && dstReg.mmx
}
// encodeMmxShift routes the MMX shift between its immediate form
// (OP $imm, dst, the 0F 71/72/73 /digit group) and its variable-count form
// (OP count, dst, the 0F D1-F3 row).
func (e *enc) encodeMmxShift(upper string, ops []Operand) (bool, error) {
if spec, ok := mmxShiftImm[upper]; ok {
if _, isImm := ops[0].(Imm); isImm {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
immByte, err := imm8(int64(ops[0].(Imm)))
if err != nil {
return true, err
}
i := &instr{opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, ops[1], 8); err != nil {
return true, err
}
i.imm = []byte{immByte}
return true, e.emit(i)
}
}
if op, ok := mmxShiftVar[upper]; ok {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
if !vecOrMem(ops[0]) && !isMmx(ops[0]) {
return true, fmt.Errorf("%s count must be an immediate, an MMX register or memory", upper)
}
dstReg, ok := ops[1].(Reg)
if !ok || !dstReg.mmx {
return true, fmt.Errorf("%s destination must be an MMX register", upper)
}
i := &instr{opcode: []byte{0x0F, op}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
return true, fmt.Errorf("unsupported instruction %q", upper)
}
// encodeMmxShiftForm emits one XMM octa shift: OP $imm, dst, the 0x66
// prefix carried.
func (e *enc) encodeMmxShiftForm(upper string, spec sseShift, prefix byte, ops []Operand) (bool, error) {
if len(ops) != 2 {
return true, fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return true, fmt.Errorf("%s needs an immediate count", upper)
}
immByte, err := imm8(int64(imm))
if err != nil {
return true, err
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return true, fmt.Errorf("%s destination must be the second, vector operand", upper)
}
i := &instr{prefix: prefix, opcode: []byte{0x0F, spec.op}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.digit, dstReg, 8); err != nil {
return true, err
}
i.imm = []byte{immByte}
return true, e.emit(i)
}
File diff suppressed because it is too large. Load diff
+467
View File
@@ -0,0 +1,467 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements the system, flag, string and segment families the Go
// assembler carries: the no-operand controls, the string primitives, the
// sign-extension pair, the multi-byte no-ops, the cache controls, MOVBE, the
// compare-exchange doubles, the random source pair, the FS/GS base pair, the
// descriptor-table controls and the LAR/LSL selector reads. Every encoding
// here is pinned byte for byte against go tool asm through the corpus lines
// in amd64_system_test.go.
// systemNoOperand maps a fixed no-operand mnemonic to its opcode bytes, the
// prefixes spelled out in full.
var systemNoOperand = map[string][]byte{
"CLC": {0xF8},
"STC": {0xF9},
"CMC": {0xF5},
"CLI": {0xFA},
"STI": {0xFB},
"HLT": {0xF4},
"ICEBP": {0xF1},
"XLAT": {0xD7},
"LAHF": {0x9F},
"SAHF": {0x9E},
"PUSHFW": {0x66, 0x9C},
"POPFW": {0x66, 0x9D},
"IRETW": {0x66, 0xCF},
"IRETL": {0xCF},
"IRETQ": {0x48, 0xCF},
"UD1": {0x0F, 0xB9},
"UD2": {0x0F, 0x0B},
"CLAC": {0x0F, 0x01, 0xCA},
"STAC": {0x0F, 0x01, 0xCB},
"CLTS": {0x0F, 0x06},
"INVD": {0x0F, 0x08},
"WBINVD": {0x0F, 0x09},
"SWAPGS": {0x0F, 0x01, 0xF8},
"RSM": {0x0F, 0xAA},
"MONITOR": {0x0F, 0x01, 0xC8},
"MWAIT": {0x0F, 0x01, 0xC9},
"RDMSR": {0x0F, 0x32},
"WRMSR": {0x0F, 0x30},
"RDPMC": {0x0F, 0x33},
"RDPKRU": {0x0F, 0x01, 0xEE},
"WRPKRU": {0x0F, 0x01, 0xEF},
"XSETBV": {0x0F, 0x01, 0xD1},
"SYSENTER": {0x0F, 0x34},
"SYSENTER64": {0x48, 0x0F, 0x34},
"SYSEXIT": {0x0F, 0x35},
"SYSEXIT64": {0x48, 0x0F, 0x35},
"SYSRET": {0x0F, 0x07},
"LEAVE": {0xC9},
"LEAVEQ": {0xC9},
"XEND": {0x0F, 0x01, 0xD5},
"XTEST": {0x0F, 0x01, 0xD6},
"CBW": {0x66, 0x98},
"CWDE": {0x98},
"CDQE": {0x48, 0x98},
"CWD": {0x66, 0x99},
"CDQ": {0x99},
"CQO": {0x48, 0x99},
}
// stringOp maps the string-primitive bases to their 32-bit opcode; the byte
// form is one lower, the word spelling carries 0x66 and the quad spelling
// REX.W, exactly the prefix ladder newInstr applies.
var stringOp = map[string]byte{
"CMPS": 0xA7,
"INS": 0x6D,
"LODS": 0xAD,
"OUTS": 0x6F,
"SCAS": 0xAF,
}
// nopWidth maps the multi-byte no-op spellings to their operand size.
var nopWidth = map[string]int{
"NOPW": 2,
"NOPL": 4,
"NOPQ": 8,
}
// cacheControl maps the one-memory-operand cache controls to their mandatory
// prefix, opcode group and /digit.
var cacheControl = map[string]struct {
prefix byte
op []byte
digit int
}{
"CLFLUSH": {0, []byte{0x0F, 0xAE}, 7},
"CLFLUSHOPT": {0x66, []byte{0x0F, 0xAE}, 7},
"INVLPG": {0, []byte{0x0F, 0x01}, 7},
}
// movbeSize maps the MOVBE spellings to their operand size.
var movbeSize = map[string]int{
"MOVBEW": 2,
"MOVBEL": 4,
"MOVBEQ": 8,
}
// randSource maps the random-source bases to their /digit (RDRAND /6,
// RDSEED /7); the destination register rides r/m, mod 11.
var randSource = map[string]int{
"RDRAND": 6,
"RDSEED": 7,
}
// fsGsBase maps the FS/GS base accessors to their /digit in the F3-prefixed
// 0F AE group; the L and Q spellings exist.
var fsGsBase = map[string]int{
"RDFSBASE": 0,
"RDGSBASE": 1,
"WRFSBASE": 2,
"WRGSBASE": 3,
}
// descTable maps the descriptor-table accesses to their /digit in 0F 01;
// each takes one memory operand alone.
var descTable = map[string]int{
"LGDT": 2,
"LIDT": 3,
"SGDT": 0,
"SIDT": 1,
}
// sysRmEntry is one 0F 00/01 register-or-memory access. sized marks the
// members whose trailing width letter (SLDTW, STRQ, SMSWL) carries the width
// prefix ladder; the rest are fixed-width single names.
type sysRmEntry struct {
group byte
digit int
sized bool
}
// sysRm maps the system register accesses LLDT/LTR/VERR/VERW/SLDT/STR (group
// 0F 00), LMSW/SMSW (0F 01).
var sysRm = map[string]sysRmEntry{
"LLDT": {0x00, 2, false},
"LTR": {0x00, 3, false},
"VERR": {0x00, 4, false},
"VERW": {0x00, 5, false},
"SLDT": {0x00, 0, true},
"STR": {0x00, 1, true},
"LMSW": {0x01, 6, false},
"SMSW": {0x01, 4, true},
}
// selectorRead maps the selector reads LAR and LSL to their opcodes; both
// load the destination register from an r/m selector, width prefixes per the
// suffix.
var selectorRead = map[string]byte{
"LAR": 0x02,
"LSL": 0x03,
}
// farSegLoad maps the far-segment loads to their opcodes; memory source
// alone, destination register, width prefixes per the suffix.
var farSegLoad = map[string]byte{
"LFS": 0xB4,
"LGS": 0xB5,
"LSS": 0xB2,
}
// encodeSystem encodes the system, flag, string and segment families. It
// reports whether the mnemonic belongs to the family.
func (e *enc) encodeSystem(upper string, ops []Operand) (bool, error) {
if op, ok := systemNoOperand[upper]; ok {
if len(ops) != 0 {
return true, fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
return true, e.emit(&instr{opcode: append([]byte(nil), op...), modrm: -1, sib: -1})
}
// The string primitives carry a B/W/L/Q suffix only; CMPSD and friends
// are the SSE compare family's names and must reach their own dispatch.
if b, size := splitSize(upper); size != 0 {
switch upper[len(upper)-1] {
case 'B', 'W', 'L', 'Q':
if op32, ok := stringOp[b]; ok {
return true, e.encodeSystemString(upper, op32, ops)
}
}
}
if size, ok := nopWidth[upper]; ok {
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
}
i := newInstr(size, []byte{0x0F, 0x1F})
if err := setRMDigit(i, 0, ops[0], size); err != nil {
return true, err
}
return true, e.emit(i)
}
if m, ok := cacheControl[upper]; ok {
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 memory operand, got %d", upper, len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("%s requires a memory operand", upper)
}
i := &instr{prefix: m.prefix, opcode: m.op, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
if _, ok := movbeSize[upper]; ok {
return true, e.encodeSystemMovbe(upper, ops)
}
if digit, ok := randSource[base(upper, 6)]; ok {
return true, e.encodeSystemRand(upper, digit, ops)
}
if digit, ok := fsGsBase[base(upper, 8)]; ok {
return true, e.encodeSystemFsGsBase(upper, digit, ops)
}
if digit, ok := descTable[upper]; ok {
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 memory operand, got %d", upper, len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("%s requires a memory operand", upper)
}
i := &instr{opcode: []byte{0x0F, 0x01}, modrm: -1, sib: -1}
if err := setRMDigit(i, digit, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
if m, ok := sysRm[upper]; ok {
return true, e.encodeSystemRm(upper, m, ops)
}
if b, size := splitSize(upper); size != 0 {
if m, ok := sysRm[b]; ok && m.sized {
return true, e.encodeSystemRm(upper, m, ops)
}
}
if op, ok := selectorRead[base(upper, 3)]; ok {
return true, e.encodeSystemSelectorRead(upper, op, ops)
}
if op, ok := farSegLoad[base(upper, 3)]; ok {
return true, e.encodeSystemFarLoad(upper, op, ops)
}
if upper == "CMPXCHG8B" || upper == "CMPXCHG16B" {
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 memory operand, got %d", upper, len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("%s requires a memory operand", upper)
}
i := newInstr(0, []byte{0x0F, 0xC7})
i.rexW = upper == "CMPXCHG16B"
if err := setRMDigit(i, 1, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// INVPCID invalidates a translation-cache entry: 66 0F38 82 with the
// type in a general register and the descriptor in memory.
if upper == "INVPCID" {
if len(ops) != 2 {
return true, fmt.Errorf("INVPCID expects 2 operands, got %d", len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("INVPCID requires a memory descriptor first")
}
srcReg, ok := ops[1].(Reg)
if !ok || srcReg.isVec() {
return true, fmt.Errorf("INVPCID: the second operand must be a general register")
}
i := &instr{prefix: 0x66, opcode: []byte{0x0F, 0x38, 0x82}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
// XABORT carries its imm8 status byte in the C6 F8 group form.
if upper == "XABORT" {
if len(ops) != 1 {
return true, fmt.Errorf("XABORT expects 1 immediate operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return true, fmt.Errorf("XABORT requires an immediate")
}
immByte, err := imm8(int64(imm))
if err != nil {
return true, err
}
return true, e.emit(&instr{opcode: []byte{0xC6, 0xF8}, modrm: -1, sib: -1, imm: []byte{immByte}})
}
return false, nil
}
// base returns the first n characters of an upper-case mnemonic, or the empty
// string when the mnemonic is shorter: the safe head lookup for the families
// whose width suffix rides the tail (RDRANDW, RDFSBASEQ, LARW).
func base(upper string, n int) string {
if len(upper) <= n {
return ""
}
return upper[:n]
}
// encodeSystemString encodes a string primitive: no operands, the width
// suffix picks the byte form, the 0x66 prefix or REX.W. Only the B/W/L/Q
// suffixes belong to the family: CMPSD and friends are the SSE compare
// family's names and must reach their own dispatch.
func (e *enc) encodeSystemString(upper string, op32 byte, ops []Operand) error {
switch upper[len(upper)-1] {
case 'B', 'W', 'L', 'Q':
default:
return fmt.Errorf("unsupported instruction %q", upper)
}
b, size := splitSize(upper)
if _, ok := stringOp[b]; !ok || size == 0 {
return fmt.Errorf("unsupported instruction %q", upper)
}
if len(ops) != 0 {
return fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
// The byte spelling is the 32-bit opcode minus one; the W and Q forms
// ride newInstr's prefix ladder, the L form the bare opcode.
op := op32
if size == 1 {
op--
}
return e.emit(newInstr(size, []byte{op}))
}
// encodeSystemMovbe encodes MOVBE: a register source stores (F1, reg = the
// register, r/m = memory), a register destination loads (F0, same fields).
func (e *enc) encodeSystemMovbe(mnem string, ops []Operand) error {
size := movbeSize[mnem]
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
srcReg, srcIsReg := ops[0].(Reg)
dstReg, dstIsReg := ops[1].(Reg)
var op byte
var reg Reg
var rm Operand
switch {
case srcIsReg && isX86Mem(ops[1]):
op, reg, rm = 0xF1, srcReg, ops[1] // store
case dstIsReg && isX86Mem(ops[0]):
op, reg, rm = 0xF0, dstReg, ops[0] // load
default:
return fmt.Errorf("%s takes one register and one memory operand", mnem)
}
i := newInstr(size, []byte{0x0F, 0x38, op})
if err := setRM(i, reg, rm, size); err != nil {
return err
}
return e.emit(i)
}
// encodeSystemRand encodes RDRAND/RDSEED: the single register operand rides
// r/m under the /digit, mod 11, with the width prefix the suffix picks.
func (e *enc) encodeSystemRand(mnem string, digit int, ops []Operand) error {
b, size := splitSize(mnem)
if _, ok := randSource[b]; !ok || size == 0 {
return fmt.Errorf("unsupported instruction %q", mnem)
}
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 register operand, got %d", mnem, len(ops))
}
dstReg, ok := ops[0].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", mnem)
}
i := newInstr(size, []byte{0x0F, 0xC7})
if err := setRMDigit(i, digit, dstReg, size); err != nil {
return err
}
return e.emit(i)
}
// encodeSystemFsGsBase encodes the FS/GS base accessors: F3-prefixed 0F AE
// under the /digit, the register in r/m; the Q spellings add REX.W.
func (e *enc) encodeSystemFsGsBase(mnem string, digit int, ops []Operand) error {
b, size := splitSize(mnem)
if _, ok := fsGsBase[b]; !ok || (size != 4 && size != 8) {
return fmt.Errorf("unsupported instruction %q", mnem)
}
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 register operand, got %d", mnem, len(ops))
}
dstReg, ok := ops[0].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", mnem)
}
i := newInstr(0, []byte{0x0F, 0xAE})
i.prefix = 0xF3
i.rexW = size == 8
if err := setRMDigit(i, digit, dstReg, 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSystemRm encodes a 0F 00/01 r/m access: the operand is a register or
// memory; the sized members carry the width prefix ladder the suffix fixes.
func (e *enc) encodeSystemRm(mnem string, m sysRmEntry, ops []Operand) error {
size := 0
if m.sized {
_, size = splitSize(mnem)
if size == 0 {
return fmt.Errorf("unsupported instruction %q", mnem)
}
}
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
i := newInstr(size, []byte{0x0F, m.group})
if err := setRMDigit(i, m.digit, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// encodeSystemSelectorRead encodes LAR/LSL: the destination register loads
// from an r/m selector, width prefixes per the suffix.
func (e *enc) encodeSystemSelectorRead(mnem string, op byte, ops []Operand) error {
b, size := splitSize(mnem)
if _, ok := selectorRead[b]; !ok || size == 0 {
return fmt.Errorf("unsupported instruction %q", mnem)
}
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", mnem)
}
i := newInstr(size, []byte{0x0F, op})
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
// encodeSystemFarLoad encodes LFS/LGS/LSS: the destination register loads a
// far pointer from memory, width prefixes per the suffix.
func (e *enc) encodeSystemFarLoad(mnem string, op byte, ops []Operand) error {
b, size := splitSize(mnem)
if _, ok := farSegLoad[b]; !ok || size == 0 {
return fmt.Errorf("unsupported instruction %q", mnem)
}
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if !isX86Mem(ops[0]) {
return fmt.Errorf("%s requires a memory source", mnem)
}
dstReg, ok := ops[1].(Reg)
if !ok || dstReg.isVec() {
return fmt.Errorf("%s destination must be a general register", mnem)
}
i := newInstr(size, []byte{0x0F, op})
if err := setRM(i, dstReg, ops[0], size); err != nil {
return err
}
return e.emit(i)
}
+311
View File
@@ -0,0 +1,311 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "testing"
// amd64SystemCorpus holds every line the Go toolchain's own
// amd64enc.s carries for the system, flag, string and segment families, with the bytes go tool asm
// emits for each: the differential ground truth the family is proven
// against, line for line.
var amd64SystemCorpus = []struct {
line string
want string
}{
{"CBW", "66 98"},
{"CDQ", "99"},
{"CDQE", "48 98"},
{"CLAC", "0f 01 ca"},
{"CLC", "f8"},
{"CLFLUSH (BX)", "0f ae 3b"},
{"CLFLUSH (R11)", "41 0f ae 3b"},
{"CLFLUSHOPT (BX)", "66 0f ae 3b"},
{"CLFLUSHOPT (R11)", "66 41 0f ae 3b"},
{"CLI", "fa"},
{"CLTS", "0f 06"},
{"CMC", "f5"},
{"CMPSB", "a6"},
{"CMPSL", "a7"},
{"CMPSQ", "48 a7"},
{"CMPSW", "66 a7"},
{"CMPXCHG16B (BX)", "48 0f c7 0b"},
{"CMPXCHG16B (R11)", "49 0f c7 0b"},
{"CMPXCHG8B (BX)", "0f c7 0b"},
{"CMPXCHG8B (R11)", "41 0f c7 0b"},
{"CQO", "48 99"},
{"CWD", "66 99"},
{"CWDE", "98"},
{"HLT", "f4"},
{"ICEBP", "f1"},
{"INSB", "6c"},
{"INSL", "6d"},
{"INSW", "66 6d"},
{"INVD", "0f 08"},
{"INVLPG (BX)", "0f 01 3b"},
{"INVLPG (R11)", "41 0f 01 3b"},
{"IRETW", "66 cf"},
{"IRETL", "cf"},
{"IRETQ", "48 cf"},
{"LAHF", "9f"},
{"LARW (BX), DX", "66 0f 02 13"},
{"LARW (R11), DX", "66 41 0f 02 13"},
{"LARW DX, DX", "66 0f 02 d2"},
{"LARW R11, DX", "66 41 0f 02 d3"},
{"LARW (BX), R11", "66 44 0f 02 1b"},
{"LARW (R11), R11", "66 45 0f 02 1b"},
{"LARW DX, R11", "66 44 0f 02 da"},
{"LARW R11, R11", "66 45 0f 02 db"},
{"LARL (BX), DX", "0f 02 13"},
{"LARL (R11), DX", "41 0f 02 13"},
{"LARL DX, DX", "0f 02 d2"},
{"LARL R11, DX", "41 0f 02 d3"},
{"LARL (BX), R11", "44 0f 02 1b"},
{"LARL (R11), R11", "45 0f 02 1b"},
{"LARL DX, R11", "44 0f 02 da"},
{"LARL R11, R11", "45 0f 02 db"},
{"LARQ (BX), DX", "48 0f 02 13"},
{"LARQ (R11), DX", "49 0f 02 13"},
{"LARQ DX, DX", "48 0f 02 d2"},
{"LARQ R11, DX", "49 0f 02 d3"},
{"LARQ (BX), R11", "4c 0f 02 1b"},
{"LARQ (R11), R11", "4d 0f 02 1b"},
{"LARQ DX, R11", "4c 0f 02 da"},
{"LARQ R11, R11", "4d 0f 02 db"},
{"LFSW (BX), DX", "66 0f b4 13"},
{"LFSW (R11), DX", "66 41 0f b4 13"},
{"LFSW (BX), R11", "66 44 0f b4 1b"},
{"LFSW (R11), R11", "66 45 0f b4 1b"},
{"LFSL (BX), DX", "0f b4 13"},
{"LFSL (R11), DX", "41 0f b4 13"},
{"LFSL (BX), R11", "44 0f b4 1b"},
{"LFSL (R11), R11", "45 0f b4 1b"},
{"LFSQ (BX), DX", "48 0f b4 13"},
{"LFSQ (R11), DX", "49 0f b4 13"},
{"LFSQ (BX), R11", "4c 0f b4 1b"},
{"LFSQ (R11), R11", "4d 0f b4 1b"},
{"LGDT (BX)", "0f 01 13"},
{"LGDT (R11)", "41 0f 01 13"},
{"LGSW (BX), DX", "66 0f b5 13"},
{"LGSW (R11), DX", "66 41 0f b5 13"},
{"LGSW (BX), R11", "66 44 0f b5 1b"},
{"LGSW (R11), R11", "66 45 0f b5 1b"},
{"LGSL (BX), DX", "0f b5 13"},
{"LGSL (R11), DX", "41 0f b5 13"},
{"LGSL (BX), R11", "44 0f b5 1b"},
{"LGSL (R11), R11", "45 0f b5 1b"},
{"LGSQ (BX), DX", "48 0f b5 13"},
{"LGSQ (R11), DX", "49 0f b5 13"},
{"LGSQ (BX), R11", "4c 0f b5 1b"},
{"LGSQ (R11), R11", "4d 0f b5 1b"},
{"LIDT (BX)", "0f 01 1b"},
{"LIDT (R11)", "41 0f 01 1b"},
{"LLDT (BX)", "0f 00 13"},
{"LLDT (R11)", "41 0f 00 13"},
{"LLDT DX", "0f 00 d2"},
{"LLDT R11", "41 0f 00 d3"},
{"LMSW (BX)", "0f 01 33"},
{"LMSW (R11)", "41 0f 01 33"},
{"LMSW DX", "0f 01 f2"},
{"LMSW R11", "41 0f 01 f3"},
{"LODSB", "ac"},
{"LODSL", "ad"},
{"LODSQ", "48 ad"},
{"LODSW", "66 ad"},
{"LSLW (BX), DX", "66 0f 03 13"},
{"LSLW (R11), DX", "66 41 0f 03 13"},
{"LSLW DX, DX", "66 0f 03 d2"},
{"LSLW R11, DX", "66 41 0f 03 d3"},
{"LSLW (BX), R11", "66 44 0f 03 1b"},
{"LSLW (R11), R11", "66 45 0f 03 1b"},
{"LSLW DX, R11", "66 44 0f 03 da"},
{"LSLW R11, R11", "66 45 0f 03 db"},
{"LSLL (BX), DX", "0f 03 13"},
{"LSLL (R11), DX", "41 0f 03 13"},
{"LSLL DX, DX", "0f 03 d2"},
{"LSLL R11, DX", "41 0f 03 d3"},
{"LSLL (BX), R11", "44 0f 03 1b"},
{"LSLL (R11), R11", "45 0f 03 1b"},
{"LSLL DX, R11", "44 0f 03 da"},
{"LSLL R11, R11", "45 0f 03 db"},
{"LSLQ (BX), DX", "48 0f 03 13"},
{"LSLQ (R11), DX", "49 0f 03 13"},
{"LSLQ DX, DX", "48 0f 03 d2"},
{"LSLQ R11, DX", "49 0f 03 d3"},
{"LSLQ (BX), R11", "4c 0f 03 1b"},
{"LSLQ (R11), R11", "4d 0f 03 1b"},
{"LSLQ DX, R11", "4c 0f 03 da"},
{"LSLQ R11, R11", "4d 0f 03 db"},
{"LSSW (BX), DX", "66 0f b2 13"},
{"LSSW (R11), DX", "66 41 0f b2 13"},
{"LSSW (BX), R11", "66 44 0f b2 1b"},
{"LSSW (R11), R11", "66 45 0f b2 1b"},
{"LSSL (BX), DX", "0f b2 13"},
{"LSSL (R11), DX", "41 0f b2 13"},
{"LSSL (BX), R11", "44 0f b2 1b"},
{"LSSL (R11), R11", "45 0f b2 1b"},
{"LSSQ (BX), DX", "48 0f b2 13"},
{"LSSQ (R11), DX", "49 0f b2 13"},
{"LSSQ (BX), R11", "4c 0f b2 1b"},
{"LSSQ (R11), R11", "4d 0f b2 1b"},
{"LTR (BX)", "0f 00 1b"},
{"LTR (R11)", "41 0f 00 1b"},
{"LTR DX", "0f 00 da"},
{"LTR R11", "41 0f 00 db"},
{"MONITOR", "0f 01 c8"},
{"MOVBEW DX, (BX)", "66 0f 38 f1 13"},
{"MOVBEW R11, (BX)", "66 44 0f 38 f1 1b"},
{"MOVBEW DX, (R11)", "66 41 0f 38 f1 13"},
{"MOVBEW R11, (R11)", "66 45 0f 38 f1 1b"},
{"MOVBEW (BX), DX", "66 0f 38 f0 13"},
{"MOVBEW (R11), DX", "66 41 0f 38 f0 13"},
{"MOVBEW (BX), R11", "66 44 0f 38 f0 1b"},
{"MOVBEW (R11), R11", "66 45 0f 38 f0 1b"},
{"MOVBEL DX, (BX)", "0f 38 f1 13"},
{"MOVBEL R11, (BX)", "44 0f 38 f1 1b"},
{"MOVBEL DX, (R11)", "41 0f 38 f1 13"},
{"MOVBEL R11, (R11)", "45 0f 38 f1 1b"},
{"MOVBEL (BX), DX", "0f 38 f0 13"},
{"MOVBEL (R11), DX", "41 0f 38 f0 13"},
{"MOVBEL (BX), R11", "44 0f 38 f0 1b"},
{"MOVBEL (R11), R11", "45 0f 38 f0 1b"},
{"MOVBEQ DX, (BX)", "48 0f 38 f1 13"},
{"MOVBEQ R11, (BX)", "4c 0f 38 f1 1b"},
{"MOVBEQ DX, (R11)", "49 0f 38 f1 13"},
{"MOVBEQ R11, (R11)", "4d 0f 38 f1 1b"},
{"MOVBEQ (BX), DX", "48 0f 38 f0 13"},
{"MOVBEQ (R11), DX", "49 0f 38 f0 13"},
{"MOVBEQ (BX), R11", "4c 0f 38 f0 1b"},
{"MOVBEQ (R11), R11", "4d 0f 38 f0 1b"},
{"MWAIT", "0f 01 c9"},
{"NOPW (BX)", "66 0f 1f 03"},
{"NOPW (R11)", "66 41 0f 1f 03"},
{"NOPW DX", "66 0f 1f c2"},
{"NOPW R11", "66 41 0f 1f c3"},
{"NOPL (BX)", "0f 1f 03"},
{"NOPL (R11)", "41 0f 1f 03"},
{"NOPL DX", "0f 1f c2"},
{"NOPL R11", "41 0f 1f c3"},
{"OUTSB", "6e"},
{"OUTSL", "6f"},
{"OUTSW", "66 6f"},
{"POPFW", "66 9d"},
{"PUSHFW", "66 9c"},
{"RDFSBASEL DX", "f3 0f ae c2"},
{"RDFSBASEL R11", "f3 41 0f ae c3"},
{"RDGSBASEL DX", "f3 0f ae ca"},
{"RDGSBASEL R11", "f3 41 0f ae cb"},
{"RDFSBASEQ DX", "f3 48 0f ae c2"},
{"RDFSBASEQ R11", "f3 49 0f ae c3"},
{"RDGSBASEQ DX", "f3 48 0f ae ca"},
{"RDGSBASEQ R11", "f3 49 0f ae cb"},
{"RDMSR", "0f 32"},
{"RDPKRU", "0f 01 ee"},
{"RDPMC", "0f 33"},
{"RDRANDW DX", "66 0f c7 f2"},
{"RDRANDW R11", "66 41 0f c7 f3"},
{"RDRANDL DX", "0f c7 f2"},
{"RDRANDL R11", "41 0f c7 f3"},
{"RDRANDQ DX", "48 0f c7 f2"},
{"RDRANDQ R11", "49 0f c7 f3"},
{"RDSEEDW DX", "66 0f c7 fa"},
{"RDSEEDW R11", "66 41 0f c7 fb"},
{"RDSEEDL DX", "0f c7 fa"},
{"RDSEEDL R11", "41 0f c7 fb"},
{"RDSEEDQ DX", "48 0f c7 fa"},
{"RDSEEDQ R11", "49 0f c7 fb"},
{"RSM", "0f aa"},
{"SAHF", "9e"},
{"SCASB", "ae"},
{"SCASL", "af"},
{"SCASQ", "48 af"},
{"SCASW", "66 af"},
{"SGDT (BX)", "0f 01 03"},
{"SGDT (R11)", "41 0f 01 03"},
{"SIDT (BX)", "0f 01 0b"},
{"SIDT (R11)", "41 0f 01 0b"},
{"SLDTW (BX)", "66 0f 00 03"},
{"SLDTW (R11)", "66 41 0f 00 03"},
{"SLDTW DX", "66 0f 00 c2"},
{"SLDTW R11", "66 41 0f 00 c3"},
{"SLDTL (BX)", "0f 00 03"},
{"SLDTL (R11)", "41 0f 00 03"},
{"SLDTL DX", "0f 00 c2"},
{"SLDTL R11", "41 0f 00 c3"},
{"SLDTQ (BX)", "48 0f 00 03"},
{"SLDTQ (R11)", "49 0f 00 03"},
{"SLDTQ DX", "48 0f 00 c2"},
{"SLDTQ R11", "49 0f 00 c3"},
{"SMSWW (BX)", "66 0f 01 23"},
{"SMSWW (R11)", "66 41 0f 01 23"},
{"SMSWW DX", "66 0f 01 e2"},
{"SMSWW R11", "66 41 0f 01 e3"},
{"SMSWL (BX)", "0f 01 23"},
{"SMSWL (R11)", "41 0f 01 23"},
{"SMSWL DX", "0f 01 e2"},
{"SMSWL R11", "41 0f 01 e3"},
{"SMSWQ (BX)", "48 0f 01 23"},
{"SMSWQ (R11)", "49 0f 01 23"},
{"SMSWQ DX", "48 0f 01 e2"},
{"SMSWQ R11", "49 0f 01 e3"},
{"STAC", "0f 01 cb"},
{"STC", "f9"},
{"STI", "fb"},
{"STRW (BX)", "66 0f 00 0b"},
{"STRW (R11)", "66 41 0f 00 0b"},
{"STRW DX", "66 0f 00 ca"},
{"STRW R11", "66 41 0f 00 cb"},
{"STRL (BX)", "0f 00 0b"},
{"STRL (R11)", "41 0f 00 0b"},
{"STRL DX", "0f 00 ca"},
{"STRL R11", "41 0f 00 cb"},
{"STRQ (BX)", "48 0f 00 0b"},
{"STRQ (R11)", "49 0f 00 0b"},
{"STRQ DX", "48 0f 00 ca"},
{"STRQ R11", "49 0f 00 cb"},
{"SWAPGS", "0f 01 f8"},
{"SYSENTER", "0f 34"},
{"SYSENTER64", "48 0f 34"},
{"SYSEXIT", "0f 35"},
{"SYSEXIT64", "48 0f 35"},
{"SYSRET", "0f 07"},
{"UD1", "0f b9"},
{"UD2", "0f 0b"},
{"VERR (BX)", "0f 00 23"},
{"VERR (R11)", "41 0f 00 23"},
{"VERR DX", "0f 00 e2"},
{"VERR R11", "41 0f 00 e3"},
{"VERW (BX)", "0f 00 2b"},
{"VERW (R11)", "41 0f 00 2b"},
{"VERW DX", "0f 00 ea"},
{"VERW R11", "41 0f 00 eb"},
{"WBINVD", "0f 09"},
{"WRFSBASEL DX", "f3 0f ae d2"},
{"WRFSBASEL R11", "f3 41 0f ae d3"},
{"WRGSBASEL DX", "f3 0f ae da"},
{"WRGSBASEL R11", "f3 41 0f ae db"},
{"WRFSBASEQ DX", "f3 48 0f ae d2"},
{"WRFSBASEQ R11", "f3 49 0f ae d3"},
{"WRGSBASEQ DX", "f3 48 0f ae da"},
{"WRGSBASEQ R11", "f3 49 0f ae db"},
{"WRMSR", "0f 30"},
{"WRPKRU", "0f 01 ef"},
{"XLAT", "d7"},
{"XSETBV", "0f 01 d1"},
}
// TestAmd64SystemCorpus assembles every corpus line and requires the same bytes
// go tool asm emits for it.
func TestAmd64SystemCorpus(t *testing.T) {
for _, tc := range amd64SystemCorpus {
fn := firstText(t, "TEXT ·p(SB), 4, $0\n\t"+tc.line+"\n")
code, _, err := Assemble(fn)
if err != nil {
t.Errorf("%s: %v", tc.line, err)
continue
}
if got := hexBytes(code); got != tc.want {
t.Errorf("%s: got %s, want %s", tc.line, got, tc.want)
}
}
}
+884
View File
@@ -0,0 +1,884 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "testing"
// amd64VexCorpus holds every line the Go toolchain's own
// amd64enc.s carries for the VEX (AVX/AVX2) families and the general-register count forms, with the bytes go
// tool asm emits for each: the differential ground truth the layer is
// proven against, line for line.
var amd64VexCorpus = []struct {
line string
want string
}{
{"BEXTRL R9, (BX), DX", "c4 e2 30 f7 13"},
{"BEXTRL R9, (BX), R11", "c4 62 30 f7 1b"},
{"BEXTRL R9, (R11), DX", "c4 c2 30 f7 13"},
{"BEXTRL R9, (R11), R11", "c4 42 30 f7 1b"},
{"BEXTRL R9, DX, DX", "c4 e2 30 f7 d2"},
{"BEXTRL R9, DX, R11", "c4 62 30 f7 da"},
{"BEXTRL R9, R11, DX", "c4 c2 30 f7 d3"},
{"BEXTRL R9, R11, R11", "c4 42 30 f7 db"},
{"BEXTRQ R14, (BX), DX", "c4 e2 88 f7 13"},
{"BEXTRQ R14, (BX), R11", "c4 62 88 f7 1b"},
{"BEXTRQ R14, (R11), DX", "c4 c2 88 f7 13"},
{"BEXTRQ R14, (R11), R11", "c4 42 88 f7 1b"},
{"BEXTRQ R14, DX, DX", "c4 e2 88 f7 d2"},
{"BEXTRQ R14, DX, R11", "c4 62 88 f7 da"},
{"BEXTRQ R14, R11, DX", "c4 c2 88 f7 d3"},
{"BEXTRQ R14, R11, R11", "c4 42 88 f7 db"},
{"BZHIL R9, (BX), DX", "c4 e2 30 f5 13"},
{"BZHIL R9, (BX), R11", "c4 62 30 f5 1b"},
{"BZHIL R9, (R11), DX", "c4 c2 30 f5 13"},
{"BZHIL R9, (R11), R11", "c4 42 30 f5 1b"},
{"BZHIL R9, DX, DX", "c4 e2 30 f5 d2"},
{"BZHIL R9, DX, R11", "c4 62 30 f5 da"},
{"BZHIL R9, R11, DX", "c4 c2 30 f5 d3"},
{"BZHIL R9, R11, R11", "c4 42 30 f5 db"},
{"BZHIQ R14, (BX), DX", "c4 e2 88 f5 13"},
{"BZHIQ R14, (BX), R11", "c4 62 88 f5 1b"},
{"BZHIQ R14, (R11), DX", "c4 c2 88 f5 13"},
{"BZHIQ R14, (R11), R11", "c4 42 88 f5 1b"},
{"BZHIQ R14, DX, DX", "c4 e2 88 f5 d2"},
{"BZHIQ R14, DX, R11", "c4 62 88 f5 da"},
{"BZHIQ R14, R11, DX", "c4 c2 88 f5 d3"},
{"BZHIQ R14, R11, R11", "c4 42 88 f5 db"},
{"SARXL R9, (BX), DX", "c4 e2 32 f7 13"},
{"SARXL R9, (BX), R11", "c4 62 32 f7 1b"},
{"SARXL R9, (R11), DX", "c4 c2 32 f7 13"},
{"SARXL R9, (R11), R11", "c4 42 32 f7 1b"},
{"SARXL R9, DX, DX", "c4 e2 32 f7 d2"},
{"SARXL R9, DX, R11", "c4 62 32 f7 da"},
{"SARXL R9, R11, DX", "c4 c2 32 f7 d3"},
{"SARXL R9, R11, R11", "c4 42 32 f7 db"},
{"SARXQ R14, (BX), DX", "c4 e2 8a f7 13"},
{"SARXQ R14, (BX), R11", "c4 62 8a f7 1b"},
{"SARXQ R14, (R11), DX", "c4 c2 8a f7 13"},
{"SARXQ R14, (R11), R11", "c4 42 8a f7 1b"},
{"SARXQ R14, DX, DX", "c4 e2 8a f7 d2"},
{"SARXQ R14, DX, R11", "c4 62 8a f7 da"},
{"SARXQ R14, R11, DX", "c4 c2 8a f7 d3"},
{"SARXQ R14, R11, R11", "c4 42 8a f7 db"},
{"SHLXL R9, (BX), DX", "c4 e2 31 f7 13"},
{"SHLXL R9, (BX), R11", "c4 62 31 f7 1b"},
{"SHLXL R9, (R11), DX", "c4 c2 31 f7 13"},
{"SHLXL R9, (R11), R11", "c4 42 31 f7 1b"},
{"SHLXL R9, DX, DX", "c4 e2 31 f7 d2"},
{"SHLXL R9, DX, R11", "c4 62 31 f7 da"},
{"SHLXL R9, R11, DX", "c4 c2 31 f7 d3"},
{"SHLXL R9, R11, R11", "c4 42 31 f7 db"},
{"SHLXQ R14, (BX), DX", "c4 e2 89 f7 13"},
{"SHLXQ R14, (BX), R11", "c4 62 89 f7 1b"},
{"SHLXQ R14, (R11), DX", "c4 c2 89 f7 13"},
{"SHLXQ R14, (R11), R11", "c4 42 89 f7 1b"},
{"SHLXQ R14, DX, DX", "c4 e2 89 f7 d2"},
{"SHLXQ R14, DX, R11", "c4 62 89 f7 da"},
{"SHLXQ R14, R11, DX", "c4 c2 89 f7 d3"},
{"SHLXQ R14, R11, R11", "c4 42 89 f7 db"},
{"SHRXL R9, (BX), DX", "c4 e2 33 f7 13"},
{"SHRXL R9, (BX), R11", "c4 62 33 f7 1b"},
{"SHRXL R9, (R11), DX", "c4 c2 33 f7 13"},
{"SHRXL R9, (R11), R11", "c4 42 33 f7 1b"},
{"SHRXL R9, DX, DX", "c4 e2 33 f7 d2"},
{"SHRXL R9, DX, R11", "c4 62 33 f7 da"},
{"SHRXL R9, R11, DX", "c4 c2 33 f7 d3"},
{"SHRXL R9, R11, R11", "c4 42 33 f7 db"},
{"SHRXQ R14, (BX), DX", "c4 e2 8b f7 13"},
{"SHRXQ R14, (BX), R11", "c4 62 8b f7 1b"},
{"SHRXQ R14, (R11), DX", "c4 c2 8b f7 13"},
{"SHRXQ R14, (R11), R11", "c4 42 8b f7 1b"},
{"SHRXQ R14, DX, DX", "c4 e2 8b f7 d2"},
{"SHRXQ R14, DX, R11", "c4 62 8b f7 da"},
{"SHRXQ R14, R11, DX", "c4 c2 8b f7 d3"},
{"SHRXQ R14, R11, R11", "c4 42 8b f7 db"},
{"VADDSUBPD (BX), X9, X11", "c5 31 d0 1b"},
{"VADDSUBPD (BX), X9, X2", "c5 b1 d0 13"},
{"VADDSUBPD (BX), Y15, Y11", "c5 05 d0 1b"},
{"VADDSUBPD (BX), Y15, Y2", "c5 85 d0 13"},
{"VADDSUBPD (R11), X9, X11", "c4 41 31 d0 1b"},
{"VADDSUBPD (R11), X9, X2", "c4 c1 31 d0 13"},
{"VADDSUBPD (R11), Y15, Y11", "c4 41 05 d0 1b"},
{"VADDSUBPD (R11), Y15, Y2", "c4 c1 05 d0 13"},
{"VADDSUBPD X11, X9, X11", "c4 41 31 d0 db"},
{"VADDSUBPD X11, X9, X2", "c4 c1 31 d0 d3"},
{"VADDSUBPD X2, X9, X11", "c5 31 d0 da"},
{"VADDSUBPD X2, X9, X2", "c5 b1 d0 d2"},
{"VADDSUBPD Y11, Y15, Y11", "c4 41 05 d0 db"},
{"VADDSUBPD Y11, Y15, Y2", "c4 c1 05 d0 d3"},
{"VADDSUBPD Y2, Y15, Y11", "c5 05 d0 da"},
{"VADDSUBPD Y2, Y15, Y2", "c5 85 d0 d2"},
{"VADDSUBPS (BX), X9, X11", "c5 33 d0 1b"},
{"VADDSUBPS (BX), X9, X2", "c5 b3 d0 13"},
{"VADDSUBPS (BX), Y15, Y11", "c5 07 d0 1b"},
{"VADDSUBPS (BX), Y15, Y2", "c5 87 d0 13"},
{"VADDSUBPS (R11), X9, X11", "c4 41 33 d0 1b"},
{"VADDSUBPS (R11), X9, X2", "c4 c1 33 d0 13"},
{"VADDSUBPS (R11), Y15, Y11", "c4 41 07 d0 1b"},
{"VADDSUBPS (R11), Y15, Y2", "c4 c1 07 d0 13"},
{"VADDSUBPS X11, X9, X11", "c4 41 33 d0 db"},
{"VADDSUBPS X11, X9, X2", "c4 c1 33 d0 d3"},
{"VADDSUBPS X2, X9, X11", "c5 33 d0 da"},
{"VADDSUBPS X2, X9, X2", "c5 b3 d0 d2"},
{"VADDSUBPS Y11, Y15, Y11", "c4 41 07 d0 db"},
{"VADDSUBPS Y11, Y15, Y2", "c4 c1 07 d0 d3"},
{"VADDSUBPS Y2, Y15, Y11", "c5 07 d0 da"},
{"VADDSUBPS Y2, Y15, Y2", "c5 87 d0 d2"},
{"VAESIMC (BX), X11", "c4 62 79 db 1b"},
{"VAESIMC (BX), X2", "c4 e2 79 db 13"},
{"VAESIMC (R11), X11", "c4 42 79 db 1b"},
{"VAESIMC (R11), X2", "c4 c2 79 db 13"},
{"VAESIMC X11, X11", "c4 42 79 db db"},
{"VAESIMC X11, X2", "c4 c2 79 db d3"},
{"VAESIMC X2, X11", "c4 62 79 db da"},
{"VAESIMC X2, X2", "c4 e2 79 db d2"},
{"VBLENDPD $7, (BX), X9, X11", "c4 63 31 0d 1b 07"},
{"VBLENDPD $7, (BX), X9, X2", "c4 e3 31 0d 13 07"},
{"VBLENDPD $7, (BX), Y15, Y11", "c4 63 05 0d 1b 07"},
{"VBLENDPD $7, (BX), Y15, Y2", "c4 e3 05 0d 13 07"},
{"VBLENDPD $7, (R11), X9, X11", "c4 43 31 0d 1b 07"},
{"VBLENDPD $7, (R11), X9, X2", "c4 c3 31 0d 13 07"},
{"VBLENDPD $7, (R11), Y15, Y11", "c4 43 05 0d 1b 07"},
{"VBLENDPD $7, (R11), Y15, Y2", "c4 c3 05 0d 13 07"},
{"VBLENDPD $7, X11, X9, X11", "c4 43 31 0d db 07"},
{"VBLENDPD $7, X11, X9, X2", "c4 c3 31 0d d3 07"},
{"VBLENDPD $7, X2, X9, X11", "c4 63 31 0d da 07"},
{"VBLENDPD $7, X2, X9, X2", "c4 e3 31 0d d2 07"},
{"VBLENDPD $7, Y11, Y15, Y11", "c4 43 05 0d db 07"},
{"VBLENDPD $7, Y11, Y15, Y2", "c4 c3 05 0d d3 07"},
{"VBLENDPD $7, Y2, Y15, Y11", "c4 63 05 0d da 07"},
{"VBLENDPD $7, Y2, Y15, Y2", "c4 e3 05 0d d2 07"},
{"VBLENDPS $7, (BX), X9, X11", "c4 63 31 0c 1b 07"},
{"VBLENDPS $7, (BX), X9, X2", "c4 e3 31 0c 13 07"},
{"VBLENDPS $7, (BX), Y15, Y11", "c4 63 05 0c 1b 07"},
{"VBLENDPS $7, (BX), Y15, Y2", "c4 e3 05 0c 13 07"},
{"VBLENDPS $7, (R11), X9, X11", "c4 43 31 0c 1b 07"},
{"VBLENDPS $7, (R11), X9, X2", "c4 c3 31 0c 13 07"},
{"VBLENDPS $7, (R11), Y15, Y11", "c4 43 05 0c 1b 07"},
{"VBLENDPS $7, (R11), Y15, Y2", "c4 c3 05 0c 13 07"},
{"VBLENDPS $7, X11, X9, X11", "c4 43 31 0c db 07"},
{"VBLENDPS $7, X11, X9, X2", "c4 c3 31 0c d3 07"},
{"VBLENDPS $7, X2, X9, X11", "c4 63 31 0c da 07"},
{"VBLENDPS $7, X2, X9, X2", "c4 e3 31 0c d2 07"},
{"VBLENDPS $7, Y11, Y15, Y11", "c4 43 05 0c db 07"},
{"VBLENDPS $7, Y11, Y15, Y2", "c4 c3 05 0c d3 07"},
{"VBLENDPS $7, Y2, Y15, Y11", "c4 63 05 0c da 07"},
{"VBLENDPS $7, Y2, Y15, Y2", "c4 e3 05 0c d2 07"},
{"VBLENDVPD X12, (BX), X9, X11", "c4 63 31 4b 1b c0"},
{"VBLENDVPD X12, (BX), X9, X2", "c4 e3 31 4b 13 c0"},
{"VBLENDVPD X12, (R11), X9, X11", "c4 43 31 4b 1b c0"},
{"VBLENDVPD X12, (R11), X9, X2", "c4 c3 31 4b 13 c0"},
{"VBLENDVPD X12, X11, X9, X11", "c4 43 31 4b db c0"},
{"VBLENDVPD X12, X11, X9, X2", "c4 c3 31 4b d3 c0"},
{"VBLENDVPD X12, X2, X9, X11", "c4 63 31 4b da c0"},
{"VBLENDVPD X12, X2, X9, X2", "c4 e3 31 4b d2 c0"},
{"VBLENDVPD Y13, (BX), Y15, Y11", "c4 63 05 4b 1b d0"},
{"VBLENDVPD Y13, (BX), Y15, Y2", "c4 e3 05 4b 13 d0"},
{"VBLENDVPD Y13, (R11), Y15, Y11", "c4 43 05 4b 1b d0"},
{"VBLENDVPD Y13, (R11), Y15, Y2", "c4 c3 05 4b 13 d0"},
{"VBLENDVPD Y13, Y11, Y15, Y11", "c4 43 05 4b db d0"},
{"VBLENDVPD Y13, Y11, Y15, Y2", "c4 c3 05 4b d3 d0"},
{"VBLENDVPD Y13, Y2, Y15, Y11", "c4 63 05 4b da d0"},
{"VBLENDVPD Y13, Y2, Y15, Y2", "c4 e3 05 4b d2 d0"},
{"VBLENDVPS X12, (BX), X9, X11", "c4 63 31 4a 1b c0"},
{"VBLENDVPS X12, (BX), X9, X2", "c4 e3 31 4a 13 c0"},
{"VBLENDVPS X12, (R11), X9, X11", "c4 43 31 4a 1b c0"},
{"VBLENDVPS X12, (R11), X9, X2", "c4 c3 31 4a 13 c0"},
{"VBLENDVPS X12, X11, X9, X11", "c4 43 31 4a db c0"},
{"VBLENDVPS X12, X11, X9, X2", "c4 c3 31 4a d3 c0"},
{"VBLENDVPS X12, X2, X9, X11", "c4 63 31 4a da c0"},
{"VBLENDVPS X12, X2, X9, X2", "c4 e3 31 4a d2 c0"},
{"VBLENDVPS Y13, (BX), Y15, Y11", "c4 63 05 4a 1b d0"},
{"VBLENDVPS Y13, (BX), Y15, Y2", "c4 e3 05 4a 13 d0"},
{"VBLENDVPS Y13, (R11), Y15, Y11", "c4 43 05 4a 1b d0"},
{"VBLENDVPS Y13, (R11), Y15, Y2", "c4 c3 05 4a 13 d0"},
{"VBLENDVPS Y13, Y11, Y15, Y11", "c4 43 05 4a db d0"},
{"VBLENDVPS Y13, Y11, Y15, Y2", "c4 c3 05 4a d3 d0"},
{"VBLENDVPS Y13, Y2, Y15, Y11", "c4 63 05 4a da d0"},
{"VBLENDVPS Y13, Y2, Y15, Y2", "c4 e3 05 4a d2 d0"},
{"VBROADCASTF128 (BX), Y11", "c4 62 7d 1a 1b"},
{"VBROADCASTF128 (BX), Y2", "c4 e2 7d 1a 13"},
{"VBROADCASTF128 (R11), Y11", "c4 42 7d 1a 1b"},
{"VBROADCASTF128 (R11), Y2", "c4 c2 7d 1a 13"},
{"VDPPD $7, (BX), X9, X11", "c4 63 31 41 1b 07"},
{"VDPPD $7, (BX), X9, X2", "c4 e3 31 41 13 07"},
{"VDPPD $7, (R11), X9, X11", "c4 43 31 41 1b 07"},
{"VDPPD $7, (R11), X9, X2", "c4 c3 31 41 13 07"},
{"VDPPD $7, X11, X9, X11", "c4 43 31 41 db 07"},
{"VDPPD $7, X11, X9, X2", "c4 c3 31 41 d3 07"},
{"VDPPD $7, X2, X9, X11", "c4 63 31 41 da 07"},
{"VDPPD $7, X2, X9, X2", "c4 e3 31 41 d2 07"},
{"VDPPS $7, (BX), X9, X11", "c4 63 31 40 1b 07"},
{"VDPPS $7, (BX), X9, X2", "c4 e3 31 40 13 07"},
{"VDPPS $7, (BX), Y15, Y11", "c4 63 05 40 1b 07"},
{"VDPPS $7, (BX), Y15, Y2", "c4 e3 05 40 13 07"},
{"VDPPS $7, (R11), X9, X11", "c4 43 31 40 1b 07"},
{"VDPPS $7, (R11), X9, X2", "c4 c3 31 40 13 07"},
{"VDPPS $7, (R11), Y15, Y11", "c4 43 05 40 1b 07"},
{"VDPPS $7, (R11), Y15, Y2", "c4 c3 05 40 13 07"},
{"VDPPS $7, X11, X9, X11", "c4 43 31 40 db 07"},
{"VDPPS $7, X11, X9, X2", "c4 c3 31 40 d3 07"},
{"VDPPS $7, X2, X9, X11", "c4 63 31 40 da 07"},
{"VDPPS $7, X2, X9, X2", "c4 e3 31 40 d2 07"},
{"VDPPS $7, Y11, Y15, Y11", "c4 43 05 40 db 07"},
{"VDPPS $7, Y11, Y15, Y2", "c4 c3 05 40 d3 07"},
{"VDPPS $7, Y2, Y15, Y11", "c4 63 05 40 da 07"},
{"VDPPS $7, Y2, Y15, Y2", "c4 e3 05 40 d2 07"},
{"VHADDPD (BX), X9, X11", "c5 31 7c 1b"},
{"VHADDPD (BX), X9, X2", "c5 b1 7c 13"},
{"VHADDPD (BX), Y15, Y11", "c5 05 7c 1b"},
{"VHADDPD (BX), Y15, Y2", "c5 85 7c 13"},
{"VHADDPD (R11), X9, X11", "c4 41 31 7c 1b"},
{"VHADDPD (R11), X9, X2", "c4 c1 31 7c 13"},
{"VHADDPD (R11), Y15, Y11", "c4 41 05 7c 1b"},
{"VHADDPD (R11), Y15, Y2", "c4 c1 05 7c 13"},
{"VHADDPD X11, X9, X11", "c4 41 31 7c db"},
{"VHADDPD X11, X9, X2", "c4 c1 31 7c d3"},
{"VHADDPD X2, X9, X11", "c5 31 7c da"},
{"VHADDPD X2, X9, X2", "c5 b1 7c d2"},
{"VHADDPD Y11, Y15, Y11", "c4 41 05 7c db"},
{"VHADDPD Y11, Y15, Y2", "c4 c1 05 7c d3"},
{"VHADDPD Y2, Y15, Y11", "c5 05 7c da"},
{"VHADDPD Y2, Y15, Y2", "c5 85 7c d2"},
{"VHADDPS (BX), X9, X11", "c5 33 7c 1b"},
{"VHADDPS (BX), X9, X2", "c5 b3 7c 13"},
{"VHADDPS (BX), Y15, Y11", "c5 07 7c 1b"},
{"VHADDPS (BX), Y15, Y2", "c5 87 7c 13"},
{"VHADDPS (R11), X9, X11", "c4 41 33 7c 1b"},
{"VHADDPS (R11), X9, X2", "c4 c1 33 7c 13"},
{"VHADDPS (R11), Y15, Y11", "c4 41 07 7c 1b"},
{"VHADDPS (R11), Y15, Y2", "c4 c1 07 7c 13"},
{"VHADDPS X11, X9, X11", "c4 41 33 7c db"},
{"VHADDPS X11, X9, X2", "c4 c1 33 7c d3"},
{"VHADDPS X2, X9, X11", "c5 33 7c da"},
{"VHADDPS X2, X9, X2", "c5 b3 7c d2"},
{"VHADDPS Y11, Y15, Y11", "c4 41 07 7c db"},
{"VHADDPS Y11, Y15, Y2", "c4 c1 07 7c d3"},
{"VHADDPS Y2, Y15, Y11", "c5 07 7c da"},
{"VHADDPS Y2, Y15, Y2", "c5 87 7c d2"},
{"VHSUBPD (BX), X9, X11", "c5 31 7d 1b"},
{"VHSUBPD (BX), X9, X2", "c5 b1 7d 13"},
{"VHSUBPD (BX), Y15, Y11", "c5 05 7d 1b"},
{"VHSUBPD (BX), Y15, Y2", "c5 85 7d 13"},
{"VHSUBPD (R11), X9, X11", "c4 41 31 7d 1b"},
{"VHSUBPD (R11), X9, X2", "c4 c1 31 7d 13"},
{"VHSUBPD (R11), Y15, Y11", "c4 41 05 7d 1b"},
{"VHSUBPD (R11), Y15, Y2", "c4 c1 05 7d 13"},
{"VHSUBPD X11, X9, X11", "c4 41 31 7d db"},
{"VHSUBPD X11, X9, X2", "c4 c1 31 7d d3"},
{"VHSUBPD X2, X9, X11", "c5 31 7d da"},
{"VHSUBPD X2, X9, X2", "c5 b1 7d d2"},
{"VHSUBPD Y11, Y15, Y11", "c4 41 05 7d db"},
{"VHSUBPD Y11, Y15, Y2", "c4 c1 05 7d d3"},
{"VHSUBPD Y2, Y15, Y11", "c5 05 7d da"},
{"VHSUBPD Y2, Y15, Y2", "c5 85 7d d2"},
{"VHSUBPS (BX), X9, X11", "c5 33 7d 1b"},
{"VHSUBPS (BX), X9, X2", "c5 b3 7d 13"},
{"VHSUBPS (BX), Y15, Y11", "c5 07 7d 1b"},
{"VHSUBPS (BX), Y15, Y2", "c5 87 7d 13"},
{"VHSUBPS (R11), X9, X11", "c4 41 33 7d 1b"},
{"VHSUBPS (R11), X9, X2", "c4 c1 33 7d 13"},
{"VHSUBPS (R11), Y15, Y11", "c4 41 07 7d 1b"},
{"VHSUBPS (R11), Y15, Y2", "c4 c1 07 7d 13"},
{"VHSUBPS X11, X9, X11", "c4 41 33 7d db"},
{"VHSUBPS X11, X9, X2", "c4 c1 33 7d d3"},
{"VHSUBPS X2, X9, X11", "c5 33 7d da"},
{"VHSUBPS X2, X9, X2", "c5 b3 7d d2"},
{"VHSUBPS Y11, Y15, Y11", "c4 41 07 7d db"},
{"VHSUBPS Y11, Y15, Y2", "c4 c1 07 7d d3"},
{"VHSUBPS Y2, Y15, Y11", "c5 07 7d da"},
{"VHSUBPS Y2, Y15, Y2", "c5 87 7d d2"},
{"VINSERTF128 $7, (BX), Y15, Y11", "c4 63 05 18 1b 07"},
{"VINSERTF128 $7, (BX), Y15, Y2", "c4 e3 05 18 13 07"},
{"VINSERTF128 $7, (R11), Y15, Y11", "c4 43 05 18 1b 07"},
{"VINSERTF128 $7, (R11), Y15, Y2", "c4 c3 05 18 13 07"},
{"VINSERTF128 $7, X11, Y15, Y11", "c4 43 05 18 db 07"},
{"VINSERTF128 $7, X11, Y15, Y2", "c4 c3 05 18 d3 07"},
{"VINSERTF128 $7, X2, Y15, Y11", "c4 63 05 18 da 07"},
{"VINSERTF128 $7, X2, Y15, Y2", "c4 e3 05 18 d2 07"},
{"VINSERTPS $7, (BX), X9, X11", "c4 63 31 21 1b 07"},
{"VINSERTPS $7, (BX), X9, X2", "c4 e3 31 21 13 07"},
{"VINSERTPS $7, (R11), X9, X11", "c4 43 31 21 1b 07"},
{"VINSERTPS $7, (R11), X9, X2", "c4 c3 31 21 13 07"},
{"VINSERTPS $7, X11, X9, X11", "c4 43 31 21 db 07"},
{"VINSERTPS $7, X11, X9, X2", "c4 c3 31 21 d3 07"},
{"VINSERTPS $7, X2, X9, X11", "c4 63 31 21 da 07"},
{"VINSERTPS $7, X2, X9, X2", "c4 e3 31 21 d2 07"},
{"VLDDQU (BX), X11", "c5 7b f0 1b"},
{"VLDDQU (BX), X2", "c5 fb f0 13"},
{"VLDDQU (BX), Y11", "c5 7f f0 1b"},
{"VLDDQU (BX), Y2", "c5 ff f0 13"},
{"VLDDQU (R11), X11", "c4 41 7b f0 1b"},
{"VLDDQU (R11), X2", "c4 c1 7b f0 13"},
{"VLDDQU (R11), Y11", "c4 41 7f f0 1b"},
{"VLDDQU (R11), Y2", "c4 c1 7f f0 13"},
{"VLDMXCSR (BX)", "c5 f8 ae 13"},
{"VLDMXCSR (R11)", "c4 c1 78 ae 13"},
{"VMASKMOVDQU X11, X11", "c4 41 79 f7 db"},
{"VMASKMOVDQU X11, X2", "c4 c1 79 f7 d3"},
{"VMASKMOVDQU X2, X11", "c5 79 f7 da"},
{"VMASKMOVDQU X2, X2", "c5 f9 f7 d2"},
{"VMASKMOVPD (BX), X9, X11", "c4 62 31 2d 1b"},
{"VMASKMOVPD (BX), X9, X2", "c4 e2 31 2d 13"},
{"VMASKMOVPD (BX), Y15, Y11", "c4 62 05 2d 1b"},
{"VMASKMOVPD (BX), Y15, Y2", "c4 e2 05 2d 13"},
{"VMASKMOVPD (R11), X9, X11", "c4 42 31 2d 1b"},
{"VMASKMOVPD (R11), X9, X2", "c4 c2 31 2d 13"},
{"VMASKMOVPD (R11), Y15, Y11", "c4 42 05 2d 1b"},
{"VMASKMOVPD (R11), Y15, Y2", "c4 c2 05 2d 13"},
{"VMASKMOVPD X11, X9, (BX)", "c4 62 31 2f 1b"},
{"VMASKMOVPD X11, X9, (R11)", "c4 42 31 2f 1b"},
{"VMASKMOVPD X2, X9, (BX)", "c4 e2 31 2f 13"},
{"VMASKMOVPD X2, X9, (R11)", "c4 c2 31 2f 13"},
{"VMASKMOVPD Y11, Y15, (BX)", "c4 62 05 2f 1b"},
{"VMASKMOVPD Y11, Y15, (R11)", "c4 42 05 2f 1b"},
{"VMASKMOVPD Y2, Y15, (BX)", "c4 e2 05 2f 13"},
{"VMASKMOVPD Y2, Y15, (R11)", "c4 c2 05 2f 13"},
{"VMASKMOVPS (BX), X9, X11", "c4 62 31 2c 1b"},
{"VMASKMOVPS (BX), X9, X2", "c4 e2 31 2c 13"},
{"VMASKMOVPS (BX), Y15, Y11", "c4 62 05 2c 1b"},
{"VMASKMOVPS (BX), Y15, Y2", "c4 e2 05 2c 13"},
{"VMASKMOVPS (R11), X9, X11", "c4 42 31 2c 1b"},
{"VMASKMOVPS (R11), X9, X2", "c4 c2 31 2c 13"},
{"VMASKMOVPS (R11), Y15, Y11", "c4 42 05 2c 1b"},
{"VMASKMOVPS (R11), Y15, Y2", "c4 c2 05 2c 13"},
{"VMASKMOVPS X11, X9, (BX)", "c4 62 31 2e 1b"},
{"VMASKMOVPS X11, X9, (R11)", "c4 42 31 2e 1b"},
{"VMASKMOVPS X2, X9, (BX)", "c4 e2 31 2e 13"},
{"VMASKMOVPS X2, X9, (R11)", "c4 c2 31 2e 13"},
{"VMASKMOVPS Y11, Y15, (BX)", "c4 62 05 2e 1b"},
{"VMASKMOVPS Y11, Y15, (R11)", "c4 42 05 2e 1b"},
{"VMASKMOVPS Y2, Y15, (BX)", "c4 e2 05 2e 13"},
{"VMASKMOVPS Y2, Y15, (R11)", "c4 c2 05 2e 13"},
{"VMOVHLPS X11, X9, X11", "c4 41 30 12 db"},
{"VMOVHLPS X11, X9, X2", "c4 c1 30 12 d3"},
{"VMOVHLPS X2, X9, X11", "c5 30 12 da"},
{"VMOVHLPS X2, X9, X2", "c5 b0 12 d2"},
{"VMOVLPS (BX), X9, X11", "c5 30 12 1b"},
{"VMOVLPS (BX), X9, X2", "c5 b0 12 13"},
{"VMOVLPS (R11), X9, X11", "c4 41 30 12 1b"},
{"VMOVLPS (R11), X9, X2", "c4 c1 30 12 13"},
{"VMOVLPS X11, (BX)", "c5 78 13 1b"},
{"VMOVLPS X11, (R11)", "c4 41 78 13 1b"},
{"VMOVLPS X2, (BX)", "c5 f8 13 13"},
{"VMOVLPS X2, (R11)", "c4 c1 78 13 13"},
{"VMOVMSKPD X11, DX", "c4 c1 79 50 d3"},
{"VMOVMSKPD X11, R11", "c4 41 79 50 db"},
{"VMOVMSKPD X2, DX", "c5 f9 50 d2"},
{"VMOVMSKPD X2, R11", "c5 79 50 da"},
{"VMOVMSKPD Y11, DX", "c4 c1 7d 50 d3"},
{"VMOVMSKPD Y11, R11", "c4 41 7d 50 db"},
{"VMOVMSKPD Y2, DX", "c5 fd 50 d2"},
{"VMOVMSKPD Y2, R11", "c5 7d 50 da"},
{"VMOVSD (BX), X11", "c5 7b 10 1b"},
{"VMOVSD (BX), X2", "c5 fb 10 13"},
{"VMOVSD (R11), X11", "c4 41 7b 10 1b"},
{"VMOVSD (R11), X2", "c4 c1 7b 10 13"},
{"VMOVSD X11, (BX)", "c5 7b 11 1b"},
{"VMOVSD X11, (R11)", "c4 41 7b 11 1b"},
{"VMOVSD X11, X9, X11", "c4 41 33 11 db"},
{"VMOVSD X11, X9, X2", "c5 33 11 da"},
{"VMOVSD X2, (BX)", "c5 fb 11 13"},
{"VMOVSD X2, (R11)", "c4 c1 7b 11 13"},
{"VMOVSD X2, X9, X11", "c4 c1 33 11 d3"},
{"VMOVSD X2, X9, X2", "c5 b3 11 d2"},
{"VMOVSS (BX), X11", "c5 7a 10 1b"},
{"VMOVSS (BX), X2", "c5 fa 10 13"},
{"VMOVSS (R11), X11", "c4 41 7a 10 1b"},
{"VMOVSS (R11), X2", "c4 c1 7a 10 13"},
{"VMOVSS X11, (BX)", "c5 7a 11 1b"},
{"VMOVSS X11, (R11)", "c4 41 7a 11 1b"},
{"VMOVSS X11, X9, X11", "c4 41 32 11 db"},
{"VMOVSS X11, X9, X2", "c5 32 11 da"},
{"VMOVSS X2, (BX)", "c5 fa 11 13"},
{"VMOVSS X2, (R11)", "c4 c1 7a 11 13"},
{"VMOVSS X2, X9, X11", "c4 c1 32 11 d3"},
{"VMOVSS X2, X9, X2", "c5 b2 11 d2"},
{"VMPSADBW $7, (BX), X9, X11", "c4 63 31 42 1b 07"},
{"VMPSADBW $7, (BX), X9, X2", "c4 e3 31 42 13 07"},
{"VMPSADBW $7, (BX), Y15, Y11", "c4 63 05 42 1b 07"},
{"VMPSADBW $7, (BX), Y15, Y2", "c4 e3 05 42 13 07"},
{"VMPSADBW $7, (R11), X9, X11", "c4 43 31 42 1b 07"},
{"VMPSADBW $7, (R11), X9, X2", "c4 c3 31 42 13 07"},
{"VMPSADBW $7, (R11), Y15, Y11", "c4 43 05 42 1b 07"},
{"VMPSADBW $7, (R11), Y15, Y2", "c4 c3 05 42 13 07"},
{"VMPSADBW $7, X11, X9, X11", "c4 43 31 42 db 07"},
{"VMPSADBW $7, X11, X9, X2", "c4 c3 31 42 d3 07"},
{"VMPSADBW $7, X2, X9, X11", "c4 63 31 42 da 07"},
{"VMPSADBW $7, X2, X9, X2", "c4 e3 31 42 d2 07"},
{"VMPSADBW $7, Y11, Y15, Y11", "c4 43 05 42 db 07"},
{"VMPSADBW $7, Y11, Y15, Y2", "c4 c3 05 42 d3 07"},
{"VMPSADBW $7, Y2, Y15, Y11", "c4 63 05 42 da 07"},
{"VMPSADBW $7, Y2, Y15, Y2", "c4 e3 05 42 d2 07"},
{"VPBLENDVB X12, (BX), X9, X11", "c4 63 31 4c 1b c0"},
{"VPBLENDVB X12, (BX), X9, X2", "c4 e3 31 4c 13 c0"},
{"VPBLENDVB X12, (R11), X9, X11", "c4 43 31 4c 1b c0"},
{"VPBLENDVB X12, (R11), X9, X2", "c4 c3 31 4c 13 c0"},
{"VPBLENDVB X12, X11, X9, X11", "c4 43 31 4c db c0"},
{"VPBLENDVB X12, X11, X9, X2", "c4 c3 31 4c d3 c0"},
{"VPBLENDVB X12, X2, X9, X11", "c4 63 31 4c da c0"},
{"VPBLENDVB X12, X2, X9, X2", "c4 e3 31 4c d2 c0"},
{"VPBLENDVB Y13, (BX), Y15, Y11", "c4 63 05 4c 1b d0"},
{"VPBLENDVB Y13, (BX), Y15, Y2", "c4 e3 05 4c 13 d0"},
{"VPBLENDVB Y13, (R11), Y15, Y11", "c4 43 05 4c 1b d0"},
{"VPBLENDVB Y13, (R11), Y15, Y2", "c4 c3 05 4c 13 d0"},
{"VPBLENDVB Y13, Y11, Y15, Y11", "c4 43 05 4c db d0"},
{"VPBLENDVB Y13, Y11, Y15, Y2", "c4 c3 05 4c d3 d0"},
{"VPBLENDVB Y13, Y2, Y15, Y11", "c4 63 05 4c da d0"},
{"VPBLENDVB Y13, Y2, Y15, Y2", "c4 e3 05 4c d2 d0"},
{"VPBLENDW $7, (BX), X9, X11", "c4 63 31 0e 1b 07"},
{"VPBLENDW $7, (BX), X9, X2", "c4 e3 31 0e 13 07"},
{"VPBLENDW $7, (BX), Y15, Y11", "c4 63 05 0e 1b 07"},
{"VPBLENDW $7, (BX), Y15, Y2", "c4 e3 05 0e 13 07"},
{"VPBLENDW $7, (R11), X9, X11", "c4 43 31 0e 1b 07"},
{"VPBLENDW $7, (R11), X9, X2", "c4 c3 31 0e 13 07"},
{"VPBLENDW $7, (R11), Y15, Y11", "c4 43 05 0e 1b 07"},
{"VPBLENDW $7, (R11), Y15, Y2", "c4 c3 05 0e 13 07"},
{"VPBLENDW $7, X11, X9, X11", "c4 43 31 0e db 07"},
{"VPBLENDW $7, X11, X9, X2", "c4 c3 31 0e d3 07"},
{"VPBLENDW $7, X2, X9, X11", "c4 63 31 0e da 07"},
{"VPBLENDW $7, X2, X9, X2", "c4 e3 31 0e d2 07"},
{"VPBLENDW $7, Y11, Y15, Y11", "c4 43 05 0e db 07"},
{"VPBLENDW $7, Y11, Y15, Y2", "c4 c3 05 0e d3 07"},
{"VPBLENDW $7, Y2, Y15, Y11", "c4 63 05 0e da 07"},
{"VPBLENDW $7, Y2, Y15, Y2", "c4 e3 05 0e d2 07"},
{"VPERMILPD $7, (BX), X11", "c4 63 79 05 1b 07"},
{"VPERMILPD $7, (BX), X2", "c4 e3 79 05 13 07"},
{"VPERMILPD $7, (BX), Y11", "c4 63 7d 05 1b 07"},
{"VPERMILPD $7, (BX), Y2", "c4 e3 7d 05 13 07"},
{"VPERMILPD $7, (R11), X11", "c4 43 79 05 1b 07"},
{"VPERMILPD $7, (R11), X2", "c4 c3 79 05 13 07"},
{"VPERMILPD $7, (R11), Y11", "c4 43 7d 05 1b 07"},
{"VPERMILPD $7, (R11), Y2", "c4 c3 7d 05 13 07"},
{"VPERMILPD $7, X11, X11", "c4 43 79 05 db 07"},
{"VPERMILPD $7, X11, X2", "c4 c3 79 05 d3 07"},
{"VPERMILPD $7, X2, X11", "c4 63 79 05 da 07"},
{"VPERMILPD $7, X2, X2", "c4 e3 79 05 d2 07"},
{"VPERMILPD $7, Y11, Y11", "c4 43 7d 05 db 07"},
{"VPERMILPD $7, Y11, Y2", "c4 c3 7d 05 d3 07"},
{"VPERMILPD $7, Y2, Y11", "c4 63 7d 05 da 07"},
{"VPERMILPD $7, Y2, Y2", "c4 e3 7d 05 d2 07"},
{"VPERMILPD (BX), X9, X11", "c4 62 31 0d 1b"},
{"VPERMILPD (BX), X9, X2", "c4 e2 31 0d 13"},
{"VPERMILPD (BX), Y15, Y11", "c4 62 05 0d 1b"},
{"VPERMILPD (BX), Y15, Y2", "c4 e2 05 0d 13"},
{"VPERMILPD (R11), X9, X11", "c4 42 31 0d 1b"},
{"VPERMILPD (R11), X9, X2", "c4 c2 31 0d 13"},
{"VPERMILPD (R11), Y15, Y11", "c4 42 05 0d 1b"},
{"VPERMILPD (R11), Y15, Y2", "c4 c2 05 0d 13"},
{"VPERMILPD X11, X9, X11", "c4 42 31 0d db"},
{"VPERMILPD X11, X9, X2", "c4 c2 31 0d d3"},
{"VPERMILPD X2, X9, X11", "c4 62 31 0d da"},
{"VPERMILPD X2, X9, X2", "c4 e2 31 0d d2"},
{"VPERMILPD Y11, Y15, Y11", "c4 42 05 0d db"},
{"VPERMILPD Y11, Y15, Y2", "c4 c2 05 0d d3"},
{"VPERMILPD Y2, Y15, Y11", "c4 62 05 0d da"},
{"VPERMILPD Y2, Y15, Y2", "c4 e2 05 0d d2"},
{"VPERMILPS $7, (BX), X11", "c4 63 79 04 1b 07"},
{"VPERMILPS $7, (BX), X2", "c4 e3 79 04 13 07"},
{"VPERMILPS $7, (BX), Y11", "c4 63 7d 04 1b 07"},
{"VPERMILPS $7, (BX), Y2", "c4 e3 7d 04 13 07"},
{"VPERMILPS $7, (R11), X11", "c4 43 79 04 1b 07"},
{"VPERMILPS $7, (R11), X2", "c4 c3 79 04 13 07"},
{"VPERMILPS $7, (R11), Y11", "c4 43 7d 04 1b 07"},
{"VPERMILPS $7, (R11), Y2", "c4 c3 7d 04 13 07"},
{"VPERMILPS $7, X11, X11", "c4 43 79 04 db 07"},
{"VPERMILPS $7, X11, X2", "c4 c3 79 04 d3 07"},
{"VPERMILPS $7, X2, X11", "c4 63 79 04 da 07"},
{"VPERMILPS $7, X2, X2", "c4 e3 79 04 d2 07"},
{"VPERMILPS $7, Y11, Y11", "c4 43 7d 04 db 07"},
{"VPERMILPS $7, Y11, Y2", "c4 c3 7d 04 d3 07"},
{"VPERMILPS $7, Y2, Y11", "c4 63 7d 04 da 07"},
{"VPERMILPS $7, Y2, Y2", "c4 e3 7d 04 d2 07"},
{"VPERMILPS (BX), X9, X11", "c4 62 31 0c 1b"},
{"VPERMILPS (BX), X9, X2", "c4 e2 31 0c 13"},
{"VPERMILPS (BX), Y15, Y11", "c4 62 05 0c 1b"},
{"VPERMILPS (BX), Y15, Y2", "c4 e2 05 0c 13"},
{"VPERMILPS (R11), X9, X11", "c4 42 31 0c 1b"},
{"VPERMILPS (R11), X9, X2", "c4 c2 31 0c 13"},
{"VPERMILPS (R11), Y15, Y11", "c4 42 05 0c 1b"},
{"VPERMILPS (R11), Y15, Y2", "c4 c2 05 0c 13"},
{"VPERMILPS X11, X9, X11", "c4 42 31 0c db"},
{"VPERMILPS X11, X9, X2", "c4 c2 31 0c d3"},
{"VPERMILPS X2, X9, X11", "c4 62 31 0c da"},
{"VPERMILPS X2, X9, X2", "c4 e2 31 0c d2"},
{"VPERMILPS Y11, Y15, Y11", "c4 42 05 0c db"},
{"VPERMILPS Y11, Y15, Y2", "c4 c2 05 0c d3"},
{"VPERMILPS Y2, Y15, Y11", "c4 62 05 0c da"},
{"VPERMILPS Y2, Y15, Y2", "c4 e2 05 0c d2"},
{"VPHADDD (BX), X9, X11", "c4 62 31 02 1b"},
{"VPHADDD (BX), X9, X2", "c4 e2 31 02 13"},
{"VPHADDD (BX), Y15, Y11", "c4 62 05 02 1b"},
{"VPHADDD (BX), Y15, Y2", "c4 e2 05 02 13"},
{"VPHADDD (R11), X9, X11", "c4 42 31 02 1b"},
{"VPHADDD (R11), X9, X2", "c4 c2 31 02 13"},
{"VPHADDD (R11), Y15, Y11", "c4 42 05 02 1b"},
{"VPHADDD (R11), Y15, Y2", "c4 c2 05 02 13"},
{"VPHADDD X11, X9, X11", "c4 42 31 02 db"},
{"VPHADDD X11, X9, X2", "c4 c2 31 02 d3"},
{"VPHADDD X2, X9, X11", "c4 62 31 02 da"},
{"VPHADDD X2, X9, X2", "c4 e2 31 02 d2"},
{"VPHADDD Y11, Y15, Y11", "c4 42 05 02 db"},
{"VPHADDD Y11, Y15, Y2", "c4 c2 05 02 d3"},
{"VPHADDD Y2, Y15, Y11", "c4 62 05 02 da"},
{"VPHADDD Y2, Y15, Y2", "c4 e2 05 02 d2"},
{"VPHADDSW (BX), X9, X11", "c4 62 31 03 1b"},
{"VPHADDSW (BX), X9, X2", "c4 e2 31 03 13"},
{"VPHADDSW (BX), Y15, Y11", "c4 62 05 03 1b"},
{"VPHADDSW (BX), Y15, Y2", "c4 e2 05 03 13"},
{"VPHADDSW (R11), X9, X11", "c4 42 31 03 1b"},
{"VPHADDSW (R11), X9, X2", "c4 c2 31 03 13"},
{"VPHADDSW (R11), Y15, Y11", "c4 42 05 03 1b"},
{"VPHADDSW (R11), Y15, Y2", "c4 c2 05 03 13"},
{"VPHADDSW X11, X9, X11", "c4 42 31 03 db"},
{"VPHADDSW X11, X9, X2", "c4 c2 31 03 d3"},
{"VPHADDSW X2, X9, X11", "c4 62 31 03 da"},
{"VPHADDSW X2, X9, X2", "c4 e2 31 03 d2"},
{"VPHADDSW Y11, Y15, Y11", "c4 42 05 03 db"},
{"VPHADDSW Y11, Y15, Y2", "c4 c2 05 03 d3"},
{"VPHADDSW Y2, Y15, Y11", "c4 62 05 03 da"},
{"VPHADDSW Y2, Y15, Y2", "c4 e2 05 03 d2"},
{"VPHADDW (BX), X9, X11", "c4 62 31 01 1b"},
{"VPHADDW (BX), X9, X2", "c4 e2 31 01 13"},
{"VPHADDW (BX), Y15, Y11", "c4 62 05 01 1b"},
{"VPHADDW (BX), Y15, Y2", "c4 e2 05 01 13"},
{"VPHADDW (R11), X9, X11", "c4 42 31 01 1b"},
{"VPHADDW (R11), X9, X2", "c4 c2 31 01 13"},
{"VPHADDW (R11), Y15, Y11", "c4 42 05 01 1b"},
{"VPHADDW (R11), Y15, Y2", "c4 c2 05 01 13"},
{"VPHADDW X11, X9, X11", "c4 42 31 01 db"},
{"VPHADDW X11, X9, X2", "c4 c2 31 01 d3"},
{"VPHADDW X2, X9, X11", "c4 62 31 01 da"},
{"VPHADDW X2, X9, X2", "c4 e2 31 01 d2"},
{"VPHADDW Y11, Y15, Y11", "c4 42 05 01 db"},
{"VPHADDW Y11, Y15, Y2", "c4 c2 05 01 d3"},
{"VPHADDW Y2, Y15, Y11", "c4 62 05 01 da"},
{"VPHADDW Y2, Y15, Y2", "c4 e2 05 01 d2"},
{"VPHMINPOSUW (BX), X11", "c4 62 79 41 1b"},
{"VPHMINPOSUW (BX), X2", "c4 e2 79 41 13"},
{"VPHMINPOSUW (R11), X11", "c4 42 79 41 1b"},
{"VPHMINPOSUW (R11), X2", "c4 c2 79 41 13"},
{"VPHMINPOSUW X11, X11", "c4 42 79 41 db"},
{"VPHMINPOSUW X11, X2", "c4 c2 79 41 d3"},
{"VPHMINPOSUW X2, X11", "c4 62 79 41 da"},
{"VPHMINPOSUW X2, X2", "c4 e2 79 41 d2"},
{"VPHSUBD (BX), X9, X11", "c4 62 31 06 1b"},
{"VPHSUBD (BX), X9, X2", "c4 e2 31 06 13"},
{"VPHSUBD (BX), Y15, Y11", "c4 62 05 06 1b"},
{"VPHSUBD (BX), Y15, Y2", "c4 e2 05 06 13"},
{"VPHSUBD (R11), X9, X11", "c4 42 31 06 1b"},
{"VPHSUBD (R11), X9, X2", "c4 c2 31 06 13"},
{"VPHSUBD (R11), Y15, Y11", "c4 42 05 06 1b"},
{"VPHSUBD (R11), Y15, Y2", "c4 c2 05 06 13"},
{"VPHSUBD X11, X9, X11", "c4 42 31 06 db"},
{"VPHSUBD X11, X9, X2", "c4 c2 31 06 d3"},
{"VPHSUBD X2, X9, X11", "c4 62 31 06 da"},
{"VPHSUBD X2, X9, X2", "c4 e2 31 06 d2"},
{"VPHSUBD Y11, Y15, Y11", "c4 42 05 06 db"},
{"VPHSUBD Y11, Y15, Y2", "c4 c2 05 06 d3"},
{"VPHSUBD Y2, Y15, Y11", "c4 62 05 06 da"},
{"VPHSUBD Y2, Y15, Y2", "c4 e2 05 06 d2"},
{"VPHSUBSW (BX), X9, X11", "c4 62 31 07 1b"},
{"VPHSUBSW (BX), X9, X2", "c4 e2 31 07 13"},
{"VPHSUBSW (BX), Y15, Y11", "c4 62 05 07 1b"},
{"VPHSUBSW (BX), Y15, Y2", "c4 e2 05 07 13"},
{"VPHSUBSW (R11), X9, X11", "c4 42 31 07 1b"},
{"VPHSUBSW (R11), X9, X2", "c4 c2 31 07 13"},
{"VPHSUBSW (R11), Y15, Y11", "c4 42 05 07 1b"},
{"VPHSUBSW (R11), Y15, Y2", "c4 c2 05 07 13"},
{"VPHSUBSW X11, X9, X11", "c4 42 31 07 db"},
{"VPHSUBSW X11, X9, X2", "c4 c2 31 07 d3"},
{"VPHSUBSW X2, X9, X11", "c4 62 31 07 da"},
{"VPHSUBSW X2, X9, X2", "c4 e2 31 07 d2"},
{"VPHSUBSW Y11, Y15, Y11", "c4 42 05 07 db"},
{"VPHSUBSW Y11, Y15, Y2", "c4 c2 05 07 d3"},
{"VPHSUBSW Y2, Y15, Y11", "c4 62 05 07 da"},
{"VPHSUBSW Y2, Y15, Y2", "c4 e2 05 07 d2"},
{"VPHSUBW (BX), X9, X11", "c4 62 31 05 1b"},
{"VPHSUBW (BX), X9, X2", "c4 e2 31 05 13"},
{"VPHSUBW (BX), Y15, Y11", "c4 62 05 05 1b"},
{"VPHSUBW (BX), Y15, Y2", "c4 e2 05 05 13"},
{"VPHSUBW (R11), X9, X11", "c4 42 31 05 1b"},
{"VPHSUBW (R11), X9, X2", "c4 c2 31 05 13"},
{"VPHSUBW (R11), Y15, Y11", "c4 42 05 05 1b"},
{"VPHSUBW (R11), Y15, Y2", "c4 c2 05 05 13"},
{"VPHSUBW X11, X9, X11", "c4 42 31 05 db"},
{"VPHSUBW X11, X9, X2", "c4 c2 31 05 d3"},
{"VPHSUBW X2, X9, X11", "c4 62 31 05 da"},
{"VPHSUBW X2, X9, X2", "c4 e2 31 05 d2"},
{"VPHSUBW Y11, Y15, Y11", "c4 42 05 05 db"},
{"VPHSUBW Y11, Y15, Y2", "c4 c2 05 05 d3"},
{"VPHSUBW Y2, Y15, Y11", "c4 62 05 05 da"},
{"VPHSUBW Y2, Y15, Y2", "c4 e2 05 05 d2"},
{"VPINSRB $7, (BX), X9, X11", "c4 63 31 20 1b 07"},
{"VPINSRB $7, (BX), X9, X2", "c4 e3 31 20 13 07"},
{"VPINSRB $7, (R11), X9, X11", "c4 43 31 20 1b 07"},
{"VPINSRB $7, (R11), X9, X2", "c4 c3 31 20 13 07"},
{"VPINSRB $7, DX, X9, X11", "c4 63 31 20 da 07"},
{"VPINSRB $7, DX, X9, X2", "c4 e3 31 20 d2 07"},
{"VPINSRB $7, R11, X9, X11", "c4 43 31 20 db 07"},
{"VPINSRB $7, R11, X9, X2", "c4 c3 31 20 d3 07"},
{"VPINSRW $7, (BX), X9, X11", "c5 31 c4 1b 07"},
{"VPINSRW $7, (BX), X9, X2", "c5 b1 c4 13 07"},
{"VPINSRW $7, (R11), X9, X11", "c4 41 31 c4 1b 07"},
{"VPINSRW $7, (R11), X9, X2", "c4 c1 31 c4 13 07"},
{"VPINSRW $7, DX, X9, X11", "c5 31 c4 da 07"},
{"VPINSRW $7, DX, X9, X2", "c5 b1 c4 d2 07"},
{"VPINSRW $7, R11, X9, X11", "c4 41 31 c4 db 07"},
{"VPINSRW $7, R11, X9, X2", "c4 c1 31 c4 d3 07"},
{"VPMASKMOVD (BX), X9, X11", "c4 62 31 8c 1b"},
{"VPMASKMOVD (BX), X9, X2", "c4 e2 31 8c 13"},
{"VPMASKMOVD (BX), Y15, Y11", "c4 62 05 8c 1b"},
{"VPMASKMOVD (BX), Y15, Y2", "c4 e2 05 8c 13"},
{"VPMASKMOVD (R11), X9, X11", "c4 42 31 8c 1b"},
{"VPMASKMOVD (R11), X9, X2", "c4 c2 31 8c 13"},
{"VPMASKMOVD (R11), Y15, Y11", "c4 42 05 8c 1b"},
{"VPMASKMOVD (R11), Y15, Y2", "c4 c2 05 8c 13"},
{"VPMASKMOVD X11, X9, (BX)", "c4 62 31 8e 1b"},
{"VPMASKMOVD X11, X9, (R11)", "c4 42 31 8e 1b"},
{"VPMASKMOVD X2, X9, (BX)", "c4 e2 31 8e 13"},
{"VPMASKMOVD X2, X9, (R11)", "c4 c2 31 8e 13"},
{"VPMASKMOVD Y11, Y15, (BX)", "c4 62 05 8e 1b"},
{"VPMASKMOVD Y11, Y15, (R11)", "c4 42 05 8e 1b"},
{"VPMASKMOVD Y2, Y15, (BX)", "c4 e2 05 8e 13"},
{"VPMASKMOVD Y2, Y15, (R11)", "c4 c2 05 8e 13"},
{"VPMASKMOVQ (BX), X9, X11", "c4 62 b1 8c 1b"},
{"VPMASKMOVQ (BX), X9, X2", "c4 e2 b1 8c 13"},
{"VPMASKMOVQ (BX), Y15, Y11", "c4 62 85 8c 1b"},
{"VPMASKMOVQ (BX), Y15, Y2", "c4 e2 85 8c 13"},
{"VPMASKMOVQ (R11), X9, X11", "c4 42 b1 8c 1b"},
{"VPMASKMOVQ (R11), X9, X2", "c4 c2 b1 8c 13"},
{"VPMASKMOVQ (R11), Y15, Y11", "c4 42 85 8c 1b"},
{"VPMASKMOVQ (R11), Y15, Y2", "c4 c2 85 8c 13"},
{"VPMASKMOVQ X11, X9, (BX)", "c4 62 b1 8e 1b"},
{"VPMASKMOVQ X11, X9, (R11)", "c4 42 b1 8e 1b"},
{"VPMASKMOVQ X2, X9, (BX)", "c4 e2 b1 8e 13"},
{"VPMASKMOVQ X2, X9, (R11)", "c4 c2 b1 8e 13"},
{"VPMASKMOVQ Y11, Y15, (BX)", "c4 62 85 8e 1b"},
{"VPMASKMOVQ Y11, Y15, (R11)", "c4 42 85 8e 1b"},
{"VPMASKMOVQ Y2, Y15, (BX)", "c4 e2 85 8e 13"},
{"VPMASKMOVQ Y2, Y15, (R11)", "c4 c2 85 8e 13"},
{"VPSIGNB (BX), X9, X11", "c4 62 31 08 1b"},
{"VPSIGNB (BX), X9, X2", "c4 e2 31 08 13"},
{"VPSIGNB (BX), Y15, Y11", "c4 62 05 08 1b"},
{"VPSIGNB (BX), Y15, Y2", "c4 e2 05 08 13"},
{"VPSIGNB (R11), X9, X11", "c4 42 31 08 1b"},
{"VPSIGNB (R11), X9, X2", "c4 c2 31 08 13"},
{"VPSIGNB (R11), Y15, Y11", "c4 42 05 08 1b"},
{"VPSIGNB (R11), Y15, Y2", "c4 c2 05 08 13"},
{"VPSIGNB X11, X9, X11", "c4 42 31 08 db"},
{"VPSIGNB X11, X9, X2", "c4 c2 31 08 d3"},
{"VPSIGNB X2, X9, X11", "c4 62 31 08 da"},
{"VPSIGNB X2, X9, X2", "c4 e2 31 08 d2"},
{"VPSIGNB Y11, Y15, Y11", "c4 42 05 08 db"},
{"VPSIGNB Y11, Y15, Y2", "c4 c2 05 08 d3"},
{"VPSIGNB Y2, Y15, Y11", "c4 62 05 08 da"},
{"VPSIGNB Y2, Y15, Y2", "c4 e2 05 08 d2"},
{"VPSIGND (BX), X9, X11", "c4 62 31 0a 1b"},
{"VPSIGND (BX), X9, X2", "c4 e2 31 0a 13"},
{"VPSIGND (BX), Y15, Y11", "c4 62 05 0a 1b"},
{"VPSIGND (BX), Y15, Y2", "c4 e2 05 0a 13"},
{"VPSIGND (R11), X9, X11", "c4 42 31 0a 1b"},
{"VPSIGND (R11), X9, X2", "c4 c2 31 0a 13"},
{"VPSIGND (R11), Y15, Y11", "c4 42 05 0a 1b"},
{"VPSIGND (R11), Y15, Y2", "c4 c2 05 0a 13"},
{"VPSIGND X11, X9, X11", "c4 42 31 0a db"},
{"VPSIGND X11, X9, X2", "c4 c2 31 0a d3"},
{"VPSIGND X2, X9, X11", "c4 62 31 0a da"},
{"VPSIGND X2, X9, X2", "c4 e2 31 0a d2"},
{"VPSIGND Y11, Y15, Y11", "c4 42 05 0a db"},
{"VPSIGND Y11, Y15, Y2", "c4 c2 05 0a d3"},
{"VPSIGND Y2, Y15, Y11", "c4 62 05 0a da"},
{"VPSIGND Y2, Y15, Y2", "c4 e2 05 0a d2"},
{"VPSIGNW (BX), X9, X11", "c4 62 31 09 1b"},
{"VPSIGNW (BX), X9, X2", "c4 e2 31 09 13"},
{"VPSIGNW (BX), Y15, Y11", "c4 62 05 09 1b"},
{"VPSIGNW (BX), Y15, Y2", "c4 e2 05 09 13"},
{"VPSIGNW (R11), X9, X11", "c4 42 31 09 1b"},
{"VPSIGNW (R11), X9, X2", "c4 c2 31 09 13"},
{"VPSIGNW (R11), Y15, Y11", "c4 42 05 09 1b"},
{"VPSIGNW (R11), Y15, Y2", "c4 c2 05 09 13"},
{"VPSIGNW X11, X9, X11", "c4 42 31 09 db"},
{"VPSIGNW X11, X9, X2", "c4 c2 31 09 d3"},
{"VPSIGNW X2, X9, X11", "c4 62 31 09 da"},
{"VPSIGNW X2, X9, X2", "c4 e2 31 09 d2"},
{"VPSIGNW Y11, Y15, Y11", "c4 42 05 09 db"},
{"VPSIGNW Y11, Y15, Y2", "c4 c2 05 09 d3"},
{"VPSIGNW Y2, Y15, Y11", "c4 62 05 09 da"},
{"VPSIGNW Y2, Y15, Y2", "c4 e2 05 09 d2"},
{"VPSLLW $7, X11, X9", "c4 c1 31 71 f3 07"},
{"VPSLLW $7, X2, X9", "c5 b1 71 f2 07"},
{"VPSLLW $7, Y11, Y15", "c4 c1 05 71 f3 07"},
{"VPSLLW $7, Y2, Y15", "c5 85 71 f2 07"},
{"VPSLLW (BX), X9, X11", "c5 31 f1 1b"},
{"VPSLLW (BX), X9, X2", "c5 b1 f1 13"},
{"VPSLLW (BX), Y15, Y11", "c5 05 f1 1b"},
{"VPSLLW (BX), Y15, Y2", "c5 85 f1 13"},
{"VPSLLW (R11), X9, X11", "c4 41 31 f1 1b"},
{"VPSLLW (R11), X9, X2", "c4 c1 31 f1 13"},
{"VPSLLW (R11), Y15, Y11", "c4 41 05 f1 1b"},
{"VPSLLW (R11), Y15, Y2", "c4 c1 05 f1 13"},
{"VPSLLW X11, X9, X11", "c4 41 31 f1 db"},
{"VPSLLW X11, X9, X2", "c4 c1 31 f1 d3"},
{"VPSLLW X11, Y15, Y11", "c4 41 05 f1 db"},
{"VPSLLW X11, Y15, Y2", "c4 c1 05 f1 d3"},
{"VPSLLW X2, X9, X11", "c5 31 f1 da"},
{"VPSLLW X2, X9, X2", "c5 b1 f1 d2"},
{"VPSLLW X2, Y15, Y11", "c5 05 f1 da"},
{"VPSLLW X2, Y15, Y2", "c5 85 f1 d2"},
{"VPSRAW $7, X11, X9", "c4 c1 31 71 e3 07"},
{"VPSRAW $7, X2, X9", "c5 b1 71 e2 07"},
{"VPSRAW $7, Y11, Y15", "c4 c1 05 71 e3 07"},
{"VPSRAW $7, Y2, Y15", "c5 85 71 e2 07"},
{"VPSRAW (BX), X9, X11", "c5 31 e1 1b"},
{"VPSRAW (BX), X9, X2", "c5 b1 e1 13"},
{"VPSRAW (BX), Y15, Y11", "c5 05 e1 1b"},
{"VPSRAW (BX), Y15, Y2", "c5 85 e1 13"},
{"VPSRAW (R11), X9, X11", "c4 41 31 e1 1b"},
{"VPSRAW (R11), X9, X2", "c4 c1 31 e1 13"},
{"VPSRAW (R11), Y15, Y11", "c4 41 05 e1 1b"},
{"VPSRAW (R11), Y15, Y2", "c4 c1 05 e1 13"},
{"VPSRAW X11, X9, X11", "c4 41 31 e1 db"},
{"VPSRAW X11, X9, X2", "c4 c1 31 e1 d3"},
{"VPSRAW X11, Y15, Y11", "c4 41 05 e1 db"},
{"VPSRAW X11, Y15, Y2", "c4 c1 05 e1 d3"},
{"VPSRAW X2, X9, X11", "c5 31 e1 da"},
{"VPSRAW X2, X9, X2", "c5 b1 e1 d2"},
{"VPSRAW X2, Y15, Y11", "c5 05 e1 da"},
{"VPSRAW X2, Y15, Y2", "c5 85 e1 d2"},
{"VPSRLW $7, X11, X9", "c4 c1 31 71 d3 07"},
{"VPSRLW $7, X2, X9", "c5 b1 71 d2 07"},
{"VPSRLW $7, Y11, Y15", "c4 c1 05 71 d3 07"},
{"VPSRLW $7, Y2, Y15", "c5 85 71 d2 07"},
{"VPSRLW (BX), X9, X11", "c5 31 d1 1b"},
{"VPSRLW (BX), X9, X2", "c5 b1 d1 13"},
{"VPSRLW (BX), Y15, Y11", "c5 05 d1 1b"},
{"VPSRLW (BX), Y15, Y2", "c5 85 d1 13"},
{"VPSRLW (R11), X9, X11", "c4 41 31 d1 1b"},
{"VPSRLW (R11), X9, X2", "c4 c1 31 d1 13"},
{"VPSRLW (R11), Y15, Y11", "c4 41 05 d1 1b"},
{"VPSRLW (R11), Y15, Y2", "c4 c1 05 d1 13"},
{"VPSRLW X11, X9, X11", "c4 41 31 d1 db"},
{"VPSRLW X11, X9, X2", "c4 c1 31 d1 d3"},
{"VPSRLW X11, Y15, Y11", "c4 41 05 d1 db"},
{"VPSRLW X11, Y15, Y2", "c4 c1 05 d1 d3"},
{"VPSRLW X2, X9, X11", "c5 31 d1 da"},
{"VPSRLW X2, X9, X2", "c5 b1 d1 d2"},
{"VPSRLW X2, Y15, Y11", "c5 05 d1 da"},
{"VPSRLW X2, Y15, Y2", "c5 85 d1 d2"},
{"VRCPPS (BX), X11", "c5 78 53 1b"},
{"VRCPPS (BX), X2", "c5 f8 53 13"},
{"VRCPPS (BX), Y11", "c5 7c 53 1b"},
{"VRCPPS (BX), Y2", "c5 fc 53 13"},
{"VRCPPS (R11), X11", "c4 41 78 53 1b"},
{"VRCPPS (R11), X2", "c4 c1 78 53 13"},
{"VRCPPS (R11), Y11", "c4 41 7c 53 1b"},
{"VRCPPS (R11), Y2", "c4 c1 7c 53 13"},
{"VRCPPS X11, X11", "c4 41 78 53 db"},
{"VRCPPS X11, X2", "c4 c1 78 53 d3"},
{"VRCPPS X2, X11", "c5 78 53 da"},
{"VRCPPS X2, X2", "c5 f8 53 d2"},
{"VRCPPS Y11, Y11", "c4 41 7c 53 db"},
{"VRCPPS Y11, Y2", "c4 c1 7c 53 d3"},
{"VRCPPS Y2, Y11", "c5 7c 53 da"},
{"VRCPPS Y2, Y2", "c5 fc 53 d2"},
{"VRCPSS (BX), X9, X11", "c5 32 53 1b"},
{"VRCPSS (BX), X9, X2", "c5 b2 53 13"},
{"VRCPSS (R11), X9, X11", "c4 41 32 53 1b"},
{"VRCPSS (R11), X9, X2", "c4 c1 32 53 13"},
{"VRCPSS X11, X9, X11", "c4 41 32 53 db"},
{"VRCPSS X11, X9, X2", "c4 c1 32 53 d3"},
{"VRCPSS X2, X9, X11", "c5 32 53 da"},
{"VRCPSS X2, X9, X2", "c5 b2 53 d2"},
{"VROUNDSD $7, (BX), X9, X11", "c4 63 31 0b 1b 07"},
{"VROUNDSD $7, (BX), X9, X2", "c4 e3 31 0b 13 07"},
{"VROUNDSD $7, (R11), X9, X11", "c4 43 31 0b 1b 07"},
{"VROUNDSD $7, (R11), X9, X2", "c4 c3 31 0b 13 07"},
{"VROUNDSD $7, X11, X9, X11", "c4 43 31 0b db 07"},
{"VROUNDSD $7, X11, X9, X2", "c4 c3 31 0b d3 07"},
{"VROUNDSD $7, X2, X9, X11", "c4 63 31 0b da 07"},
{"VROUNDSD $7, X2, X9, X2", "c4 e3 31 0b d2 07"},
{"VROUNDSS $7, (BX), X9, X11", "c4 63 31 0a 1b 07"},
{"VROUNDSS $7, (BX), X9, X2", "c4 e3 31 0a 13 07"},
{"VROUNDSS $7, (R11), X9, X11", "c4 43 31 0a 1b 07"},
{"VROUNDSS $7, (R11), X9, X2", "c4 c3 31 0a 13 07"},
{"VROUNDSS $7, X11, X9, X11", "c4 43 31 0a db 07"},
{"VROUNDSS $7, X11, X9, X2", "c4 c3 31 0a d3 07"},
{"VROUNDSS $7, X2, X9, X11", "c4 63 31 0a da 07"},
{"VROUNDSS $7, X2, X9, X2", "c4 e3 31 0a d2 07"},
{"VRSQRTPS (BX), X11", "c5 78 52 1b"},
{"VRSQRTPS (BX), X2", "c5 f8 52 13"},
{"VRSQRTPS (BX), Y11", "c5 7c 52 1b"},
{"VRSQRTPS (BX), Y2", "c5 fc 52 13"},
{"VRSQRTPS (R11), X11", "c4 41 78 52 1b"},
{"VRSQRTPS (R11), X2", "c4 c1 78 52 13"},
{"VRSQRTPS (R11), Y11", "c4 41 7c 52 1b"},
{"VRSQRTPS (R11), Y2", "c4 c1 7c 52 13"},
{"VRSQRTPS X11, X11", "c4 41 78 52 db"},
{"VRSQRTPS X11, X2", "c4 c1 78 52 d3"},
{"VRSQRTPS X2, X11", "c5 78 52 da"},
{"VRSQRTPS X2, X2", "c5 f8 52 d2"},
{"VRSQRTPS Y11, Y11", "c4 41 7c 52 db"},
{"VRSQRTPS Y11, Y2", "c4 c1 7c 52 d3"},
{"VRSQRTPS Y2, Y11", "c5 7c 52 da"},
{"VRSQRTPS Y2, Y2", "c5 fc 52 d2"},
{"VRSQRTSS (BX), X9, X11", "c5 32 52 1b"},
{"VRSQRTSS (BX), X9, X2", "c5 b2 52 13"},
{"VRSQRTSS (R11), X9, X11", "c4 41 32 52 1b"},
{"VRSQRTSS (R11), X9, X2", "c4 c1 32 52 13"},
{"VRSQRTSS X11, X9, X11", "c4 41 32 52 db"},
{"VRSQRTSS X11, X9, X2", "c4 c1 32 52 d3"},
{"VRSQRTSS X2, X9, X11", "c5 32 52 da"},
{"VRSQRTSS X2, X9, X2", "c5 b2 52 d2"},
{"VSTMXCSR (BX)", "c5 f8 ae 1b"},
{"VSTMXCSR (R11)", "c4 c1 78 ae 1b"},
{"VTESTPD (BX), X11", "c4 62 79 0f 1b"},
{"VTESTPD (BX), X2", "c4 e2 79 0f 13"},
{"VTESTPD (BX), Y11", "c4 62 7d 0f 1b"},
{"VTESTPD (BX), Y2", "c4 e2 7d 0f 13"},
{"VTESTPD (R11), X11", "c4 42 79 0f 1b"},
{"VTESTPD (R11), X2", "c4 c2 79 0f 13"},
{"VTESTPD (R11), Y11", "c4 42 7d 0f 1b"},
{"VTESTPD (R11), Y2", "c4 c2 7d 0f 13"},
{"VTESTPD X11, X11", "c4 42 79 0f db"},
{"VTESTPD X11, X2", "c4 c2 79 0f d3"},
{"VTESTPD X2, X11", "c4 62 79 0f da"},
{"VTESTPD X2, X2", "c4 e2 79 0f d2"},
{"VTESTPD Y11, Y11", "c4 42 7d 0f db"},
{"VTESTPD Y11, Y2", "c4 c2 7d 0f d3"},
{"VTESTPD Y2, Y11", "c4 62 7d 0f da"},
{"VTESTPD Y2, Y2", "c4 e2 7d 0f d2"},
{"VTESTPS (BX), X11", "c4 62 79 0e 1b"},
{"VTESTPS (BX), X2", "c4 e2 79 0e 13"},
{"VTESTPS (BX), Y11", "c4 62 7d 0e 1b"},
{"VTESTPS (BX), Y2", "c4 e2 7d 0e 13"},
{"VTESTPS (R11), X11", "c4 42 79 0e 1b"},
{"VTESTPS (R11), X2", "c4 c2 79 0e 13"},
{"VTESTPS (R11), Y11", "c4 42 7d 0e 1b"},
{"VTESTPS (R11), Y2", "c4 c2 7d 0e 13"},
{"VTESTPS X11, X11", "c4 42 79 0e db"},
{"VTESTPS X11, X2", "c4 c2 79 0e d3"},
{"VTESTPS X2, X11", "c4 62 79 0e da"},
{"VTESTPS X2, X2", "c4 e2 79 0e d2"},
{"VTESTPS Y11, Y11", "c4 42 7d 0e db"},
{"VTESTPS Y11, Y2", "c4 c2 7d 0e d3"},
{"VTESTPS Y2, Y11", "c4 62 7d 0e da"},
{"VTESTPS Y2, Y2", "c4 e2 7d 0e d2"},
}
// TestAmd64VexCorpus assembles every corpus line and requires the same bytes
// go tool asm emits for it.
func TestAmd64VexCorpus(t *testing.T) {
for _, tc := range amd64VexCorpus {
fn := firstText(t, "TEXT ·p(SB), 4, $0\n\t"+tc.line+"\n")
code, _, err := Assemble(fn)
if err != nil {
t.Errorf("%s: %v", tc.line, err)
continue
}
if got := hexBytes(code); got != tc.want {
t.Errorf("%s: got %s, want %s", tc.line, got, tc.want)
}
}
}
+835
View File
@@ -0,0 +1,835 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"slices"
"strings"
"testing"
)
// amd64VexParityCorpus holds every line the Go toolchain's own
// amd64enc.s carries for the mnemonics the full-file byte sweep
// flagged: the VEX forms the toolchain prefers for plain vector
// registers, the compare-with-predicate family and the byte-level
// corrections (PEXTRW's GPR fields, PUSHW and POPW widths, the double
// shift's high register field, VCOMISS's prefix, RORX's destination
// bit, the variable shifts' consolidated rows). Each entry lists the
// byte strings go tool asm emits or accepts for the line; a line
// passes when the assembled bytes match one of them.
var amd64VexParityCorpus = []struct {
line string
want string
}{
{"PEXTRW $7, X11, (BX)", "66 44 0f 3a 15 1b 07"},
{"PEXTRW $7, X11, (R11)", "66 45 0f 3a 15 1b 07"},
{"PEXTRW $7, X11, DX", "66 41 0f c5 d3 07|66 44 0f 3a 15 da 07"},
{"PEXTRW $7, X11, R11", "66 45 0f c5 db 07|66 45 0f 3a 15 db 07"},
{"PEXTRW $7, X2, (BX)", "66 0f 3a 15 13 07"},
{"PEXTRW $7, X2, (R11)", "66 41 0f 3a 15 13 07"},
{"PEXTRW $7, X2, DX", "66 0f c5 d2 07|66 0f 3a 15 d2 07"},
{"PEXTRW $7, X2, R11", "66 44 0f c5 da 07|66 41 0f 3a 15 d3 07"},
{"POPW (BX)", "66 8f 03"},
{"POPW (R11)", "66 41 8f 03"},
{"POPW DX", "66 8f c2|66 5a"},
{"POPW R11", "66 41 8f c3|66 41 5b"},
{"PUSHW $61731", "66 68 23 f1"},
{"PUSHW (BX)", "66 ff 33"},
{"PUSHW (R11)", "66 41 ff 33"},
{"PUSHW DX", "66 ff f2|66 52"},
{"PUSHW R11", "66 41 ff f3|66 41 53"},
{"RORXL $7, (BX), DX", "c4 e3 7b f0 13 07"},
{"RORXL $7, (BX), R11", "c4 63 7b f0 1b 07"},
{"RORXL $7, (R11), DX", "c4 c3 7b f0 13 07"},
{"RORXL $7, (R11), R11", "c4 43 7b f0 1b 07"},
{"RORXL $7, DX, DX", "c4 e3 7b f0 d2 07"},
{"RORXL $7, DX, R11", "c4 63 7b f0 da 07"},
{"RORXL $7, R11, DX", "c4 c3 7b f0 d3 07"},
{"RORXL $7, R11, R11", "c4 43 7b f0 db 07"},
{"RORXQ $7, (BX), DX", "c4 e3 fb f0 13 07"},
{"RORXQ $7, (BX), R11", "c4 63 fb f0 1b 07"},
{"RORXQ $7, (R11), DX", "c4 c3 fb f0 13 07"},
{"RORXQ $7, (R11), R11", "c4 43 fb f0 1b 07"},
{"RORXQ $7, DX, DX", "c4 e3 fb f0 d2 07"},
{"RORXQ $7, DX, R11", "c4 63 fb f0 da 07"},
{"RORXQ $7, R11, DX", "c4 c3 fb f0 d3 07"},
{"RORXQ $7, R11, R11", "c4 43 fb f0 db 07"},
{"SHLL $1, (BX)", "d1 23"},
{"SHLL $1, (R11)", "41 d1 23"},
{"SHLL $1, DX", "d1 e2"},
{"SHLL $1, R11", "41 d1 e3"},
{"SHLL $7, (BX)", "c1 23 07"},
{"SHLL $7, (R11)", "41 c1 23 07"},
{"SHLL $7, DX", "c1 e2 07"},
{"SHLL $7, DX, (BX)", "0f a4 13 07"},
{"SHLL $7, DX, (R11)", "41 0f a4 13 07"},
{"SHLL $7, DX, DX", "0f a4 d2 07"},
{"SHLL $7, DX, R11", "41 0f a4 d3 07"},
{"SHLL $7, R11", "41 c1 e3 07"},
{"SHLL $7, R11, (BX)", "44 0f a4 1b 07"},
{"SHLL $7, R11, (R11)", "45 0f a4 1b 07"},
{"SHLL $7, R11, DX", "44 0f a4 da 07"},
{"SHLL $7, R11, R11", "45 0f a4 db 07"},
{"SHLL CL, (BX)", "d3 23"},
{"SHLL CL, (R11)", "41 d3 23"},
{"SHLL CL, DX", "d3 e2"},
{"SHLL CL, DX, (BX)", "0f a5 13"},
{"SHLL CL, DX, (R11)", "41 0f a5 13"},
{"SHLL CL, DX, DX", "0f a5 d2"},
{"SHLL CL, DX, R11", "41 0f a5 d3"},
{"SHLL CL, R11", "41 d3 e3"},
{"SHLL CL, R11, (BX)", "44 0f a5 1b"},
{"SHLL CL, R11, (R11)", "45 0f a5 1b"},
{"SHLL CL, R11, DX", "44 0f a5 da"},
{"SHLL CL, R11, R11", "45 0f a5 db"},
{"SHLQ $1, (BX)", "48 d1 23"},
{"SHLQ $1, (R11)", "49 d1 23"},
{"SHLQ $1, DX", "48 d1 e2"},
{"SHLQ $1, R11", "49 d1 e3"},
{"SHLQ $7, (BX)", "48 c1 23 07"},
{"SHLQ $7, (R11)", "49 c1 23 07"},
{"SHLQ $7, DX", "48 c1 e2 07"},
{"SHLQ $7, DX, (BX)", "48 0f a4 13 07"},
{"SHLQ $7, DX, (R11)", "49 0f a4 13 07"},
{"SHLQ $7, DX, DX", "48 0f a4 d2 07"},
{"SHLQ $7, DX, R11", "49 0f a4 d3 07"},
{"SHLQ $7, R11", "49 c1 e3 07"},
{"SHLQ $7, R11, (BX)", "4c 0f a4 1b 07"},
{"SHLQ $7, R11, (R11)", "4d 0f a4 1b 07"},
{"SHLQ $7, R11, DX", "4c 0f a4 da 07"},
{"SHLQ $7, R11, R11", "4d 0f a4 db 07"},
{"SHLQ CL, (BX)", "48 d3 23"},
{"SHLQ CL, (R11)", "49 d3 23"},
{"SHLQ CL, DX", "48 d3 e2"},
{"SHLQ CL, DX, (BX)", "48 0f a5 13"},
{"SHLQ CL, DX, (R11)", "49 0f a5 13"},
{"SHLQ CL, DX, DX", "48 0f a5 d2"},
{"SHLQ CL, DX, R11", "49 0f a5 d3"},
{"SHLQ CL, R11", "49 d3 e3"},
{"SHLQ CL, R11, (BX)", "4c 0f a5 1b"},
{"SHLQ CL, R11, (R11)", "4d 0f a5 1b"},
{"SHLQ CL, R11, DX", "4c 0f a5 da"},
{"SHLQ CL, R11, R11", "4d 0f a5 db"},
{"SHLW $1, (BX)", "66 d1 23"},
{"SHLW $1, (R11)", "66 41 d1 23"},
{"SHLW $1, DX", "66 d1 e2"},
{"SHLW $1, R11", "66 41 d1 e3"},
{"SHLW $7, (BX)", "66 c1 23 07"},
{"SHLW $7, (R11)", "66 41 c1 23 07"},
{"SHLW $7, DX", "66 c1 e2 07"},
{"SHLW $7, DX, (BX)", "66 0f a4 13 07"},
{"SHLW $7, DX, (R11)", "66 41 0f a4 13 07"},
{"SHLW $7, DX, DX", "66 0f a4 d2 07"},
{"SHLW $7, DX, R11", "66 41 0f a4 d3 07"},
{"SHLW $7, R11", "66 41 c1 e3 07"},
{"SHLW $7, R11, (BX)", "66 44 0f a4 1b 07"},
{"SHLW $7, R11, (R11)", "66 45 0f a4 1b 07"},
{"SHLW $7, R11, DX", "66 44 0f a4 da 07"},
{"SHLW $7, R11, R11", "66 45 0f a4 db 07"},
{"SHLW CL, (BX)", "66 d3 23"},
{"SHLW CL, (R11)", "66 41 d3 23"},
{"SHLW CL, DX", "66 d3 e2"},
{"SHLW CL, DX, (BX)", "66 0f a5 13"},
{"SHLW CL, DX, (R11)", "66 41 0f a5 13"},
{"SHLW CL, DX, DX", "66 0f a5 d2"},
{"SHLW CL, DX, R11", "66 41 0f a5 d3"},
{"SHLW CL, R11", "66 41 d3 e3"},
{"SHLW CL, R11, (BX)", "66 44 0f a5 1b"},
{"SHLW CL, R11, (R11)", "66 45 0f a5 1b"},
{"SHLW CL, R11, DX", "66 44 0f a5 da"},
{"SHLW CL, R11, R11", "66 45 0f a5 db"},
{"VCMPPD $7, (BX), X9, X11", "c4 61 31 c2 1b 07|c5 31 c2 1b 07"},
{"VCMPPD $7, (BX), X9, X2", "c4 e1 31 c2 13 07|c5 b1 c2 13 07"},
{"VCMPPD $7, (BX), Y15, Y11", "c4 61 05 c2 1b 07|c5 05 c2 1b 07"},
{"VCMPPD $7, (BX), Y15, Y2", "c4 e1 05 c2 13 07|c5 85 c2 13 07"},
{"VCMPPD $7, (R11), X9, X11", "c4 41 31 c2 1b 07"},
{"VCMPPD $7, (R11), X9, X2", "c4 c1 31 c2 13 07"},
{"VCMPPD $7, (R11), Y15, Y11", "c4 41 05 c2 1b 07"},
{"VCMPPD $7, (R11), Y15, Y2", "c4 c1 05 c2 13 07"},
{"VCMPPD $7, X11, X9, X11", "c4 41 31 c2 db 07"},
{"VCMPPD $7, X11, X9, X2", "c4 c1 31 c2 d3 07"},
{"VCMPPD $7, X2, X9, X11", "c4 61 31 c2 da 07|c5 31 c2 da 07"},
{"VCMPPD $7, X2, X9, X2", "c4 e1 31 c2 d2 07|c5 b1 c2 d2 07"},
{"VCMPPD $7, Y11, Y15, Y11", "c4 41 05 c2 db 07"},
{"VCMPPD $7, Y11, Y15, Y2", "c4 c1 05 c2 d3 07"},
{"VCMPPD $7, Y2, Y15, Y11", "c4 61 05 c2 da 07|c5 05 c2 da 07"},
{"VCMPPD $7, Y2, Y15, Y2", "c4 e1 05 c2 d2 07|c5 85 c2 d2 07"},
{"VCMPPS $7, (BX), X9, X11", "c4 61 30 c2 1b 07|c5 30 c2 1b 07"},
{"VCMPPS $7, (BX), X9, X2", "c4 e1 30 c2 13 07|c5 b0 c2 13 07"},
{"VCMPPS $7, (BX), Y15, Y11", "c4 61 04 c2 1b 07|c5 04 c2 1b 07"},
{"VCMPPS $7, (BX), Y15, Y2", "c4 e1 04 c2 13 07|c5 84 c2 13 07"},
{"VCMPPS $7, (R11), X9, X11", "c4 41 30 c2 1b 07"},
{"VCMPPS $7, (R11), X9, X2", "c4 c1 30 c2 13 07"},
{"VCMPPS $7, (R11), Y15, Y11", "c4 41 04 c2 1b 07"},
{"VCMPPS $7, (R11), Y15, Y2", "c4 c1 04 c2 13 07"},
{"VCMPPS $7, X11, X9, X11", "c4 41 30 c2 db 07"},
{"VCMPPS $7, X11, X9, X2", "c4 c1 30 c2 d3 07"},
{"VCMPPS $7, X2, X9, X11", "c4 61 30 c2 da 07|c5 30 c2 da 07"},
{"VCMPPS $7, X2, X9, X2", "c4 e1 30 c2 d2 07|c5 b0 c2 d2 07"},
{"VCMPPS $7, Y11, Y15, Y11", "c4 41 04 c2 db 07"},
{"VCMPPS $7, Y11, Y15, Y2", "c4 c1 04 c2 d3 07"},
{"VCMPPS $7, Y2, Y15, Y11", "c4 61 04 c2 da 07|c5 04 c2 da 07"},
{"VCMPPS $7, Y2, Y15, Y2", "c4 e1 04 c2 d2 07|c5 84 c2 d2 07"},
{"VCMPSD $7, (BX), X9, X11", "c4 61 33 c2 1b 07|c5 33 c2 1b 07"},
{"VCMPSD $7, (BX), X9, X2", "c4 e1 33 c2 13 07|c5 b3 c2 13 07"},
{"VCMPSD $7, (R11), X9, X11", "c4 41 33 c2 1b 07"},
{"VCMPSD $7, (R11), X9, X2", "c4 c1 33 c2 13 07"},
{"VCMPSD $7, X11, X9, X11", "c4 41 33 c2 db 07"},
{"VCMPSD $7, X11, X9, X2", "c4 c1 33 c2 d3 07"},
{"VCMPSD $7, X2, X9, X11", "c4 61 33 c2 da 07|c5 33 c2 da 07"},
{"VCMPSD $7, X2, X9, X2", "c4 e1 33 c2 d2 07|c5 b3 c2 d2 07"},
{"VCMPSS $7, (BX), X9, X11", "c4 61 32 c2 1b 07|c5 32 c2 1b 07"},
{"VCMPSS $7, (BX), X9, X2", "c4 e1 32 c2 13 07|c5 b2 c2 13 07"},
{"VCMPSS $7, (R11), X9, X11", "c4 41 32 c2 1b 07"},
{"VCMPSS $7, (R11), X9, X2", "c4 c1 32 c2 13 07"},
{"VCMPSS $7, X11, X9, X11", "c4 41 32 c2 db 07"},
{"VCMPSS $7, X11, X9, X2", "c4 c1 32 c2 d3 07"},
{"VCMPSS $7, X2, X9, X11", "c4 61 32 c2 da 07|c5 32 c2 da 07"},
{"VCMPSS $7, X2, X9, X2", "c4 e1 32 c2 d2 07|c5 b2 c2 d2 07"},
{"VCOMISS (BX), X11", "c4 61 78 2f 1b|c5 78 2f 1b"},
{"VCOMISS (BX), X2", "c4 e1 78 2f 13|c5 f8 2f 13"},
{"VCOMISS (R11), X11", "c4 41 78 2f 1b"},
{"VCOMISS (R11), X2", "c4 c1 78 2f 13"},
{"VCOMISS X11, X11", "c4 41 78 2f db"},
{"VCOMISS X11, X2", "c4 c1 78 2f d3"},
{"VCOMISS X2, X11", "c4 61 78 2f da|c5 78 2f da"},
{"VCOMISS X2, X2", "c4 e1 78 2f d2|c5 f8 2f d2"},
{"VCVTPS2DQ (BX), X11", "c4 61 79 5b 1b|c5 79 5b 1b"},
{"VCVTPS2DQ (BX), X2", "c4 e1 79 5b 13|c5 f9 5b 13"},
{"VCVTPS2DQ (BX), Y11", "c4 61 7d 5b 1b|c5 7d 5b 1b"},
{"VCVTPS2DQ (BX), Y2", "c4 e1 7d 5b 13|c5 fd 5b 13"},
{"VCVTPS2DQ (R11), X11", "c4 41 79 5b 1b"},
{"VCVTPS2DQ (R11), X2", "c4 c1 79 5b 13"},
{"VCVTPS2DQ (R11), Y11", "c4 41 7d 5b 1b"},
{"VCVTPS2DQ (R11), Y2", "c4 c1 7d 5b 13"},
{"VCVTPS2DQ X11, X11", "c4 41 79 5b db"},
{"VCVTPS2DQ X11, X2", "c4 c1 79 5b d3"},
{"VCVTPS2DQ X2, X11", "c4 61 79 5b da|c5 79 5b da"},
{"VCVTPS2DQ X2, X2", "c4 e1 79 5b d2|c5 f9 5b d2"},
{"VCVTPS2DQ Y11, Y11", "c4 41 7d 5b db"},
{"VCVTPS2DQ Y11, Y2", "c4 c1 7d 5b d3"},
{"VCVTPS2DQ Y2, Y11", "c4 61 7d 5b da|c5 7d 5b da"},
{"VCVTPS2DQ Y2, Y2", "c4 e1 7d 5b d2|c5 fd 5b d2"},
{"VCVTTPS2DQ (BX), X11", "c4 61 7a 5b 1b|c5 7a 5b 1b"},
{"VCVTTPS2DQ (BX), X2", "c4 e1 7a 5b 13|c5 fa 5b 13"},
{"VCVTTPS2DQ (BX), Y11", "c4 61 7e 5b 1b|c5 7e 5b 1b"},
{"VCVTTPS2DQ (BX), Y2", "c4 e1 7e 5b 13|c5 fe 5b 13"},
{"VCVTTPS2DQ (R11), X11", "c4 41 7a 5b 1b"},
{"VCVTTPS2DQ (R11), X2", "c4 c1 7a 5b 13"},
{"VCVTTPS2DQ (R11), Y11", "c4 41 7e 5b 1b"},
{"VCVTTPS2DQ (R11), Y2", "c4 c1 7e 5b 13"},
{"VCVTTPS2DQ X11, X11", "c4 41 7a 5b db"},
{"VCVTTPS2DQ X11, X2", "c4 c1 7a 5b d3"},
{"VCVTTPS2DQ X2, X11", "c4 61 7a 5b da|c5 7a 5b da"},
{"VCVTTPS2DQ X2, X2", "c4 e1 7a 5b d2|c5 fa 5b d2"},
{"VCVTTPS2DQ Y11, Y11", "c4 41 7e 5b db"},
{"VCVTTPS2DQ Y11, Y2", "c4 c1 7e 5b d3"},
{"VCVTTPS2DQ Y2, Y11", "c4 61 7e 5b da|c5 7e 5b da"},
{"VCVTTPS2DQ Y2, Y2", "c4 e1 7e 5b d2|c5 fe 5b d2"},
{"VMOVLHPS X11, X9, X11", "c4 41 30 16 db"},
{"VMOVLHPS X11, X9, X2", "c4 c1 30 16 d3"},
{"VMOVLHPS X2, X9, X11", "c4 61 30 16 da|c5 30 16 da"},
{"VMOVLHPS X2, X9, X2", "c4 e1 30 16 d2|c5 b0 16 d2"},
{"VMOVUPS (BX), X11", "c4 61 78 10 1b|c5 78 10 1b"},
{"VMOVUPS (BX), X2", "c4 e1 78 10 13|c5 f8 10 13"},
{"VMOVUPS (BX), Y11", "c4 61 7c 10 1b|c5 7c 10 1b"},
{"VMOVUPS (BX), Y2", "c4 e1 7c 10 13|c5 fc 10 13"},
{"VMOVUPS (R11), X11", "c4 41 78 10 1b"},
{"VMOVUPS (R11), X2", "c4 c1 78 10 13"},
{"VMOVUPS (R11), Y11", "c4 41 7c 10 1b"},
{"VMOVUPS (R11), Y2", "c4 c1 7c 10 13"},
{"VMOVUPS X11, (BX)", "c4 61 78 11 1b|c5 78 11 1b"},
{"VMOVUPS X11, (R11)", "c4 41 78 11 1b"},
{"VMOVUPS X11, X11", "c4 41 78 10 db|c4 41 78 11 db"},
{"VMOVUPS X11, X2", "c4 c1 78 10 d3|c4 61 78 11 da|c5 78 11 da"},
{"VMOVUPS X2, (BX)", "c4 e1 78 11 13|c5 f8 11 13"},
{"VMOVUPS X2, (R11)", "c4 c1 78 11 13"},
{"VMOVUPS X2, X11", "c4 61 78 10 da|c5 78 10 da|c4 c1 78 11 d3"},
{"VMOVUPS X2, X2", "c4 e1 78 10 d2|c5 f8 10 d2|c4 e1 78 11 d2|c5 f8 11 d2"},
{"VMOVUPS Y11, (BX)", "c4 61 7c 11 1b|c5 7c 11 1b"},
{"VMOVUPS Y11, (R11)", "c4 41 7c 11 1b"},
{"VMOVUPS Y11, Y11", "c4 41 7c 10 db|c4 41 7c 11 db"},
{"VMOVUPS Y11, Y2", "c4 c1 7c 10 d3|c4 61 7c 11 da|c5 7c 11 da"},
{"VMOVUPS Y2, (BX)", "c4 e1 7c 11 13|c5 fc 11 13"},
{"VMOVUPS Y2, (R11)", "c4 c1 7c 11 13"},
{"VMOVUPS Y2, Y11", "c4 61 7c 10 da|c5 7c 10 da|c4 c1 7c 11 d3"},
{"VMOVUPS Y2, Y2", "c4 e1 7c 10 d2|c5 fc 10 d2|c4 e1 7c 11 d2|c5 fc 11 d2"},
{"VPABSB (BX), X11", "c4 62 79 1c 1b"},
{"VPABSB (BX), X2", "c4 e2 79 1c 13"},
{"VPABSB (BX), Y11", "c4 62 7d 1c 1b"},
{"VPABSB (BX), Y2", "c4 e2 7d 1c 13"},
{"VPABSB (R11), X11", "c4 42 79 1c 1b"},
{"VPABSB (R11), X2", "c4 c2 79 1c 13"},
{"VPABSB (R11), Y11", "c4 42 7d 1c 1b"},
{"VPABSB (R11), Y2", "c4 c2 7d 1c 13"},
{"VPABSB X11, X11", "c4 42 79 1c db"},
{"VPABSB X11, X2", "c4 c2 79 1c d3"},
{"VPABSB X2, X11", "c4 62 79 1c da"},
{"VPABSB X2, X2", "c4 e2 79 1c d2"},
{"VPABSB Y11, Y11", "c4 42 7d 1c db"},
{"VPABSB Y11, Y2", "c4 c2 7d 1c d3"},
{"VPABSB Y2, Y11", "c4 62 7d 1c da"},
{"VPABSB Y2, Y2", "c4 e2 7d 1c d2"},
{"VPABSD (BX), X11", "c4 62 79 1e 1b"},
{"VPABSD (BX), X2", "c4 e2 79 1e 13"},
{"VPABSD (BX), Y11", "c4 62 7d 1e 1b"},
{"VPABSD (BX), Y2", "c4 e2 7d 1e 13"},
{"VPABSD (R11), X11", "c4 42 79 1e 1b"},
{"VPABSD (R11), X2", "c4 c2 79 1e 13"},
{"VPABSD (R11), Y11", "c4 42 7d 1e 1b"},
{"VPABSD (R11), Y2", "c4 c2 7d 1e 13"},
{"VPABSD X11, X11", "c4 42 79 1e db"},
{"VPABSD X11, X2", "c4 c2 79 1e d3"},
{"VPABSD X2, X11", "c4 62 79 1e da"},
{"VPABSD X2, X2", "c4 e2 79 1e d2"},
{"VPABSD Y11, Y11", "c4 42 7d 1e db"},
{"VPABSD Y11, Y2", "c4 c2 7d 1e d3"},
{"VPABSD Y2, Y11", "c4 62 7d 1e da"},
{"VPABSD Y2, Y2", "c4 e2 7d 1e d2"},
{"VPABSW (BX), X11", "c4 62 79 1d 1b"},
{"VPABSW (BX), X2", "c4 e2 79 1d 13"},
{"VPABSW (BX), Y11", "c4 62 7d 1d 1b"},
{"VPABSW (BX), Y2", "c4 e2 7d 1d 13"},
{"VPABSW (R11), X11", "c4 42 79 1d 1b"},
{"VPABSW (R11), X2", "c4 c2 79 1d 13"},
{"VPABSW (R11), Y11", "c4 42 7d 1d 1b"},
{"VPABSW (R11), Y2", "c4 c2 7d 1d 13"},
{"VPABSW X11, X11", "c4 42 79 1d db"},
{"VPABSW X11, X2", "c4 c2 79 1d d3"},
{"VPABSW X2, X11", "c4 62 79 1d da"},
{"VPABSW X2, X2", "c4 e2 79 1d d2"},
{"VPABSW Y11, Y11", "c4 42 7d 1d db"},
{"VPABSW Y11, Y2", "c4 c2 7d 1d d3"},
{"VPABSW Y2, Y11", "c4 62 7d 1d da"},
{"VPABSW Y2, Y2", "c4 e2 7d 1d d2"},
{"VPACKSSWB (BX), X9, X11", "c4 61 31 63 1b|c5 31 63 1b"},
{"VPACKSSWB (BX), X9, X2", "c4 e1 31 63 13|c5 b1 63 13"},
{"VPACKSSWB (BX), Y15, Y11", "c4 61 05 63 1b|c5 05 63 1b"},
{"VPACKSSWB (BX), Y15, Y2", "c4 e1 05 63 13|c5 85 63 13"},
{"VPACKSSWB (R11), X9, X11", "c4 41 31 63 1b"},
{"VPACKSSWB (R11), X9, X2", "c4 c1 31 63 13"},
{"VPACKSSWB (R11), Y15, Y11", "c4 41 05 63 1b"},
{"VPACKSSWB (R11), Y15, Y2", "c4 c1 05 63 13"},
{"VPACKSSWB X11, X9, X11", "c4 41 31 63 db"},
{"VPACKSSWB X11, X9, X2", "c4 c1 31 63 d3"},
{"VPACKSSWB X2, X9, X11", "c4 61 31 63 da|c5 31 63 da"},
{"VPACKSSWB X2, X9, X2", "c4 e1 31 63 d2|c5 b1 63 d2"},
{"VPACKSSWB Y11, Y15, Y11", "c4 41 05 63 db"},
{"VPACKSSWB Y11, Y15, Y2", "c4 c1 05 63 d3"},
{"VPACKSSWB Y2, Y15, Y11", "c4 61 05 63 da|c5 05 63 da"},
{"VPACKSSWB Y2, Y15, Y2", "c4 e1 05 63 d2|c5 85 63 d2"},
{"VPACKUSDW (BX), X9, X11", "c4 62 31 2b 1b"},
{"VPACKUSDW (BX), X9, X2", "c4 e2 31 2b 13"},
{"VPACKUSDW (BX), Y15, Y11", "c4 62 05 2b 1b"},
{"VPACKUSDW (BX), Y15, Y2", "c4 e2 05 2b 13"},
{"VPACKUSDW (R11), X9, X11", "c4 42 31 2b 1b"},
{"VPACKUSDW (R11), X9, X2", "c4 c2 31 2b 13"},
{"VPACKUSDW (R11), Y15, Y11", "c4 42 05 2b 1b"},
{"VPACKUSDW (R11), Y15, Y2", "c4 c2 05 2b 13"},
{"VPACKUSDW X11, X9, X11", "c4 42 31 2b db"},
{"VPACKUSDW X11, X9, X2", "c4 c2 31 2b d3"},
{"VPACKUSDW X2, X9, X11", "c4 62 31 2b da"},
{"VPACKUSDW X2, X9, X2", "c4 e2 31 2b d2"},
{"VPACKUSDW Y11, Y15, Y11", "c4 42 05 2b db"},
{"VPACKUSDW Y11, Y15, Y2", "c4 c2 05 2b d3"},
{"VPACKUSDW Y2, Y15, Y11", "c4 62 05 2b da"},
{"VPACKUSDW Y2, Y15, Y2", "c4 e2 05 2b d2"},
{"VPACKUSWB (BX), X9, X11", "c4 61 31 67 1b|c5 31 67 1b"},
{"VPACKUSWB (BX), X9, X2", "c4 e1 31 67 13|c5 b1 67 13"},
{"VPACKUSWB (BX), Y15, Y11", "c4 61 05 67 1b|c5 05 67 1b"},
{"VPACKUSWB (BX), Y15, Y2", "c4 e1 05 67 13|c5 85 67 13"},
{"VPACKUSWB (R11), X9, X11", "c4 41 31 67 1b"},
{"VPACKUSWB (R11), X9, X2", "c4 c1 31 67 13"},
{"VPACKUSWB (R11), Y15, Y11", "c4 41 05 67 1b"},
{"VPACKUSWB (R11), Y15, Y2", "c4 c1 05 67 13"},
{"VPACKUSWB X11, X9, X11", "c4 41 31 67 db"},
{"VPACKUSWB X11, X9, X2", "c4 c1 31 67 d3"},
{"VPACKUSWB X2, X9, X11", "c4 61 31 67 da|c5 31 67 da"},
{"VPACKUSWB X2, X9, X2", "c4 e1 31 67 d2|c5 b1 67 d2"},
{"VPACKUSWB Y11, Y15, Y11", "c4 41 05 67 db"},
{"VPACKUSWB Y11, Y15, Y2", "c4 c1 05 67 d3"},
{"VPACKUSWB Y2, Y15, Y11", "c4 61 05 67 da|c5 05 67 da"},
{"VPACKUSWB Y2, Y15, Y2", "c4 e1 05 67 d2|c5 85 67 d2"},
{"VPADDB (BX), X9, X11", "c4 61 31 fc 1b|c5 31 fc 1b"},
{"VPADDB (BX), X9, X2", "c4 e1 31 fc 13|c5 b1 fc 13"},
{"VPADDB (BX), Y15, Y11", "c4 61 05 fc 1b|c5 05 fc 1b"},
{"VPADDB (BX), Y15, Y2", "c4 e1 05 fc 13|c5 85 fc 13"},
{"VPADDB (R11), X9, X11", "c4 41 31 fc 1b"},
{"VPADDB (R11), X9, X2", "c4 c1 31 fc 13"},
{"VPADDB (R11), Y15, Y11", "c4 41 05 fc 1b"},
{"VPADDB (R11), Y15, Y2", "c4 c1 05 fc 13"},
{"VPADDB X11, X9, X11", "c4 41 31 fc db"},
{"VPADDB X11, X9, X2", "c4 c1 31 fc d3"},
{"VPADDB X2, X9, X11", "c4 61 31 fc da|c5 31 fc da"},
{"VPADDB X2, X9, X2", "c4 e1 31 fc d2|c5 b1 fc d2"},
{"VPADDB Y11, Y15, Y11", "c4 41 05 fc db"},
{"VPADDB Y11, Y15, Y2", "c4 c1 05 fc d3"},
{"VPADDB Y2, Y15, Y11", "c4 61 05 fc da|c5 05 fc da"},
{"VPADDB Y2, Y15, Y2", "c4 e1 05 fc d2|c5 85 fc d2"},
{"VPADDW (BX), X9, X11", "c4 61 31 fd 1b|c5 31 fd 1b"},
{"VPADDW (BX), X9, X2", "c4 e1 31 fd 13|c5 b1 fd 13"},
{"VPADDW (BX), Y15, Y11", "c4 61 05 fd 1b|c5 05 fd 1b"},
{"VPADDW (BX), Y15, Y2", "c4 e1 05 fd 13|c5 85 fd 13"},
{"VPADDW (R11), X9, X11", "c4 41 31 fd 1b"},
{"VPADDW (R11), X9, X2", "c4 c1 31 fd 13"},
{"VPADDW (R11), Y15, Y11", "c4 41 05 fd 1b"},
{"VPADDW (R11), Y15, Y2", "c4 c1 05 fd 13"},
{"VPADDW X11, X9, X11", "c4 41 31 fd db"},
{"VPADDW X11, X9, X2", "c4 c1 31 fd d3"},
{"VPADDW X2, X9, X11", "c4 61 31 fd da|c5 31 fd da"},
{"VPADDW X2, X9, X2", "c4 e1 31 fd d2|c5 b1 fd d2"},
{"VPADDW Y11, Y15, Y11", "c4 41 05 fd db"},
{"VPADDW Y11, Y15, Y2", "c4 c1 05 fd d3"},
{"VPADDW Y2, Y15, Y11", "c4 61 05 fd da|c5 05 fd da"},
{"VPADDW Y2, Y15, Y2", "c4 e1 05 fd d2|c5 85 fd d2"},
{"VPAVGB (BX), X9, X11", "c4 61 31 e0 1b|c5 31 e0 1b"},
{"VPAVGB (BX), X9, X2", "c4 e1 31 e0 13|c5 b1 e0 13"},
{"VPAVGB (BX), Y15, Y11", "c4 61 05 e0 1b|c5 05 e0 1b"},
{"VPAVGB (BX), Y15, Y2", "c4 e1 05 e0 13|c5 85 e0 13"},
{"VPAVGB (R11), X9, X11", "c4 41 31 e0 1b"},
{"VPAVGB (R11), X9, X2", "c4 c1 31 e0 13"},
{"VPAVGB (R11), Y15, Y11", "c4 41 05 e0 1b"},
{"VPAVGB (R11), Y15, Y2", "c4 c1 05 e0 13"},
{"VPAVGB X11, X9, X11", "c4 41 31 e0 db"},
{"VPAVGB X11, X9, X2", "c4 c1 31 e0 d3"},
{"VPAVGB X2, X9, X11", "c4 61 31 e0 da|c5 31 e0 da"},
{"VPAVGB X2, X9, X2", "c4 e1 31 e0 d2|c5 b1 e0 d2"},
{"VPAVGB Y11, Y15, Y11", "c4 41 05 e0 db"},
{"VPAVGB Y11, Y15, Y2", "c4 c1 05 e0 d3"},
{"VPAVGB Y2, Y15, Y11", "c4 61 05 e0 da|c5 05 e0 da"},
{"VPAVGB Y2, Y15, Y2", "c4 e1 05 e0 d2|c5 85 e0 d2"},
{"VPAVGW (BX), X9, X11", "c4 61 31 e3 1b|c5 31 e3 1b"},
{"VPAVGW (BX), X9, X2", "c4 e1 31 e3 13|c5 b1 e3 13"},
{"VPAVGW (BX), Y15, Y11", "c4 61 05 e3 1b|c5 05 e3 1b"},
{"VPAVGW (BX), Y15, Y2", "c4 e1 05 e3 13|c5 85 e3 13"},
{"VPAVGW (R11), X9, X11", "c4 41 31 e3 1b"},
{"VPAVGW (R11), X9, X2", "c4 c1 31 e3 13"},
{"VPAVGW (R11), Y15, Y11", "c4 41 05 e3 1b"},
{"VPAVGW (R11), Y15, Y2", "c4 c1 05 e3 13"},
{"VPAVGW X11, X9, X11", "c4 41 31 e3 db"},
{"VPAVGW X11, X9, X2", "c4 c1 31 e3 d3"},
{"VPAVGW X2, X9, X11", "c4 61 31 e3 da|c5 31 e3 da"},
{"VPAVGW X2, X9, X2", "c4 e1 31 e3 d2|c5 b1 e3 d2"},
{"VPAVGW Y11, Y15, Y11", "c4 41 05 e3 db"},
{"VPAVGW Y11, Y15, Y2", "c4 c1 05 e3 d3"},
{"VPAVGW Y2, Y15, Y11", "c4 61 05 e3 da|c5 05 e3 da"},
{"VPAVGW Y2, Y15, Y2", "c4 e1 05 e3 d2|c5 85 e3 d2"},
{"VPMADDUBSW (BX), X9, X11", "c4 62 31 04 1b"},
{"VPMADDUBSW (BX), X9, X2", "c4 e2 31 04 13"},
{"VPMADDUBSW (BX), Y15, Y11", "c4 62 05 04 1b"},
{"VPMADDUBSW (BX), Y15, Y2", "c4 e2 05 04 13"},
{"VPMADDUBSW (R11), X9, X11", "c4 42 31 04 1b"},
{"VPMADDUBSW (R11), X9, X2", "c4 c2 31 04 13"},
{"VPMADDUBSW (R11), Y15, Y11", "c4 42 05 04 1b"},
{"VPMADDUBSW (R11), Y15, Y2", "c4 c2 05 04 13"},
{"VPMADDUBSW X11, X9, X11", "c4 42 31 04 db"},
{"VPMADDUBSW X11, X9, X2", "c4 c2 31 04 d3"},
{"VPMADDUBSW X2, X9, X11", "c4 62 31 04 da"},
{"VPMADDUBSW X2, X9, X2", "c4 e2 31 04 d2"},
{"VPMADDUBSW Y11, Y15, Y11", "c4 42 05 04 db"},
{"VPMADDUBSW Y11, Y15, Y2", "c4 c2 05 04 d3"},
{"VPMADDUBSW Y2, Y15, Y11", "c4 62 05 04 da"},
{"VPMADDUBSW Y2, Y15, Y2", "c4 e2 05 04 d2"},
{"VPMADDWD (BX), X9, X11", "c4 61 31 f5 1b|c5 31 f5 1b"},
{"VPMADDWD (BX), X9, X2", "c4 e1 31 f5 13|c5 b1 f5 13"},
{"VPMADDWD (BX), Y15, Y11", "c4 61 05 f5 1b|c5 05 f5 1b"},
{"VPMADDWD (BX), Y15, Y2", "c4 e1 05 f5 13|c5 85 f5 13"},
{"VPMADDWD (R11), X9, X11", "c4 41 31 f5 1b"},
{"VPMADDWD (R11), X9, X2", "c4 c1 31 f5 13"},
{"VPMADDWD (R11), Y15, Y11", "c4 41 05 f5 1b"},
{"VPMADDWD (R11), Y15, Y2", "c4 c1 05 f5 13"},
{"VPMADDWD X11, X9, X11", "c4 41 31 f5 db"},
{"VPMADDWD X11, X9, X2", "c4 c1 31 f5 d3"},
{"VPMADDWD X2, X9, X11", "c4 61 31 f5 da|c5 31 f5 da"},
{"VPMADDWD X2, X9, X2", "c4 e1 31 f5 d2|c5 b1 f5 d2"},
{"VPMADDWD Y11, Y15, Y11", "c4 41 05 f5 db"},
{"VPMADDWD Y11, Y15, Y2", "c4 c1 05 f5 d3"},
{"VPMADDWD Y2, Y15, Y11", "c4 61 05 f5 da|c5 05 f5 da"},
{"VPMADDWD Y2, Y15, Y2", "c4 e1 05 f5 d2|c5 85 f5 d2"},
{"VPMAXSB (BX), X9, X11", "c4 62 31 3c 1b"},
{"VPMAXSB (BX), X9, X2", "c4 e2 31 3c 13"},
{"VPMAXSB (BX), Y15, Y11", "c4 62 05 3c 1b"},
{"VPMAXSB (BX), Y15, Y2", "c4 e2 05 3c 13"},
{"VPMAXSB (R11), X9, X11", "c4 42 31 3c 1b"},
{"VPMAXSB (R11), X9, X2", "c4 c2 31 3c 13"},
{"VPMAXSB (R11), Y15, Y11", "c4 42 05 3c 1b"},
{"VPMAXSB (R11), Y15, Y2", "c4 c2 05 3c 13"},
{"VPMAXSB X11, X9, X11", "c4 42 31 3c db"},
{"VPMAXSB X11, X9, X2", "c4 c2 31 3c d3"},
{"VPMAXSB X2, X9, X11", "c4 62 31 3c da"},
{"VPMAXSB X2, X9, X2", "c4 e2 31 3c d2"},
{"VPMAXSB Y11, Y15, Y11", "c4 42 05 3c db"},
{"VPMAXSB Y11, Y15, Y2", "c4 c2 05 3c d3"},
{"VPMAXSB Y2, Y15, Y11", "c4 62 05 3c da"},
{"VPMAXSB Y2, Y15, Y2", "c4 e2 05 3c d2"},
{"VPMAXSD (BX), X9, X11", "c4 62 31 3d 1b"},
{"VPMAXSD (BX), X9, X2", "c4 e2 31 3d 13"},
{"VPMAXSD (BX), Y15, Y11", "c4 62 05 3d 1b"},
{"VPMAXSD (BX), Y15, Y2", "c4 e2 05 3d 13"},
{"VPMAXSD (R11), X9, X11", "c4 42 31 3d 1b"},
{"VPMAXSD (R11), X9, X2", "c4 c2 31 3d 13"},
{"VPMAXSD (R11), Y15, Y11", "c4 42 05 3d 1b"},
{"VPMAXSD (R11), Y15, Y2", "c4 c2 05 3d 13"},
{"VPMAXSD X11, X9, X11", "c4 42 31 3d db"},
{"VPMAXSD X11, X9, X2", "c4 c2 31 3d d3"},
{"VPMAXSD X2, X9, X11", "c4 62 31 3d da"},
{"VPMAXSD X2, X9, X2", "c4 e2 31 3d d2"},
{"VPMAXSD Y11, Y15, Y11", "c4 42 05 3d db"},
{"VPMAXSD Y11, Y15, Y2", "c4 c2 05 3d d3"},
{"VPMAXSD Y2, Y15, Y11", "c4 62 05 3d da"},
{"VPMAXSD Y2, Y15, Y2", "c4 e2 05 3d d2"},
{"VPMAXSW (BX), X9, X11", "c4 61 31 ee 1b|c5 31 ee 1b"},
{"VPMAXSW (BX), X9, X2", "c4 e1 31 ee 13|c5 b1 ee 13"},
{"VPMAXSW (BX), Y15, Y11", "c4 61 05 ee 1b|c5 05 ee 1b"},
{"VPMAXSW (BX), Y15, Y2", "c4 e1 05 ee 13|c5 85 ee 13"},
{"VPMAXSW (R11), X9, X11", "c4 41 31 ee 1b"},
{"VPMAXSW (R11), X9, X2", "c4 c1 31 ee 13"},
{"VPMAXSW (R11), Y15, Y11", "c4 41 05 ee 1b"},
{"VPMAXSW (R11), Y15, Y2", "c4 c1 05 ee 13"},
{"VPMAXSW X11, X9, X11", "c4 41 31 ee db"},
{"VPMAXSW X11, X9, X2", "c4 c1 31 ee d3"},
{"VPMAXSW X2, X9, X11", "c4 61 31 ee da|c5 31 ee da"},
{"VPMAXSW X2, X9, X2", "c4 e1 31 ee d2|c5 b1 ee d2"},
{"VPMAXSW Y11, Y15, Y11", "c4 41 05 ee db"},
{"VPMAXSW Y11, Y15, Y2", "c4 c1 05 ee d3"},
{"VPMAXSW Y2, Y15, Y11", "c4 61 05 ee da|c5 05 ee da"},
{"VPMAXSW Y2, Y15, Y2", "c4 e1 05 ee d2|c5 85 ee d2"},
{"VPMAXUB (BX), X9, X11", "c4 61 31 de 1b|c5 31 de 1b"},
{"VPMAXUB (BX), X9, X2", "c4 e1 31 de 13|c5 b1 de 13"},
{"VPMAXUB (BX), Y15, Y11", "c4 61 05 de 1b|c5 05 de 1b"},
{"VPMAXUB (BX), Y15, Y2", "c4 e1 05 de 13|c5 85 de 13"},
{"VPMAXUB (R11), X9, X11", "c4 41 31 de 1b"},
{"VPMAXUB (R11), X9, X2", "c4 c1 31 de 13"},
{"VPMAXUB (R11), Y15, Y11", "c4 41 05 de 1b"},
{"VPMAXUB (R11), Y15, Y2", "c4 c1 05 de 13"},
{"VPMAXUB X11, X9, X11", "c4 41 31 de db"},
{"VPMAXUB X11, X9, X2", "c4 c1 31 de d3"},
{"VPMAXUB X2, X9, X11", "c4 61 31 de da|c5 31 de da"},
{"VPMAXUB X2, X9, X2", "c4 e1 31 de d2|c5 b1 de d2"},
{"VPMAXUB Y11, Y15, Y11", "c4 41 05 de db"},
{"VPMAXUB Y11, Y15, Y2", "c4 c1 05 de d3"},
{"VPMAXUB Y2, Y15, Y11", "c4 61 05 de da|c5 05 de da"},
{"VPMAXUB Y2, Y15, Y2", "c4 e1 05 de d2|c5 85 de d2"},
{"VPMAXUD (BX), X9, X11", "c4 62 31 3f 1b"},
{"VPMAXUD (BX), X9, X2", "c4 e2 31 3f 13"},
{"VPMAXUD (BX), Y15, Y11", "c4 62 05 3f 1b"},
{"VPMAXUD (BX), Y15, Y2", "c4 e2 05 3f 13"},
{"VPMAXUD (R11), X9, X11", "c4 42 31 3f 1b"},
{"VPMAXUD (R11), X9, X2", "c4 c2 31 3f 13"},
{"VPMAXUD (R11), Y15, Y11", "c4 42 05 3f 1b"},
{"VPMAXUD (R11), Y15, Y2", "c4 c2 05 3f 13"},
{"VPMAXUD X11, X9, X11", "c4 42 31 3f db"},
{"VPMAXUD X11, X9, X2", "c4 c2 31 3f d3"},
{"VPMAXUD X2, X9, X11", "c4 62 31 3f da"},
{"VPMAXUD X2, X9, X2", "c4 e2 31 3f d2"},
{"VPMAXUD Y11, Y15, Y11", "c4 42 05 3f db"},
{"VPMAXUD Y11, Y15, Y2", "c4 c2 05 3f d3"},
{"VPMAXUD Y2, Y15, Y11", "c4 62 05 3f da"},
{"VPMAXUD Y2, Y15, Y2", "c4 e2 05 3f d2"},
{"VPMAXUW (BX), X9, X11", "c4 62 31 3e 1b"},
{"VPMAXUW (BX), X9, X2", "c4 e2 31 3e 13"},
{"VPMAXUW (BX), Y15, Y11", "c4 62 05 3e 1b"},
{"VPMAXUW (BX), Y15, Y2", "c4 e2 05 3e 13"},
{"VPMAXUW (R11), X9, X11", "c4 42 31 3e 1b"},
{"VPMAXUW (R11), X9, X2", "c4 c2 31 3e 13"},
{"VPMAXUW (R11), Y15, Y11", "c4 42 05 3e 1b"},
{"VPMAXUW (R11), Y15, Y2", "c4 c2 05 3e 13"},
{"VPMAXUW X11, X9, X11", "c4 42 31 3e db"},
{"VPMAXUW X11, X9, X2", "c4 c2 31 3e d3"},
{"VPMAXUW X2, X9, X11", "c4 62 31 3e da"},
{"VPMAXUW X2, X9, X2", "c4 e2 31 3e d2"},
{"VPMAXUW Y11, Y15, Y11", "c4 42 05 3e db"},
{"VPMAXUW Y11, Y15, Y2", "c4 c2 05 3e d3"},
{"VPMAXUW Y2, Y15, Y11", "c4 62 05 3e da"},
{"VPMAXUW Y2, Y15, Y2", "c4 e2 05 3e d2"},
{"VPMINSB (BX), X9, X11", "c4 62 31 38 1b"},
{"VPMINSB (BX), X9, X2", "c4 e2 31 38 13"},
{"VPMINSB (BX), Y15, Y11", "c4 62 05 38 1b"},
{"VPMINSB (BX), Y15, Y2", "c4 e2 05 38 13"},
{"VPMINSB (R11), X9, X11", "c4 42 31 38 1b"},
{"VPMINSB (R11), X9, X2", "c4 c2 31 38 13"},
{"VPMINSB (R11), Y15, Y11", "c4 42 05 38 1b"},
{"VPMINSB (R11), Y15, Y2", "c4 c2 05 38 13"},
{"VPMINSB X11, X9, X11", "c4 42 31 38 db"},
{"VPMINSB X11, X9, X2", "c4 c2 31 38 d3"},
{"VPMINSB X2, X9, X11", "c4 62 31 38 da"},
{"VPMINSB X2, X9, X2", "c4 e2 31 38 d2"},
{"VPMINSB Y11, Y15, Y11", "c4 42 05 38 db"},
{"VPMINSB Y11, Y15, Y2", "c4 c2 05 38 d3"},
{"VPMINSB Y2, Y15, Y11", "c4 62 05 38 da"},
{"VPMINSB Y2, Y15, Y2", "c4 e2 05 38 d2"},
{"VPMINSD (BX), X9, X11", "c4 62 31 39 1b"},
{"VPMINSD (BX), X9, X2", "c4 e2 31 39 13"},
{"VPMINSD (BX), Y15, Y11", "c4 62 05 39 1b"},
{"VPMINSD (BX), Y15, Y2", "c4 e2 05 39 13"},
{"VPMINSD (R11), X9, X11", "c4 42 31 39 1b"},
{"VPMINSD (R11), X9, X2", "c4 c2 31 39 13"},
{"VPMINSD (R11), Y15, Y11", "c4 42 05 39 1b"},
{"VPMINSD (R11), Y15, Y2", "c4 c2 05 39 13"},
{"VPMINSD X11, X9, X11", "c4 42 31 39 db"},
{"VPMINSD X11, X9, X2", "c4 c2 31 39 d3"},
{"VPMINSD X2, X9, X11", "c4 62 31 39 da"},
{"VPMINSD X2, X9, X2", "c4 e2 31 39 d2"},
{"VPMINSD Y11, Y15, Y11", "c4 42 05 39 db"},
{"VPMINSD Y11, Y15, Y2", "c4 c2 05 39 d3"},
{"VPMINSD Y2, Y15, Y11", "c4 62 05 39 da"},
{"VPMINSD Y2, Y15, Y2", "c4 e2 05 39 d2"},
{"VPMINSW (BX), X9, X11", "c4 61 31 ea 1b|c5 31 ea 1b"},
{"VPMINSW (BX), X9, X2", "c4 e1 31 ea 13|c5 b1 ea 13"},
{"VPMINSW (BX), Y15, Y11", "c4 61 05 ea 1b|c5 05 ea 1b"},
{"VPMINSW (BX), Y15, Y2", "c4 e1 05 ea 13|c5 85 ea 13"},
{"VPMINSW (R11), X9, X11", "c4 41 31 ea 1b"},
{"VPMINSW (R11), X9, X2", "c4 c1 31 ea 13"},
{"VPMINSW (R11), Y15, Y11", "c4 41 05 ea 1b"},
{"VPMINSW (R11), Y15, Y2", "c4 c1 05 ea 13"},
{"VPMINSW X11, X9, X11", "c4 41 31 ea db"},
{"VPMINSW X11, X9, X2", "c4 c1 31 ea d3"},
{"VPMINSW X2, X9, X11", "c4 61 31 ea da|c5 31 ea da"},
{"VPMINSW X2, X9, X2", "c4 e1 31 ea d2|c5 b1 ea d2"},
{"VPMINSW Y11, Y15, Y11", "c4 41 05 ea db"},
{"VPMINSW Y11, Y15, Y2", "c4 c1 05 ea d3"},
{"VPMINSW Y2, Y15, Y11", "c4 61 05 ea da|c5 05 ea da"},
{"VPMINSW Y2, Y15, Y2", "c4 e1 05 ea d2|c5 85 ea d2"},
{"VPMINUB (BX), X9, X11", "c4 61 31 da 1b|c5 31 da 1b"},
{"VPMINUB (BX), X9, X2", "c4 e1 31 da 13|c5 b1 da 13"},
{"VPMINUB (BX), Y15, Y11", "c4 61 05 da 1b|c5 05 da 1b"},
{"VPMINUB (BX), Y15, Y2", "c4 e1 05 da 13|c5 85 da 13"},
{"VPMINUB (R11), X9, X11", "c4 41 31 da 1b"},
{"VPMINUB (R11), X9, X2", "c4 c1 31 da 13"},
{"VPMINUB (R11), Y15, Y11", "c4 41 05 da 1b"},
{"VPMINUB (R11), Y15, Y2", "c4 c1 05 da 13"},
{"VPMINUB X11, X9, X11", "c4 41 31 da db"},
{"VPMINUB X11, X9, X2", "c4 c1 31 da d3"},
{"VPMINUB X2, X9, X11", "c4 61 31 da da|c5 31 da da"},
{"VPMINUB X2, X9, X2", "c4 e1 31 da d2|c5 b1 da d2"},
{"VPMINUB Y11, Y15, Y11", "c4 41 05 da db"},
{"VPMINUB Y11, Y15, Y2", "c4 c1 05 da d3"},
{"VPMINUB Y2, Y15, Y11", "c4 61 05 da da|c5 05 da da"},
{"VPMINUB Y2, Y15, Y2", "c4 e1 05 da d2|c5 85 da d2"},
{"VPMINUD (BX), X9, X11", "c4 62 31 3b 1b"},
{"VPMINUD (BX), X9, X2", "c4 e2 31 3b 13"},
{"VPMINUD (BX), Y15, Y11", "c4 62 05 3b 1b"},
{"VPMINUD (BX), Y15, Y2", "c4 e2 05 3b 13"},
{"VPMINUD (R11), X9, X11", "c4 42 31 3b 1b"},
{"VPMINUD (R11), X9, X2", "c4 c2 31 3b 13"},
{"VPMINUD (R11), Y15, Y11", "c4 42 05 3b 1b"},
{"VPMINUD (R11), Y15, Y2", "c4 c2 05 3b 13"},
{"VPMINUD X11, X9, X11", "c4 42 31 3b db"},
{"VPMINUD X11, X9, X2", "c4 c2 31 3b d3"},
{"VPMINUD X2, X9, X11", "c4 62 31 3b da"},
{"VPMINUD X2, X9, X2", "c4 e2 31 3b d2"},
{"VPMINUD Y11, Y15, Y11", "c4 42 05 3b db"},
{"VPMINUD Y11, Y15, Y2", "c4 c2 05 3b d3"},
{"VPMINUD Y2, Y15, Y11", "c4 62 05 3b da"},
{"VPMINUD Y2, Y15, Y2", "c4 e2 05 3b d2"},
{"VPMINUW (BX), X9, X11", "c4 62 31 3a 1b"},
{"VPMINUW (BX), X9, X2", "c4 e2 31 3a 13"},
{"VPMINUW (BX), Y15, Y11", "c4 62 05 3a 1b"},
{"VPMINUW (BX), Y15, Y2", "c4 e2 05 3a 13"},
{"VPMINUW (R11), X9, X11", "c4 42 31 3a 1b"},
{"VPMINUW (R11), X9, X2", "c4 c2 31 3a 13"},
{"VPMINUW (R11), Y15, Y11", "c4 42 05 3a 1b"},
{"VPMINUW (R11), Y15, Y2", "c4 c2 05 3a 13"},
{"VPMINUW X11, X9, X11", "c4 42 31 3a db"},
{"VPMINUW X11, X9, X2", "c4 c2 31 3a d3"},
{"VPMINUW X2, X9, X11", "c4 62 31 3a da"},
{"VPMINUW X2, X9, X2", "c4 e2 31 3a d2"},
{"VPMINUW Y11, Y15, Y11", "c4 42 05 3a db"},
{"VPMINUW Y11, Y15, Y2", "c4 c2 05 3a d3"},
{"VPMINUW Y2, Y15, Y11", "c4 62 05 3a da"},
{"VPMINUW Y2, Y15, Y2", "c4 e2 05 3a d2"},
{"VPMOVSXBW (BX), X11", "c4 62 79 20 1b"},
{"VPMOVSXBW (BX), X2", "c4 e2 79 20 13"},
{"VPMOVSXBW (BX), Y11", "c4 62 7d 20 1b"},
{"VPMOVSXBW (BX), Y2", "c4 e2 7d 20 13"},
{"VPMOVSXBW (R11), X11", "c4 42 79 20 1b"},
{"VPMOVSXBW (R11), X2", "c4 c2 79 20 13"},
{"VPMOVSXBW (R11), Y11", "c4 42 7d 20 1b"},
{"VPMOVSXBW (R11), Y2", "c4 c2 7d 20 13"},
{"VPMOVSXBW X11, X11", "c4 42 79 20 db"},
{"VPMOVSXBW X11, X2", "c4 c2 79 20 d3"},
{"VPMOVSXBW X11, Y11", "c4 42 7d 20 db"},
{"VPMOVSXBW X11, Y2", "c4 c2 7d 20 d3"},
{"VPMOVSXBW X2, X11", "c4 62 79 20 da"},
{"VPMOVSXBW X2, X2", "c4 e2 79 20 d2"},
{"VPMOVSXBW X2, Y11", "c4 62 7d 20 da"},
{"VPMOVSXBW X2, Y2", "c4 e2 7d 20 d2"},
{"VPMULHUW (BX), X9, X11", "c4 61 31 e4 1b|c5 31 e4 1b"},
{"VPMULHUW (BX), X9, X2", "c4 e1 31 e4 13|c5 b1 e4 13"},
{"VPMULHUW (BX), Y15, Y11", "c4 61 05 e4 1b|c5 05 e4 1b"},
{"VPMULHUW (BX), Y15, Y2", "c4 e1 05 e4 13|c5 85 e4 13"},
{"VPMULHUW (R11), X9, X11", "c4 41 31 e4 1b"},
{"VPMULHUW (R11), X9, X2", "c4 c1 31 e4 13"},
{"VPMULHUW (R11), Y15, Y11", "c4 41 05 e4 1b"},
{"VPMULHUW (R11), Y15, Y2", "c4 c1 05 e4 13"},
{"VPMULHUW X11, X9, X11", "c4 41 31 e4 db"},
{"VPMULHUW X11, X9, X2", "c4 c1 31 e4 d3"},
{"VPMULHUW X2, X9, X11", "c4 61 31 e4 da|c5 31 e4 da"},
{"VPMULHUW X2, X9, X2", "c4 e1 31 e4 d2|c5 b1 e4 d2"},
{"VPMULHUW Y11, Y15, Y11", "c4 41 05 e4 db"},
{"VPMULHUW Y11, Y15, Y2", "c4 c1 05 e4 d3"},
{"VPMULHUW Y2, Y15, Y11", "c4 61 05 e4 da|c5 05 e4 da"},
{"VPMULHUW Y2, Y15, Y2", "c4 e1 05 e4 d2|c5 85 e4 d2"},
{"VPMULLW (BX), X9, X11", "c4 61 31 d5 1b|c5 31 d5 1b"},
{"VPMULLW (BX), X9, X2", "c4 e1 31 d5 13|c5 b1 d5 13"},
{"VPMULLW (BX), Y15, Y11", "c4 61 05 d5 1b|c5 05 d5 1b"},
{"VPMULLW (BX), Y15, Y2", "c4 e1 05 d5 13|c5 85 d5 13"},
{"VPMULLW (R11), X9, X11", "c4 41 31 d5 1b"},
{"VPMULLW (R11), X9, X2", "c4 c1 31 d5 13"},
{"VPMULLW (R11), Y15, Y11", "c4 41 05 d5 1b"},
{"VPMULLW (R11), Y15, Y2", "c4 c1 05 d5 13"},
{"VPMULLW X11, X9, X11", "c4 41 31 d5 db"},
{"VPMULLW X11, X9, X2", "c4 c1 31 d5 d3"},
{"VPMULLW X2, X9, X11", "c4 61 31 d5 da|c5 31 d5 da"},
{"VPMULLW X2, X9, X2", "c4 e1 31 d5 d2|c5 b1 d5 d2"},
{"VPMULLW Y11, Y15, Y11", "c4 41 05 d5 db"},
{"VPMULLW Y11, Y15, Y2", "c4 c1 05 d5 d3"},
{"VPMULLW Y2, Y15, Y11", "c4 61 05 d5 da|c5 05 d5 da"},
{"VPMULLW Y2, Y15, Y2", "c4 e1 05 d5 d2|c5 85 d5 d2"},
{"VPSLLVD (BX), X9, X11", "c4 62 31 47 1b"},
{"VPSLLVD (BX), X9, X2", "c4 e2 31 47 13"},
{"VPSLLVD (BX), Y15, Y11", "c4 62 05 47 1b"},
{"VPSLLVD (BX), Y15, Y2", "c4 e2 05 47 13"},
{"VPSLLVD (R11), X9, X11", "c4 42 31 47 1b"},
{"VPSLLVD (R11), X9, X2", "c4 c2 31 47 13"},
{"VPSLLVD (R11), Y15, Y11", "c4 42 05 47 1b"},
{"VPSLLVD (R11), Y15, Y2", "c4 c2 05 47 13"},
{"VPSLLVD X11, X9, X11", "c4 42 31 47 db"},
{"VPSLLVD X11, X9, X2", "c4 c2 31 47 d3"},
{"VPSLLVD X2, X9, X11", "c4 62 31 47 da"},
{"VPSLLVD X2, X9, X2", "c4 e2 31 47 d2"},
{"VPSLLVD Y11, Y15, Y11", "c4 42 05 47 db"},
{"VPSLLVD Y11, Y15, Y2", "c4 c2 05 47 d3"},
{"VPSLLVD Y2, Y15, Y11", "c4 62 05 47 da"},
{"VPSLLVD Y2, Y15, Y2", "c4 e2 05 47 d2"},
{"VPSLLVQ (BX), X9, X11", "c4 62 b1 47 1b"},
{"VPSLLVQ (BX), X9, X2", "c4 e2 b1 47 13"},
{"VPSLLVQ (BX), Y15, Y11", "c4 62 85 47 1b"},
{"VPSLLVQ (BX), Y15, Y2", "c4 e2 85 47 13"},
{"VPSLLVQ (R11), X9, X11", "c4 42 b1 47 1b"},
{"VPSLLVQ (R11), X9, X2", "c4 c2 b1 47 13"},
{"VPSLLVQ (R11), Y15, Y11", "c4 42 85 47 1b"},
{"VPSLLVQ (R11), Y15, Y2", "c4 c2 85 47 13"},
{"VPSLLVQ X11, X9, X11", "c4 42 b1 47 db"},
{"VPSLLVQ X11, X9, X2", "c4 c2 b1 47 d3"},
{"VPSLLVQ X2, X9, X11", "c4 62 b1 47 da"},
{"VPSLLVQ X2, X9, X2", "c4 e2 b1 47 d2"},
{"VPSLLVQ Y11, Y15, Y11", "c4 42 85 47 db"},
{"VPSLLVQ Y11, Y15, Y2", "c4 c2 85 47 d3"},
{"VPSLLVQ Y2, Y15, Y11", "c4 62 85 47 da"},
{"VPSLLVQ Y2, Y15, Y2", "c4 e2 85 47 d2"},
{"VPSRAVD (BX), X9, X11", "c4 62 31 46 1b"},
{"VPSRAVD (BX), X9, X2", "c4 e2 31 46 13"},
{"VPSRAVD (BX), Y15, Y11", "c4 62 05 46 1b"},
{"VPSRAVD (BX), Y15, Y2", "c4 e2 05 46 13"},
{"VPSRAVD (R11), X9, X11", "c4 42 31 46 1b"},
{"VPSRAVD (R11), X9, X2", "c4 c2 31 46 13"},
{"VPSRAVD (R11), Y15, Y11", "c4 42 05 46 1b"},
{"VPSRAVD (R11), Y15, Y2", "c4 c2 05 46 13"},
{"VPSRAVD X11, X9, X11", "c4 42 31 46 db"},
{"VPSRAVD X11, X9, X2", "c4 c2 31 46 d3"},
{"VPSRAVD X2, X9, X11", "c4 62 31 46 da"},
{"VPSRAVD X2, X9, X2", "c4 e2 31 46 d2"},
{"VPSRAVD Y11, Y15, Y11", "c4 42 05 46 db"},
{"VPSRAVD Y11, Y15, Y2", "c4 c2 05 46 d3"},
{"VPSRAVD Y2, Y15, Y11", "c4 62 05 46 da"},
{"VPSRAVD Y2, Y15, Y2", "c4 e2 05 46 d2"},
{"VPSRLVD (BX), X9, X11", "c4 62 31 45 1b"},
{"VPSRLVD (BX), X9, X2", "c4 e2 31 45 13"},
{"VPSRLVD (BX), Y15, Y11", "c4 62 05 45 1b"},
{"VPSRLVD (BX), Y15, Y2", "c4 e2 05 45 13"},
{"VPSRLVD (R11), X9, X11", "c4 42 31 45 1b"},
{"VPSRLVD (R11), X9, X2", "c4 c2 31 45 13"},
{"VPSRLVD (R11), Y15, Y11", "c4 42 05 45 1b"},
{"VPSRLVD (R11), Y15, Y2", "c4 c2 05 45 13"},
{"VPSRLVD X11, X9, X11", "c4 42 31 45 db"},
{"VPSRLVD X11, X9, X2", "c4 c2 31 45 d3"},
{"VPSRLVD X2, X9, X11", "c4 62 31 45 da"},
{"VPSRLVD X2, X9, X2", "c4 e2 31 45 d2"},
{"VPSRLVD Y11, Y15, Y11", "c4 42 05 45 db"},
{"VPSRLVD Y11, Y15, Y2", "c4 c2 05 45 d3"},
{"VPSRLVD Y2, Y15, Y11", "c4 62 05 45 da"},
{"VPSRLVD Y2, Y15, Y2", "c4 e2 05 45 d2"},
{"VPSRLVQ (BX), X9, X11", "c4 62 b1 45 1b"},
{"VPSRLVQ (BX), X9, X2", "c4 e2 b1 45 13"},
{"VPSRLVQ (BX), Y15, Y11", "c4 62 85 45 1b"},
{"VPSRLVQ (BX), Y15, Y2", "c4 e2 85 45 13"},
{"VPSRLVQ (R11), X9, X11", "c4 42 b1 45 1b"},
{"VPSRLVQ (R11), X9, X2", "c4 c2 b1 45 13"},
{"VPSRLVQ (R11), Y15, Y11", "c4 42 85 45 1b"},
{"VPSRLVQ (R11), Y15, Y2", "c4 c2 85 45 13"},
{"VPSRLVQ X11, X9, X11", "c4 42 b1 45 db"},
{"VPSRLVQ X11, X9, X2", "c4 c2 b1 45 d3"},
{"VPSRLVQ X2, X9, X11", "c4 62 b1 45 da"},
{"VPSRLVQ X2, X9, X2", "c4 e2 b1 45 d2"},
{"VPSRLVQ Y11, Y15, Y11", "c4 42 85 45 db"},
{"VPSRLVQ Y11, Y15, Y2", "c4 c2 85 45 d3"},
{"VPSRLVQ Y2, Y15, Y11", "c4 62 85 45 da"},
{"VPSRLVQ Y2, Y15, Y2", "c4 e2 85 45 d2"},
{"VPSUBB (BX), X9, X11", "c4 61 31 f8 1b|c5 31 f8 1b"},
{"VPSUBB (BX), X9, X2", "c4 e1 31 f8 13|c5 b1 f8 13"},
{"VPSUBB (BX), Y15, Y11", "c4 61 05 f8 1b|c5 05 f8 1b"},
{"VPSUBB (BX), Y15, Y2", "c4 e1 05 f8 13|c5 85 f8 13"},
{"VPSUBB (R11), X9, X11", "c4 41 31 f8 1b"},
{"VPSUBB (R11), X9, X2", "c4 c1 31 f8 13"},
{"VPSUBB (R11), Y15, Y11", "c4 41 05 f8 1b"},
{"VPSUBB (R11), Y15, Y2", "c4 c1 05 f8 13"},
{"VPSUBB X11, X9, X11", "c4 41 31 f8 db"},
{"VPSUBB X11, X9, X2", "c4 c1 31 f8 d3"},
{"VPSUBB X2, X9, X11", "c4 61 31 f8 da|c5 31 f8 da"},
{"VPSUBB X2, X9, X2", "c4 e1 31 f8 d2|c5 b1 f8 d2"},
{"VPSUBB Y11, Y15, Y11", "c4 41 05 f8 db"},
{"VPSUBB Y11, Y15, Y2", "c4 c1 05 f8 d3"},
{"VPSUBB Y2, Y15, Y11", "c4 61 05 f8 da|c5 05 f8 da"},
{"VPSUBB Y2, Y15, Y2", "c4 e1 05 f8 d2|c5 85 f8 d2"},
{"VPSUBW (BX), X9, X11", "c4 61 31 f9 1b|c5 31 f9 1b"},
{"VPSUBW (BX), X9, X2", "c4 e1 31 f9 13|c5 b1 f9 13"},
{"VPSUBW (BX), Y15, Y11", "c4 61 05 f9 1b|c5 05 f9 1b"},
{"VPSUBW (BX), Y15, Y2", "c4 e1 05 f9 13|c5 85 f9 13"},
{"VPSUBW (R11), X9, X11", "c4 41 31 f9 1b"},
{"VPSUBW (R11), X9, X2", "c4 c1 31 f9 13"},
{"VPSUBW (R11), Y15, Y11", "c4 41 05 f9 1b"},
{"VPSUBW (R11), Y15, Y2", "c4 c1 05 f9 13"},
{"VPSUBW X11, X9, X11", "c4 41 31 f9 db"},
{"VPSUBW X11, X9, X2", "c4 c1 31 f9 d3"},
{"VPSUBW X2, X9, X11", "c4 61 31 f9 da|c5 31 f9 da"},
{"VPSUBW X2, X9, X2", "c4 e1 31 f9 d2|c5 b1 f9 d2"},
{"VPSUBW Y11, Y15, Y11", "c4 41 05 f9 db"},
{"VPSUBW Y11, Y15, Y2", "c4 c1 05 f9 d3"},
{"VPSUBW Y2, Y15, Y11", "c4 61 05 f9 da|c5 05 f9 da"},
{"VPSUBW Y2, Y15, Y2", "c4 e1 05 f9 d2|c5 85 f9 d2"},
{"VSHUFPS $7, (BX), X9, X11", "c4 61 30 c6 1b 07|c5 30 c6 1b 07"},
{"VSHUFPS $7, (BX), X9, X2", "c4 e1 30 c6 13 07|c5 b0 c6 13 07"},
{"VSHUFPS $7, (BX), Y15, Y11", "c4 61 04 c6 1b 07|c5 04 c6 1b 07"},
{"VSHUFPS $7, (BX), Y15, Y2", "c4 e1 04 c6 13 07|c5 84 c6 13 07"},
{"VSHUFPS $7, (R11), X9, X11", "c4 41 30 c6 1b 07"},
{"VSHUFPS $7, (R11), X9, X2", "c4 c1 30 c6 13 07"},
{"VSHUFPS $7, (R11), Y15, Y11", "c4 41 04 c6 1b 07"},
{"VSHUFPS $7, (R11), Y15, Y2", "c4 c1 04 c6 13 07"},
{"VSHUFPS $7, X11, X9, X11", "c4 41 30 c6 db 07"},
{"VSHUFPS $7, X11, X9, X2", "c4 c1 30 c6 d3 07"},
{"VSHUFPS $7, X2, X9, X11", "c4 61 30 c6 da 07|c5 30 c6 da 07"},
{"VSHUFPS $7, X2, X9, X2", "c4 e1 30 c6 d2 07|c5 b0 c6 d2 07"},
{"VSHUFPS $7, Y11, Y15, Y11", "c4 41 04 c6 db 07"},
{"VSHUFPS $7, Y11, Y15, Y2", "c4 c1 04 c6 d3 07"},
{"VSHUFPS $7, Y2, Y15, Y11", "c4 61 04 c6 da 07|c5 04 c6 da 07"},
{"VSHUFPS $7, Y2, Y15, Y2", "c4 e1 04 c6 d2 07|c5 84 c6 d2 07"},
}
// TestAmd64VexParityCorpus assembles every corpus line and requires the
// bytes to match one of the alternatives go tool asm accepts.
func TestAmd64VexParityCorpus(t *testing.T) {
for _, tc := range amd64VexParityCorpus {
fn := firstText(t, "TEXT ·p(SB), 4, $0\n\t"+tc.line+"\n")
code, _, err := Assemble(fn)
if err != nil {
t.Errorf("%s: %v", tc.line, err)
continue
}
got := hexBytes(code)
if !slices.Contains(strings.Split(tc.want, "|"), got) {
t.Errorf("%s: got %s, want one of %s", tc.line, got, tc.want)
}
}
}
+202
View File
@@ -0,0 +1,202 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/hex"
"os"
"path/filepath"
"testing"
)
// The byte forms of the suffixed scalar families answer to the register
// operands as much as to the mnemonic: the toolchain's own disassembly
// prints them with the L suffix or none at all (the rendered suffix rides
// the operand-size attribute), the register names carrying the width. The
// tests here pin that reconciliation: the renderer's spellings encode the
// byte form byte for byte, the W and Q spellings never ride a byte
// register, and the classic names stay size-agnostic.
// TestOperandWidthByteForms pins the renderer's spellings: an L suffix or
// no suffix with a byte-spelled register encodes the 8-bit form, every
// register joining at its low byte. Every row's bytes were cross-checked
// against `go tool asm -S` over the B-suffixed spelling of the same
// operands (the toolchain rejects the L spelling itself), and against the
// toolchain's objdump text for the byte encodings.
func TestOperandWidthByteForms(t *testing.T) {
r11 := Reg{idx: 11, size: 8}
r8 := Reg{idx: 8, size: 1}
r9 := Reg{idx: 9, size: 1}
for _, tt := range []struct {
name string
ops []Operand
want string
}{
// The atomics pair: XADD and CMPXCHG drop to the 0F C0/0F B0 byte
// opcodes the L spelling would otherwise widen past.
{"XADDL DL, DL", []Operand{DL, DL}, "0fc0d2"},
{"XADDB DL, DL", []Operand{DL, DL}, "0fc0d2"},
{"XADDL R8B, R9B", []Operand{r8, r9}, "450fc0c1"},
{"CMPXCHGL DL, DL", []Operand{DL, DL}, "0fb0d2"},
{"XCHGL DL, DL", []Operand{DL, DL}, "86d2"},
{"XCHGL DL, 0(BX)", []Operand{DL, Ptr(BX, 0, 1)}, "8613"},
// The ALU immediates: the accumulator short form for the AL
// spelling, the generic 0x80 /digit elsewhere.
{"CMPL AL, $7", []Operand{AL, Imm(7)}, "3c07"},
{"ADDL $7, AL", []Operand{Imm(7), AL}, "0407"},
{"SUBL $7, AL", []Operand{Imm(7), AL}, "2c07"},
{"ANDL $7, AL", []Operand{Imm(7), AL}, "2407"},
{"SBBL $7, AL", []Operand{Imm(7), AL}, "1c07"},
{"SBBL $7, DL", []Operand{Imm(7), DL}, "80da07"},
{"ADDB $3, AX", []Operand{Imm(3), AX}, "80c003"},
{"ORB $7, AX", []Operand{Imm(7), AX}, "80c807"},
{"SBBB $7, AX", []Operand{Imm(7), AX}, "80d807"},
// The register forms, the extended register riding REX.B and the
// byte source in the reg field.
{"SBBL DL, R11", []Operand{DL, r11}, "4118d3"},
{"TESTL R11, DL", []Operand{r11, DL}, "4484da"},
{"TESTB $7, AX", []Operand{Imm(7), AX}, "f6c007"},
{"TESTB R11, DL", []Operand{r11, DL}, "4484da"},
// CRC32 keeps the F0 byte opcode under the unsuffixed spelling.
{"CRC32 DL, R11", []Operand{DL, r11}, "f2440f38f0da"},
{"CRC32B DL, R11", []Operand{DL, r11}, "f2440f38f0da"},
{"CRC32B AX, CX", []Operand{AX, CX}, "f20f38f0c8"},
// The move and unary families follow the same rule.
{"MOVL $7, DL", []Operand{Imm(7), DL}, "b207"},
{"MOVB $7, DL", []Operand{Imm(7), DL}, "b207"},
{"MOVB AX, AL", []Operand{AX, AL}, "88c0"},
{"INCL DL", []Operand{DL}, "fec2"},
{"NEGL DL", []Operand{DL}, "f6da"},
{"IMULL DL", []Operand{DL}, "f6ea"},
{"SHLL $2, DL", []Operand{Imm(2), DL}, "c0e202"},
{"ROLL CL, DL", []Operand{CL, DL}, "d2c2"},
// The shift count never narrows the shifted value.
{"RCLW CL, 0(R11)", []Operand{CL, Ptr(r11, 0, 2)}, "6641d313"},
{"RORQ CL, AX", []Operand{CL, AX}, "48d3c8"},
// The size-agnostic spellings stay width-free: the L and Q forms
// of the same families are untouched by the reconciliation.
{"XADDL AX, CX", []Operand{AX, CX}, "0fc1c1"},
{"ADDL $7, AX", []Operand{Imm(7), AX}, "83c007"},
{"ADDL $256, AX", []Operand{Imm(256), AX}, "0500010000"},
{"CRC32L AX, CX", []Operand{AX, CX}, "f20f38f1c8"},
} {
t.Run(tt.name, func(t *testing.T) {
got, err := Encode(mnemonicOf(tt.name), tt.ops...)
if err != nil {
t.Fatalf("%s: %v", tt.name, err)
}
if want := unhex(tt.want); !bytes.Equal(got, want) {
t.Errorf("%s: % x, want % x", tt.name, got, want)
}
})
}
}
// TestOperandWidthConflicts pins the refusals: the W and Q spellings never
// ride a byte register (go tool asm rejects MOVQ AL, AX and its siblings
// outright), and neither does the MOVD alias of the quad move.
func TestOperandWidthConflicts(t *testing.T) {
for _, tt := range []struct {
name string
ops []Operand
}{
{"MOVQ AL, AX", []Operand{AL, AX}},
{"MOVQ AX, AL", []Operand{AX, AL}},
{"MOVQ DL, DL", []Operand{DL, DL}},
{"MOVW $7, DL", []Operand{Imm(7), DL}},
{"MOVD AL, AX", []Operand{AL, AX}},
{"XADDQ DL, DL", []Operand{DL, DL}},
{"CMPXCHGQ DL, DL", []Operand{DL, DL}},
{"XCHGQ DL, DL", []Operand{DL, DL}},
{"SHLQ $2, DL", []Operand{Imm(2), DL}},
{"INCQ DL", []Operand{DL}},
{"IMULQ DL", []Operand{DL}},
{"TESTQ R11, DL", []Operand{Reg{idx: 11, size: 8}, DL}},
{"CRC32Q DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}},
{"CRC32W DL, R11", []Operand{DL, Reg{idx: 11, size: 8}}},
} {
t.Run(tt.name, func(t *testing.T) {
if _, err := Encode(mnemonicOf(tt.name), tt.ops...); err == nil {
t.Errorf("%s: encoded, want the byte-register conflict refused", tt.name)
}
})
}
}
// TestOperandWidthDifferential is the byte-parity oracle for the same
// reconciliation: the B-suffixed spellings, which go tool asm accepts, must
// encode identically through both assemblers, the agnostic names included
// (their low byte joins the byte form) and the accumulator division (AL
// short, AX generic) with them.
func TestOperandWidthDifferential(t *testing.T) {
kernel := "#include \"textflag.h\"\n" +
"TEXT \u00b7bytewidth(SB), NOSPLIT, $0\n" +
"\tXADDB DL, DL\n" +
"\tXADDB R8B, R9B\n" +
"\tXADDL AX, CX\n" +
"\tCMPXCHGB DL, DL\n" +
"\tXCHGB DL, DL\n" +
"\tXCHGB DL, 0(BX)\n" +
"\tCMPB AL, $7\n" +
"\tADDB $7, AL\n" +
"\tADDB $3, AX\n" +
"\tSUBB $7, AL\n" +
"\tANDB $7, AL\n" +
"\tSBBB $7, AL\n" +
"\tSBBB $7, DL\n" +
"\tSBBB DL, R11\n" +
"\tORB $7, AX\n" +
"\tTESTB R11, DL\n" +
"\tTESTB $7, AX\n" +
"\tCRC32B DL, R11\n" +
"\tCRC32B AX, CX\n" +
"\tCRC32B R8B, CX\n" +
"\tCRC32L AX, CX\n" +
"\tMOVB $7, DL\n" +
"\tMOVB $3, AX\n" +
"\tMOVB AX, AL\n" +
"\tINCB DL\n" +
"\tNEGB DL\n" +
"\tIMULB DL\n" +
"\tMULB CL\n" +
"\tSHLB $2, DL\n" +
"\tROLB CL, DL\n" +
"\tRET\n"
path := filepath.Join(t.TempDir(), "bytewidth_amd64.s")
if err := os.WriteFile(path, []byte(kernel), 0o644); err != nil {
t.Fatal(err)
}
gt := oracleFuncCode(t, toolAsmObject(t, path, ""))
goCode, ok := gt["bytewidth.bytewidth"]
if !ok {
t.Fatalf("oracle: bytewidth function missing (%d functions)", len(gt))
}
gasmCode := code(path, kernel)
if gasmCode == nil {
t.Fatal("gasm: assemble failed")
}
if !bytes.Equal(gasmCode, goCode) {
t.Errorf("byte width kernel:\ngasm %x\ngo %x", gasmCode, goCode)
}
}
// mnemonicOf returns the first whitespace-free token of a rendered row.
func mnemonicOf(text string) string {
for i := 0; i < len(text); i++ {
if text[i] == ' ' || text[i] == ',' {
return text[:i]
}
}
return text
}
// unhex decodes a hex string, failing the test on malformed input.
func unhex(s string) []byte {
out, err := hex.DecodeString(s)
if err != nil {
panic("unhex: " + err.Error())
}
return out
}
+237
View File
@@ -0,0 +1,237 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "fmt"
// This file implements the x87 floating-point family the Go assembler
// carries: the no-operand stack controls, the two-register arithmetic pair,
// the register compares, the memory loads and stores and the FXSAVE pair.
// Every encoding here is pinned byte for byte against go tool asm through
// the corpus lines in amd64_x87_test.go.
// x87NoOperand maps the no-operand x87 instruction to its postfix byte
// inside the D9 escape: D9 <postfix>, no ModR/M, no operand.
var x87NoOperand = map[string]byte{
"F2XM1": 0xF0,
"FABS": 0xE1,
"FCHS": 0xE0,
"FCOS": 0xFF,
"FDECSTP": 0xF6,
"FINCSTP": 0xF7,
"FLD1": 0xE8,
"FLDL2E": 0xEA,
"FLDL2T": 0xE9,
"FLDLG2": 0xEC,
"FLDPI": 0xEB,
"FNOP": 0xD0,
"FPATAN": 0xF3,
"FPREM": 0xF8,
"FPREM1": 0xF5,
"FPTAN": 0xF2,
"FRNDINT": 0xFC,
"FSCALE": 0xFD,
"FSIN": 0xFE,
"FSINCOS": 0xFB,
"FSQRT": 0xFA,
"FTST": 0xE4,
"FXAM": 0xE5,
"FXTRACT": 0xF4,
"FYL2X": 0xF1,
"FYL2XP1": 0xF9,
}
// x87ArithSpec describes one member of the D8/DC two-register arithmetic
// pair. The D8 form reads ST(0) as its second operand (FADDD F2, F0), the
// DC form ST(0) as its first (FADDD F0, F2) or a memory source (FADDD (BX),
// F0). FDIV is the odd member: with ST(0) as the first operand the toolchain
// assembles the reversed register form (DC F8+i, the FDIVR digit), so the DC
// digit differs from the memory digit there.
type x87ArithSpec struct {
d8Digit int // the /digit of the D8 form (dst = ST(0))
dcDigit int // the /digit of the DC register form (src = ST(0))
memDigit int // the /digit of the DC memory form (dst = ST(0))
}
// x87Arith maps the arithmetic mnemonics to their digits.
var x87Arith = map[string]x87ArithSpec{
"FADDD": {0, 0, 0},
"FCOMD": {2, 2, 2},
"FDIVD": {6, 7, 6},
}
// x87FCmov maps the conditional x87 moves to their escape byte and postfix
// base (the C0/C8/D0/D8 group the condition selects); the compared register
// rides the postfix's low three bits.
var x87FCmov = map[string][2]byte{
"FCMOVB": {0xDA, 0xC0},
"FCMOVBE": {0xDA, 0xD0},
"FCMOVE": {0xDA, 0xC8},
"FCMOVNB": {0xDB, 0xC0},
"FCMOVNBE": {0xDB, 0xD0},
"FCMOVNE": {0xDB, 0xC8},
"FCMOVNU": {0xDB, 0xD8},
"FCMOVU": {0xDA, 0xD8},
}
// x87Compare maps the register compare pair to their escape byte; the
// register form is escape F0+i (mod 11, reg 110, rm = the compared stack
// register), ST(0) fixed as the second operand.
var x87Compare = map[string]byte{
"FCOMI": 0xDB,
"FCOMIP": 0xDF,
}
// x87MemUnary maps the one-memory-operand x87 controls to their escape byte
// and /digit.
var x87MemUnary = map[string]struct {
escape byte
digit int
}{
"FBLD": {0xDF, 4},
"FBSTP": {0xDF, 6},
"FLDCW": {0xD9, 5},
}
// x87Fxsav maps the FXSAVE pair to their /digit in the 0F AE group; the 64
// spellings carry REX.W.
var x87Fxsav = map[string]struct {
digit int
rexW bool
}{
"FXSAVE": {0, false},
"FXSAVE64": {0, true},
"FXRSTOR": {1, false},
"FXRSTOR64": {1, true},
}
// encodeX87 encodes the x87 family. It reports whether the mnemonic belongs
// to the family; a false result hands the mnemonic back to the caller, an
// error result a failed attempt to encode it.
func (e *enc) encodeX87(upper string, ops []Operand) (bool, error) {
if post, ok := x87NoOperand[upper]; ok {
if len(ops) != 0 {
return true, fmt.Errorf("%s takes no operands, got %d", upper, len(ops))
}
return true, e.emit(&instr{opcode: []byte{0xD9, post}, modrm: -1, sib: -1})
}
if spec, ok := x87Arith[upper]; ok {
return true, e.encodeX87Arith(upper, spec, ops)
}
if esc, ok := x87FCmov[upper]; ok {
src, _, err := x87PairOperands(upper, ops)
if err != nil {
return true, err
}
// The condition applies between the named register and ST(0), so the
// second operand is always F0; the first rides the postfix's low bits.
return true, e.emit(&instr{opcode: []byte{esc[0], esc[1] | byte(src.idx&7)}, modrm: -1, sib: -1})
}
if escape, ok := x87Compare[upper]; ok {
if _, _, err := x87PairOperands(upper, ops); err != nil {
return true, err
}
src, _ := ops[0].(Reg)
return true, e.emit(&instr{opcode: []byte{escape, 0xF0 | byte(src.idx&7)}, modrm: -1, sib: -1})
}
if upper == "FADDDP" {
if len(ops) != 2 {
return true, fmt.Errorf("FADDDP expects 2 operands, got %d", len(ops))
}
first, ok1 := ops[0].(Reg)
second, ok2 := ops[1].(Reg)
if !ok1 || !ok2 || !first.fp || !second.fp {
return true, fmt.Errorf("FADDDP takes two x87 stack registers")
}
if first.idx != 0 {
return true, fmt.Errorf("FADDDP: the first operand must be F0")
}
return true, e.emit(&instr{opcode: []byte{0xDE, 0xC0 | byte(second.idx&7)}, modrm: -1, sib: -1})
}
if m, ok := x87MemUnary[upper]; ok {
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 memory operand, got %d", upper, len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("%s requires a memory operand", upper)
}
i := &instr{opcode: []byte{m.escape}, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
if m, ok := x87Fxsav[upper]; ok {
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 memory operand, got %d", upper, len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("%s requires a memory operand", upper)
}
i := newInstr(0, []byte{0x0F, 0xAE})
i.rexW = m.rexW
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
return false, nil
}
// encodeX87Arith encodes one member of the D8/DC arithmetic pair. The tool-
// chain's shape set: (Fn, F0) rides D8, (F0, Fn) rides DC, ((m), F0) rides
// the DC memory digit; every other pairing is an error.
func (e *enc) encodeX87Arith(mnem string, spec x87ArithSpec, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, dst := ops[0], ops[1]
dstReg, dstIsF := dst.(Reg)
if !dstIsF || !dstReg.fp {
return fmt.Errorf("%s: the destination must be an x87 stack register", mnem)
}
switch s := src.(type) {
case Reg:
if !s.fp {
return fmt.Errorf("%s: the source must be an x87 stack register", mnem)
}
switch {
case dstReg.idx == 0:
return e.emit(&instr{opcode: []byte{0xD8, 0xC0 | byte(spec.d8Digit)<<3 | byte(s.idx&7)}, modrm: -1, sib: -1})
case s.idx == 0:
return e.emit(&instr{opcode: []byte{0xDC, 0xC0 | byte(spec.dcDigit)<<3 | byte(dstReg.idx&7)}, modrm: -1, sib: -1})
default:
return fmt.Errorf("%s: one operand must be F0", mnem)
}
default:
if !isX86Mem(src) {
return fmt.Errorf("%s: the source must be an x87 stack register or memory", mnem)
}
if dstReg.idx != 0 {
return fmt.Errorf("%s: the destination must be F0 with a memory source", mnem)
}
i := &instr{opcode: []byte{0xDC}, modrm: -1, sib: -1}
if err := setRMDigit(i, spec.memDigit, src, 8); err != nil {
return err
}
return e.emit(i)
}
}
// x87PairOperands validates the (register, F0) shape the conditional moves
// and compares take and returns the two registers.
func x87PairOperands(mnem string, ops []Operand) (Reg, Reg, error) {
if len(ops) != 2 {
return Reg{}, Reg{}, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, ok1 := ops[0].(Reg)
dst, ok2 := ops[1].(Reg)
if !ok1 || !ok2 || !src.fp || !dst.fp {
return Reg{}, Reg{}, fmt.Errorf("%s takes two x87 stack registers", mnem)
}
if dst.idx != 0 {
return Reg{}, Reg{}, fmt.Errorf("%s: the second operand must be F0", mnem)
}
return src, dst, nil
}
+110
View File
@@ -0,0 +1,110 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "testing"
// amd64X87Corpus holds every line the Go toolchain's own
// amd64enc.s carries for the x87 family (stack controls, the arithmetic pair, the compares, the memory loads and the FXSAVE pair), with the bytes go tool asm
// emits for each: the differential ground truth the family is proven
// against, line for line.
var amd64X87Corpus = []struct {
line string
want string
}{
{"F2XM1", "d9 f0"},
{"FABS", "d9 e1"},
{"FADDD F2, F0", "d8 c2"},
{"FADDD F3, F0", "d8 c3"},
{"FADDD F0, F2", "dc c2"},
{"FADDD F0, F3", "dc c3"},
{"FADDD (BX), F0", "dc 03"},
{"FADDD (R11), F0", "41 dc 03"},
{"FADDDP F0, F2", "de c2"},
{"FADDDP F0, F3", "de c3"},
{"FBLD (BX)", "df 23"},
{"FBLD (R11)", "41 df 23"},
{"FBSTP (BX)", "df 33"},
{"FBSTP (R11)", "41 df 33"},
{"FCHS", "d9 e0"},
{"FCMOVB F2, F0", "da c2"},
{"FCMOVB F3, F0", "da c3"},
{"FCMOVBE F2, F0", "da d2"},
{"FCMOVBE F3, F0", "da d3"},
{"FCMOVE F2, F0", "da ca"},
{"FCMOVE F3, F0", "da cb"},
{"FCMOVNB F2, F0", "db c2"},
{"FCMOVNB F3, F0", "db c3"},
{"FCMOVNBE F2, F0", "db d2"},
{"FCMOVNBE F3, F0", "db d3"},
{"FCMOVNE F2, F0", "db ca"},
{"FCMOVNE F3, F0", "db cb"},
{"FCMOVNU F2, F0", "db da"},
{"FCMOVNU F3, F0", "db db"},
{"FCMOVU F2, F0", "da da"},
{"FCMOVU F3, F0", "da db"},
{"FCOMD F2, F0", "d8 d2"},
{"FCOMD F3, F0", "d8 d3"},
{"FCOMD (BX), F0", "dc 13"},
{"FCOMD (R11), F0", "41 dc 13"},
{"FCOMI F2, F0", "db f2"},
{"FCOMI F3, F0", "db f3"},
{"FCOMIP F2, F0", "df f2"},
{"FCOMIP F3, F0", "df f3"},
{"FCOS", "d9 ff"},
{"FDECSTP", "d9 f6"},
{"FDIVD F2, F0", "d8 f2"},
{"FDIVD F3, F0", "d8 f3"},
{"FDIVD F0, F2", "dc fa"},
{"FDIVD F0, F3", "dc fb"},
{"FDIVD (BX), F0", "dc 33"},
{"FDIVD (R11), F0", "41 dc 33"},
{"FINCSTP", "d9 f7"},
{"FLD1", "d9 e8"},
{"FLDCW (BX)", "d9 2b"},
{"FLDCW (R11)", "41 d9 2b"},
{"FLDL2E", "d9 ea"},
{"FLDL2T", "d9 e9"},
{"FLDLG2", "d9 ec"},
{"FLDPI", "d9 eb"},
{"FNOP", "d9 d0"},
{"FPATAN", "d9 f3"},
{"FPREM", "d9 f8"},
{"FPREM1", "d9 f5"},
{"FPTAN", "d9 f2"},
{"FRNDINT", "d9 fc"},
{"FSCALE", "d9 fd"},
{"FSIN", "d9 fe"},
{"FSINCOS", "d9 fb"},
{"FSQRT", "d9 fa"},
{"FTST", "d9 e4"},
{"FXAM", "d9 e5"},
{"FXRSTOR (BX)", "0f ae 0b"},
{"FXRSTOR (R11)", "41 0f ae 0b"},
{"FXRSTOR64 (BX)", "48 0f ae 0b"},
{"FXRSTOR64 (R11)", "49 0f ae 0b"},
{"FXSAVE (BX)", "0f ae 03"},
{"FXSAVE (R11)", "41 0f ae 03"},
{"FXSAVE64 (BX)", "48 0f ae 03"},
{"FXSAVE64 (R11)", "49 0f ae 03"},
{"FXTRACT", "d9 f4"},
{"FYL2X", "d9 f1"},
{"FYL2XP1", "d9 f9"},
}
// TestAmd64X87Corpus assembles every corpus line and requires the same bytes
// go tool asm emits for it.
func TestAmd64X87Corpus(t *testing.T) {
for _, tc := range amd64X87Corpus {
fn := firstText(t, "TEXT ·p(SB), 4, $0\n\t"+tc.line+"\n")
code, _, err := Assemble(fn)
if err != nil {
t.Errorf("%s: %v", tc.line, err)
continue
}
if got := hexBytes(code); got != tc.want {
t.Errorf("%s: got %s, want %s", tc.line, got, tc.want)
}
}
}
+58
View File
@@ -0,0 +1,58 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"strings"
)
// This file implements the XSAVE family: the extended-state save and restore
// pair in its four generations (XSAVE/XRSTOR, XSAVEOPT, XSAVEC, XSAVES) and
// their 64 spellings. Each takes one memory operand alone; the 64 spellings
// carry REX.W. Every encoding here is pinned byte for byte against
// go tool asm through the corpus lines in amd64_xsave_test.go.
// xsaveSpec is one XSAVE family member: the opcode group and the /digit.
// XSAVEOPT carries no mandatory prefix in the toolchain's encoding despite
// the manual's 0x66, so the family has none anywhere.
type xsaveSpec struct {
op []byte
digit int
}
// xsaveTable maps the save/restore mnemonics to their encodings. The 64
// spellings share the base's digit with REX.W.
var xsaveTable = map[string]xsaveSpec{
"XSAVE": {[]byte{0x0F, 0xAE}, 4},
"XRSTOR": {[]byte{0x0F, 0xAE}, 5},
"XSAVEOPT": {[]byte{0x0F, 0xAE}, 6},
"XSAVEC": {[]byte{0x0F, 0xC7}, 4},
"XSAVES": {[]byte{0x0F, 0xC7}, 5},
"XRSTORS": {[]byte{0x0F, 0xC7}, 3},
}
// encodeXsave encodes the XSAVE family. It reports whether the mnemonic
// belongs to the family.
func (e *enc) encodeXsave(upper string, ops []Operand) (bool, error) {
name, rexW := upper, false
if base, ok := strings.CutSuffix(upper, "64"); ok {
name, rexW = base, true
}
spec, ok := xsaveTable[name]
if !ok {
return false, nil
}
if len(ops) != 1 {
return true, fmt.Errorf("%s expects 1 memory operand, got %d", upper, len(ops))
}
if !isX86Mem(ops[0]) {
return true, fmt.Errorf("%s requires a memory operand", upper)
}
i := &instr{opcode: append([]byte(nil), spec.op...), modrm: -1, sib: -1, rexW: rexW}
if err := setRMDigit(i, spec.digit, ops[0], 8); err != nil {
return true, err
}
return true, e.emit(i)
}
+56
View File
@@ -0,0 +1,56 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "testing"
// amd64XsaveCorpus holds every line the Go toolchain's own
// amd64enc.s carries for the XSAVE family (XSAVE, XSAVEOPT, XSAVEC, XSAVES and the restores, plain and 64), with the bytes go tool asm
// emits for each: the differential ground truth the family is proven
// against, line for line.
var amd64XsaveCorpus = []struct {
line string
want string
}{
{"XRSTOR (BX)", "0f ae 2b"},
{"XRSTOR (R11)", "41 0f ae 2b"},
{"XRSTOR64 (BX)", "48 0f ae 2b"},
{"XRSTOR64 (R11)", "49 0f ae 2b"},
{"XRSTORS (BX)", "0f c7 1b"},
{"XRSTORS (R11)", "41 0f c7 1b"},
{"XRSTORS64 (BX)", "48 0f c7 1b"},
{"XRSTORS64 (R11)", "49 0f c7 1b"},
{"XSAVE (BX)", "0f ae 23"},
{"XSAVE (R11)", "41 0f ae 23"},
{"XSAVE64 (BX)", "48 0f ae 23"},
{"XSAVE64 (R11)", "49 0f ae 23"},
{"XSAVEC (BX)", "0f c7 23"},
{"XSAVEC (R11)", "41 0f c7 23"},
{"XSAVEC64 (BX)", "48 0f c7 23"},
{"XSAVEC64 (R11)", "49 0f c7 23"},
{"XSAVEOPT (BX)", "0f ae 33"},
{"XSAVEOPT (R11)", "41 0f ae 33"},
{"XSAVEOPT64 (BX)", "48 0f ae 33"},
{"XSAVEOPT64 (R11)", "49 0f ae 33"},
{"XSAVES (BX)", "0f c7 2b"},
{"XSAVES (R11)", "41 0f c7 2b"},
{"XSAVES64 (BX)", "48 0f c7 2b"},
{"XSAVES64 (R11)", "49 0f c7 2b"},
}
// TestAmd64XsaveCorpus assembles every corpus line and requires the same bytes
// go tool asm emits for it.
func TestAmd64XsaveCorpus(t *testing.T) {
for _, tc := range amd64XsaveCorpus {
fn := firstText(t, "TEXT ·p(SB), 4, $0\n\t"+tc.line+"\n")
code, _, err := Assemble(fn)
if err != nil {
t.Errorf("%s: %v", tc.line, err)
continue
}
if got := hexBytes(code); got != tc.want {
t.Errorf("%s: got %s, want %s", tc.line, got, tc.want)
}
}
}
+2236 -245
View File
File diff suppressed because it is too large. Load diff
+293 -128
View File
@@ -33,7 +33,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
@@ -374,6 +374,7 @@ const (
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCASP // compare and swap pair: CASP
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
@@ -384,6 +385,7 @@ const (
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
@@ -563,16 +565,12 @@ func init() {
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0xd3400000}
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
fp3 := map[string]uint32{
@@ -597,6 +595,10 @@ func init() {
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
// The half-precision conversions: FPOP1S rows with the half source
// or destination (type 3), single and double beside them.
"FCVTHS": 0x1ee24000, "FCVTHD": 0x1ee2c000,
"FCVTSH": 0x1e23c000, "FCVTDH": 0x1e63c000,
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
@@ -770,7 +772,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -783,12 +785,20 @@ func init() {
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
// The scale column of a64LoadTable does not reach the pair forms, so each
// entry states its own access width through the imm7 divisor the pair
// encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q).
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000}
a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000}
a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000}
a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
@@ -806,11 +816,23 @@ func init() {
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"CASB": 0x08a07c00,
"CASAB": 0x08e07c00,
"CASH": 0x48a07c00,
"CASLD": 0xc8a0fc00,
"CASLH": 0x48a0fc00,
"CASAW": 0x88e07c00,
"CASAD": 0xc8e07c00,
"CASALH": 0x48e0fc00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDADDAD": 0xf8a00000,
"LDADDAW": 0xb8a00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDCLRAD": 0xf8a01000,
"LDCLRAW": 0xb8a01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
@@ -847,7 +869,9 @@ func init() {
"LDEORAD": 0xf8a02000,
"LDEORAH": 0x78a02000,
"LDEORALB": 0x38e02000,
"LDEORALD": 0xf8e02000,
"LDEORALH": 0x78e02000,
"LDEORALW": 0xb8e02000,
"LDEORAW": 0xb8a02000,
"LDEORB": 0x38202000,
"LDEORD": 0xf8202000,
@@ -889,6 +913,12 @@ func init() {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// Compare and swap pair: the second register of each pair is implicit
// (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset
// fixed field (asm7.go atomicCASP).
a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10}
a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
@@ -940,11 +970,13 @@ func init() {
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10}
a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
@@ -957,16 +989,37 @@ func init() {
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
// Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and
// the replicate loads VLD2R/VLD3R, each with the post-index spelling.
a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
// for instructions that exist at a single arrangement and carry that
// arrangement's bits inside the base already, the fixed flag.
// the set of arrangements it accepts as a bitmask over the a64Arr index,
// the fixed flag for instructions that exist at a single arrangement and
// carry that arrangement's bits inside the base already, and the fp flag for
// the FP rows, whose size field is the single FP bit (a64FPArrBits) instead
// of the integer size.
type a64SimdVSpec struct {
base uint32
arrs uint16
fixed bool
fp bool
}
// a64Arr names the vector arrangements the encoders deal with, indexed by
@@ -1013,11 +1066,24 @@ func a64ElemLetter(s string) bool {
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// fpSimdArrs bounds the arrangements the FP SIMD forms accept: S and D only,
// the toolchain rejecting the half-width spellings outright ("invalid
// arrangement"). fpAcrossArrs bounds the across-vector reductions, which do
// take the half width.
var fpSimdArrs = uint16(1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1 << a64Arr4S)
// a64FPArrBits carries the bits an arrangement contributes to the FP SIMD
// words: the FP size field is a single bit at bit 22 (0 for the S widths, 1
// for the D widths; the toolchain carries no half-width FP rows) and the
// 128-bit flag sits at bit 30. Word-verified against go tool asm.
var a64FPArrBits = [a64ArrCount]uint32{
a64Arr2S: 0,
a64Arr4S: 1 << 30,
a64Arr2D: 1<<30 | 1<<22,
a64Arr4H: 0,
a64Arr8H: 1 << 30,
}
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
@@ -1049,77 +1115,149 @@ var a64ArrBits = [a64ArrCount]uint32{
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VADD": {0x0e208400, 0x7f, false, false},
"VSUB": {0x2e208400, 0x7f, false, false},
"VMUL": {0x0e209c00, 0x3f, false, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false, false},
"VORR": {0x0ea01c00, 0x03, false, false},
"VADDP": {0x0e20bc00, 0x7f, false, false},
"VZIP1": {0x0e003800, 0x7f, false, false},
"VZIP2": {0x0e007800, 0x7f, false, false},
"VCMEQ": {0x2e208c00, 0x7f, false, false},
"VCMGE": {0x0e203c00, 0x7f, false, false},
"VCMGT": {0x0e203400, 0x7f, false, false},
"VCMHI": {0x2e203400, 0x7f, false, false},
"VCMHS": {0x2e203c00, 0x7f, false, false},
// FP compares take S and D arrangements only (the toolchain rejects the
// byte and half forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, fpSimdArrs, false, true},
"VFCMGE": {0x2e20e400, fpSimdArrs, false, true},
"VFCMGT": {0x2ea0e400, fpSimdArrs, false, true},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFADD": {0x0e20d400, fpSimdArrs, false, true},
"VFSUB": {0x0ea0d400, fpSimdArrs, false, true},
"VFMUL": {0x2e20dc00, fpSimdArrs, false, true},
"VFDIV": {0x2e20fc00, fpSimdArrs, false, true},
"VFMAX": {0x0e20f400, fpSimdArrs, false, true},
"VFMIN": {0x0ea0f400, fpSimdArrs, false, true},
"VFMAXNM": {0x0e20c400, fpSimdArrs, false, true},
"VFMINNM": {0x0ea0c400, fpSimdArrs, false, true},
"VFMLA": {0x0e20cc00, fpSimdArrs, false, true},
"VFMLS": {0x0ea0cc00, fpSimdArrs, false, true},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
"VBIC": {0x0e601c00, 0x03, false, false}, // logical ops accept 8B and 16B only
"VBIF": {0x2ee01c00, 0x03, false, false},
"VBIT": {0x6ea01c00, 0x03, false, false},
"VBSL": {0x6e601c00, 0x03, false, false},
"VCMTST": {0x0e208c00, 0x7f, false, false},
"VFADDP": {0x2e20d400, fpSimdArrs, false, true},
"VFMAXP": {0x2e20f400, fpSimdArrs, false, true},
"VFMINP": {0x6ea0f400, fpSimdArrs, false, true},
"VFMAXNMP": {0x2e20c400, fpSimdArrs, false, true},
"VFMINNMP": {0x6ea0c400, fpSimdArrs, false, true},
"VMLA": {0x4ea09400, 0x3f, false, false}, // no 2D: integer multiply stops at 4S
"VMLS": {0x6ea09400, 0x3f, false, false},
"VORN": {0x4ee01c00, 0x03, false, false},
"VSHADD": {0x4ea00400, 0x7f, false, false},
"VSRHADD": {0x4ea01400, 0x7f, false, false},
"VUHADD": {0x6ea00400, 0x7f, false, false},
"VURHADD": {0x6ea01400, 0x7f, false, false},
"VSMAX": {0x4ea06400, 0x3f, false, false}, // no 2D: integer max stops at 4S
"VSMIN": {0x4ea06c00, 0x3f, false, false},
"VSMAXP": {0x4ea0a400, 0x3f, false, false},
"VSMINP": {0x4ea0ac00, 0x3f, false, false},
"VUMAX": {0x2e206400, 0x3f, false, false},
"VUMIN": {0x2e206c00, 0x3f, false, false},
"VUMAXP": {0x6ea0a400, 0x3f, false, false},
"VUMINP": {0x6ea0ac00, 0x3f, false, false},
"VSQADD": {0x4ea00c00, 0x7f, false, false},
"VUQADD": {0x6ea00c00, 0x7f, false, false},
"VSQSUB": {0x4ea02c00, 0x7f, false, false},
"VUQSUB": {0x6ea02c00, 0x7f, false, false},
"VSSHL": {0x0e204400, 0x7f, false, false},
"VUSHL": {0x2e204400, 0x7f, false, false},
"VUZP1": {0x0e001800, 0x7f, false, false},
"VUZP2": {0x4ec05800, 0x7f, false, false},
"VTRN1": {0x4ec02800, 0x7f, false, false},
"VTRN2": {0x4ec06800, 0x7f, false, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true, false}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false, false},
"VUQSHL": {0x2e204c00, 0x7f, false, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
// arrangement does not travel on every operand: the encoding's size and Q
// bits read off one designated operand and the element widths pair up across
// the operands.
type a64SimdNLForm uint8
const (
a64NLTwoNarrow a64SimdNLForm = iota // (Vn.wide, Vd.narrow): size/Q from Vd
a64NLTwoLong // (Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeLongMul // (Vm.narrow, Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeWide // (Vm.narrow, Vn.wide, Vd.wide): size/Q from Vn
a64NLThreeLongShift // ($sh, Vn.narrow, Vd.long): size/Q from Vn, immh = esize+sh
a64NLThreeNarrowShift // ($sh, Vn.wide, Vd.narrow): size/Q from Vd, immh = esize-sh
)
// a64SimdNLSpec is one narrow/long/wide instruction: the base word (U, opcode
// and fixed bits positioned) and the arrangement form. qonly marks the FCVT
// family, whose size field is fixed in the base and only the Q bit follows
// the driving arrangement.
type a64SimdNLSpec struct {
base uint32
form a64SimdNLForm
qonly bool
}
// a64SimdNLTable holds the families the arrangement-driven three-register and
// two-register encoders cannot express. The .2 spellings force the 128-bit
// side of the pair through their operand arrangements, so the base carries no
// arrangement bits of its own.
var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSHRN": {0x0f008400, a64NLThreeNarrowShift, false},
"VSHRN2": {0x0f008400, a64NLThreeNarrowShift, false},
"VSXTL": {0x0f00a400, a64NLTwoLong, false},
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e212800, a64NLTwoNarrow, false},
"VXTN2": {0x0e212800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e214800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e212800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e212800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e214800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e214800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e616800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e617800, a64NLTwoLong, true},
"VFCVTL2": {0x0e617800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
"VUSHLL2": {0x2f00a400, a64NLThreeLongShift, false},
"VUADDW": {0x2e201000, a64NLThreeWide, false},
"VUADDW2": {0x2e201000, a64NLThreeWide, false},
"VUMULL": {0x2e20c000, a64NLThreeLongMul, false},
"VUMULL2": {0x2e20c000, a64NLThreeLongMul, false},
"VSMULL": {0x0e20c000, a64NLThreeLongMul, false},
"VSMULL2": {0x0e20c000, a64NLThreeLongMul, false},
"VUMLAL": {0x2e208000, a64NLThreeLongMul, false},
"VUMLAL2": {0x2e208000, a64NLThreeLongMul, false},
"VSMLAL": {0x0e208000, a64NLThreeLongMul, false},
"VSMLAL2": {0x0e208000, a64NLThreeLongMul, false},
"VUMLSL": {0x2e20a000, a64NLThreeLongMul, false},
"VUMLSL2": {0x2e20a000, a64NLThreeLongMul, false},
"VSMLSL": {0x0e20a000, a64NLThreeLongMul, false},
"VSMLSL2": {0x0e20a000, a64NLThreeLongMul, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
@@ -1145,43 +1283,43 @@ var a64SimdVZero = map[string]uint32{
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
// the register pair spelling ORR Vd, Vn, Vm.
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false, false},
"VREV64": {0x0e200800, 0x3f, false, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
"VUADDLV": {0x2e303800, 0x3f, false, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
"VABS": {0x0e20b800, 0x7f, false, false},
"VNEG": {0x2e20b800, 0x7f, false, false},
"VCLS": {0x0e204800, 0x7f, false, false},
"VCLZ": {0x2e204800, 0x7f, false, false},
"VCNT": {0x0e205800, 0x7f, false, false},
"VNOT": {0x2e205800, 0x7f, false, false},
"VSQABS": {0x0e207800, 0x7f, false, false},
"VSQNEG": {0x2e207800, 0x7f, false, false},
"VRBIT": {0x2e605800, 0x03, false, false}, // 8B and 16B only
"VSCVTF": {0x4e21d800, fpSimdArrs, false, true},
"VUCVTF": {0x6e21d800, fpSimdArrs, false, true},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false, true},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false, true},
"VFABS": {0x0ea0f800, fpSimdArrs, false, true},
"VFNEG": {0x2ea0f800, fpSimdArrs, false, true},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false, true},
"VFRINTN": {0x0e218800, fpSimdArrs, false, true},
"VFRINTP": {0x0ea18800, fpSimdArrs, false, true},
"VFRINTM": {0x0e219800, fpSimdArrs, false, true},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false, true},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
"VADDV": {0x0e31b800, 0x3f, false, false},
"VSMAXV": {0x0e30a800, 0x3f, false, false},
"VSMINV": {0x0e31a800, 0x3f, false, false},
"VUMAXV": {0x2e30a800, 0x3f, false, false},
"VUMINV": {0x2e31a800, 0x3f, false, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
@@ -1243,6 +1381,16 @@ var a64PRFOps = map[string]int{
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64VLDNBase and a64VSTNBase hold the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// fixed words (indexed by register count 2..4): the opcode field at bits
// 15:12 carries the access kind.
var a64VLDNBase = [5]uint32{0, 0, 0x0c408000, 0x0c404000, 0x0c400000}
var a64VSTNBase = [5]uint32{0, 0, 0x0c008000, 0x0c004000, 0x0c000000}
// a64VLDNReplicate holds the VLD2R/VLD3R fixed words beside the existing
// VLD1R (0x0d40c000) and VLD4R (0x0d60e000) bases.
var a64VLDNReplicate = [5]uint32{0, 0x0d40c000, 0x0d60c000, 0x0d40e000, 0x0d60e000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
@@ -1366,29 +1514,46 @@ type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
scale int // access width in bytes; the unsigned offset divides by it
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
"MOVD": {3, 0, 1, 8}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1, 4}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2, 4}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1, 2}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2, 1}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1, 4}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1, 8}, // LDR D (64-bit FP)
"FMOVQ": {0, 1, 3, 16}, // LDR/STR Q (128-bit FP): opc=11 selects it
}
// a64StoreOpc returns the store opc for a given load type: integer and FP
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
// the store form clears; FP registers are selected by V, not opc).
// the store form clears; FP registers are selected by V, not opc). The one
// exception is the 128-bit Q width, whose store is the opc=10 spelling: with
// opc=00 the size field selects STR B instead (STR Qt is size=00, opc=10).
func a64StoreOpc(t a64LSType) int {
if t.size == 0 && t.V == 1 {
return 2
}
return 0
}
// a64LSScale returns the byte width an access's unsigned offset divides by.
// The Q width carries its size in opc (the size field stays 0), yet scales
// by 16 like any other 128-bit access, so the exponent alone does not answer.
func a64LSScale(t a64LSType) int64 {
if t.size == 0 && t.V == 1 {
return 16
}
return int64(1) << uint(t.size)
}
// arm64RegClass discriminates integer (R), floating-point (F) registers for
// the MOV pseudo-instruction.
type arm64RegClass int
+1076 -9
View File
File diff suppressed because it is too large. Load diff
+64
View File
@@ -0,0 +1,64 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// arm64AcceptedErrorShapes lists the toolchain's arm64error.s spellings gasm
// still accepts, each an acceptance superset with a known shape. The list
// only shrinks: every tightening of the encoder moves spellings out of it,
// and a spelling reappearing here means a regression. The catalogue is
// empty as of Go 1.27: every line the toolchain's corpus rejects, gasm
// rejects too.
var arm64AcceptedErrorShapes = []string{}
// TestArm64ToolchainErrorParity walks the toolchain's arm64error.s (Go
// 1.27, arm64) and requires gasm to reject every case the toolchain rejects.
// A live Go toolchain is needed for the source file; the test skips without
// one or in -short.
func TestArm64ToolchainErrorParity(t *testing.T) {
if testing.Short() {
t.Skip("live arm64error.s corpus: skipped in -short mode")
}
goroot := os.Getenv("GOROOT")
if goroot == "" {
t.Skip("no GOROOT")
}
path := filepath.Join(goroot, "src", "cmd", "asm", "internal", "asm", "testdata", "arm64error.s")
data, err := os.ReadFile(path)
if err != nil {
t.Skip(err)
}
allowed := map[string]bool{}
for _, s := range arm64AcceptedErrorShapes {
allowed[s] = true
}
for raw := range strings.SplitSeq(string(data), "\n") {
line := strings.TrimSpace(raw)
if line == "" || strings.HasPrefix(line, "//") || strings.HasPrefix(line, "TEXT") || !strings.Contains(line, "ERROR") {
continue
}
body := line
if i := strings.Index(body, "//"); i >= 0 {
body = strings.TrimSpace(body[:i])
}
body = strings.ReplaceAll(body, "\t", " ")
body = strings.Join(strings.Fields(body), " ")
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t" + body + "\n\tRET\n"
f, perr := parser.Parse("errorparity.s", src)
if len(perr) > 0 {
continue // the parser already rejects the spelling
}
if _, aerr := AssembleFileARM64(f); aerr == nil && !allowed[body] {
t.Errorf("gasm accepts what the toolchain rejects: %s", body)
}
}
}
+571
View File
@@ -0,0 +1,571 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The assembler's side of the extended-instruction layer: this file turns a
// parsed arm64 statement into the operand form arch.ExtInstr.Encode consumes
// and routes statements only the layer can encode through the registry. It
// sits beside the main arm64 encoders, never inside them: the generated
// tables and the scalar, NEON and FP paths are untouched, and a statement
// reaches this file only when the mnemonic is registered in the extension
// layer and at least one operand is a scalable vector or predicate register.
//
// The spellings are the layer's own Plan 9 forms, the ones its metadata
// documents: Zn, Zm, Zd for the unpredicated three-vector class, Zm, Pg/M,
// Zdn for the predicated class, imm{, LSL #8}, Zdn for the immediate
// classes, and for the predicate family Pm.B, Pn.B, Pg/Z (or Pg.Z), Pd.B
// for the logical operations, Pn.B, Pg.Z, Pd.B for the breaks, Pm.T, Pn.T,
// Pd.T for the permutations, Rm, Rn, Pd.T for the while compares, PN8-PN15
// for the counter destinations, and the bare SETFFR. Stage three adds the
// crypto family (Zn.T, Zd.T, Zd.T read-back and the in-place Zd.T, Zd.T),
// the predicate counters (Pn.T, Pg, Rd; Pn.T, ZR; Rd, Pn.T, Rd; ZR and R
// terminators) and the reductions (Zn.T, Pg, Vd over the SIMD register
// V0-V31, with ZR and RSP accepted where the classes take them).
package asm
import (
"fmt"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64ExtStatement converts one instruction's operands into the extended
// layer's operand form. pinned reports that the statement belongs to the
// layer: the mnemonic is registered in the registry and the operand list
// carries at least one scalable vector, predicate or predicate-as-counter
// register, or no operands at all (the zero-operand forms such as SETFFR,
// which no scalar path could mean instead). A pinned statement can only
// encode through the layer, so every operand is read here and its
// diagnostic replaces whatever the scalar paths would have said about
// operands they cannot read; err is non-nil for a pinned statement whose
// operands the layer refuses, and extops is complete only when err is nil.
// Unpinned means the statement is nobody's: the caller falls through to the
// ordinary arm64 encoders, which keep their exact behaviour for every
// scalar, NEON and FP operand list.
func arm64ExtStatement(mnem string, ops []*ast.Operand) (extops []arch.ExtOperand, pinned bool, err error) {
if _, ok := LookupExtension(arch.ARM64, mnem); !ok {
return nil, false, nil
}
if !arm64ExtPinned(mnem, ops) {
return nil, false, nil
}
ops = arm64ExtMergeLists(ops)
out := make([]arch.ExtOperand, 0, len(ops))
for i, op := range ops {
text := strings.Join(strings.Fields(op.Raw), "")
// The spelled shift of an immediate class: the shift is an attribute
// of the preceding immediate operand (imm{, LSL #8}, Zdn), never an
// operand of its own.
if amount, ok := strings.CutPrefix(text, "LSL#"); ok {
if len(out) == 0 || out[len(out)-1].Kind != arch.ExtImm || out[len(out)-1].HasShift {
return nil, true, fmt.Errorf("%s: operand %d (%s): LSL belongs straight after an immediate", mnem, i+1, op.Raw)
}
n, convErr := strconv.Atoi(amount)
if convErr != nil {
return nil, true, fmt.Errorf("%s: operand %d (%s): %q is not an LSL amount", mnem, i+1, op.Raw, amount)
}
out[len(out)-1].Shift, out[len(out)-1].HasShift = n, true
continue
}
if op.Kind == ast.OpImmediate {
ext, ok := arm64ExtImmediate(op)
if !ok {
return nil, true, fmt.Errorf("%s: operand %d (%s) is not an immediate the layer can read", mnem, i+1, op.Raw)
}
out = append(out, ext)
continue
}
// The load and store destination list, [Z13.B] or the multi-register
// [Z13.B, Z14.B, Z15.B] of the multiple-structure shapes, its
// arrangement part of the instruction's identity.
if strings.HasPrefix(text, "[") && strings.HasSuffix(text, "]") {
if ext, ok := arm64ExtVectorList(mnem, strings.Trim(text, "[]")); ok {
out = append(out, ext)
continue
}
}
// The gather/scatter memory operand: a parenthesised register pair,
// an immediate-offset base or a lone vector base.
if ext, ok := arm64ExtSveMem(text); ok {
out = append(out, ext)
continue
}
if ext, ok := arm64ExtVector(text); ok {
out = append(out, ext)
continue
}
if ext, ok := arm64ExtPredicate(text); ok {
out = append(out, ext)
continue
}
if ext, ok := arm64ExtCounter(text); ok {
out = append(out, ext)
continue
}
if text == "ZR" {
out = append(out, arch.ExtZeroRegister())
continue
}
if text == "RSP" {
out = append(out, arch.ExtStackPointer())
continue
}
if ext, ok := arm64ExtSIMD(text); ok {
out = append(out, ext)
continue
}
if ext, ok := arm64ExtGeneral(text); ok {
out = append(out, ext)
continue
}
return nil, true, fmt.Errorf("%s: operand %d (%s) is not an extended-layer operand: want a scalable vector, predicate, general or counter register, or an immediate", mnem, i+1, op.Raw)
}
return out, true, nil
}
// arm64ExtPinned reports whether the statement belongs to the layer. A
// mnemonic the extension layer registers on its own, one the generated
// arm64 table does not know, owns every one of its statements: no scalar
// path could mean it instead, and the layer's diagnostics replace the
// unsupported-instruction complaint. A mnemonic both tables carry (the
// SVE aliases of ADD, SUB and MUL) keeps the operand-shape test: any
// operand is a scalable vector, predicate or predicate-as-counter
// register, the shapes only the extension layer reads, or the statement
// carries no operands at all and the mnemonic's zero-operand forms claim
// it. The shape test is deliberately loose about the suffixes: P0/B is
// not a spelling the layer takes, but the P of it makes the statement the
// layer's, and the conversion then diagnoses the operand precisely
// instead of leaving it to a scalar path that would report an unrelated
// register error.
func arm64ExtPinned(mnem string, ops []*ast.Operand) bool {
if len(ops) == 0 {
return true
}
if _, shared := a64InstrTable[mnem]; !shared {
return true
}
for _, op := range ops {
if op.Kind == ast.OpImmediate {
continue
}
text := strings.Join(strings.Fields(op.Raw), "")
if _, ok := arm64ExtVector(text); ok {
return true
}
if arm64ExtPredicateShape(text) {
return true
}
}
return false
}
// arm64ExtPredicateShape reports whether text spells a predicate or
// predicate-as-counter register at all: PN or P, digits, an optional
// arrangement suffix and an optional qualifier after a slash, whatever the
// qualifier says. The strict parses in arm64ExtPredicate and
// arm64ExtCounter judge the suffix; this shape only decides who the operand
// belongs to.
func arm64ExtPredicateShape(text string) bool {
if text == "" || text[0] != 'P' {
return false
}
text = text[1:]
if rest, found := strings.CutPrefix(text, "N"); found {
text = rest
}
if i := strings.IndexByte(text, '/'); i >= 0 {
text = text[:i]
}
if i := strings.IndexByte(text, '.'); i >= 0 {
text = text[:i]
}
_, err := strconv.Atoi(text)
return err == nil && text != ""
}
// arm64ExtImmediate converts a $ immediate into the layer's form. The
// parser folds a parenthesised constant expression in full ($(255<<8)) and
// reads a bare literal greedily, dropping any trailing operator tokens:
// $255<<8 parses as 255 with the shift silently gone. Encoding that silent
// prefix would assemble what the text did not say, so an unparenthesised
// immediate is accepted only when its whole text reads back as one integer
// carrying the parser's value.
func arm64ExtImmediate(op *ast.Operand) (arch.ExtOperand, bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return arch.ExtOperand{}, false
}
text := strings.Join(strings.Fields(strings.TrimPrefix(op.Raw, "$")), "")
if !strings.HasPrefix(text, "(") {
if _, parseErr := strconv.ParseInt(text, 0, 64); parseErr != nil {
return arch.ExtOperand{}, false
}
}
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return arch.ExtOperand{Kind: arch.ExtImm, Imm: v}, true
}
// arm64ExtVector parses a scalable vector register operand: Z0..Z31 with an
// optional element-size suffix, Z0.S. The arrangement is carried as written
// and the encoding validates it against the form.
func arm64ExtVector(text string) (arch.ExtOperand, bool) {
reg, arr, ok := arm64ExtReg(text, 'Z')
if !ok {
return arch.ExtOperand{}, false
}
return arch.ExtOperand{Kind: arch.ExtZReg, Reg: reg, Arr: arr}, true
}
// arm64ExtMergeLists rejoins the bracketed vector lists the parser reads as
// separate operands: the comma inside [Z13.B, Z14.B, Z15.B] is an operand
// boundary to the parser, so the list arrives as two or more pieces and the
// multiple-structure loads and stores need it whole. Pieces from an opening
// bracket to the one carrying the closing bracket rejoin over their commas;
// everything else passes through untouched.
func arm64ExtMergeLists(ops []*ast.Operand) []*ast.Operand {
closed := func(op *ast.Operand) bool {
return strings.HasSuffix(strings.Join(strings.Fields(op.Raw), ""), "]")
}
merged := make([]*ast.Operand, 0, len(ops))
for i := 0; i < len(ops); i++ {
text := strings.Join(strings.Fields(ops[i].Raw), "")
if !strings.HasPrefix(text, "[") || closed(ops[i]) {
merged = append(merged, ops[i])
continue
}
parts := []string{ops[i].Raw}
kind := ops[i].Kind
for i+1 < len(ops) {
i++
parts = append(parts, ops[i].Raw)
if closed(ops[i]) {
break
}
}
merged = append(merged, &ast.Operand{Kind: kind, Raw: strings.Join(parts, ",")})
}
return merged
}
// arm64ExtVectorList parses the bracketed vector list of the loads and
// stores: a single register, [Z13.B], or the multi-register lists of the
// LD2-LD4 and ST2-ST4 multiple-structure shapes, [Z13.B, Z14.B, Z15.B],
// consecutive registers under one arrangement. The instruction's own digit
// names the list's length where it carries one, so a two-register list
// under ZLD3 fails here. The encoding carries the first register alone;
// the length rides the operand for the encode side.
func arm64ExtVectorList(mnem, body string) (arch.ExtOperand, bool) {
regs := strings.Split(body, ",")
first, ok := arm64ExtVector(regs[0])
if !ok {
return arch.ExtOperand{}, false
}
for i, reg := range regs[1:] {
op, ok := arm64ExtVector(reg)
if !ok || op.Arr != first.Arr || op.Reg != first.Reg+i+1 {
return arch.ExtOperand{}, false
}
}
if count := arm64ExtListCount(mnem); count != len(regs) {
return arch.ExtOperand{}, false
}
if len(regs) > 1 {
first.List = len(regs)
}
return first, true
}
// arm64ExtListCount reads the list length a load or store mnemonic names,
// the digit straight after its ZLD or ZST prefix; the loads and stores
// without one carry a single register.
func arm64ExtListCount(mnem string) int {
rest, ok := strings.CutPrefix(mnem, "ZLD")
if !ok {
rest, ok = strings.CutPrefix(mnem, "ZST")
}
if !ok || rest == "" {
return 1
}
if c := rest[0]; c >= '2' && c <= '4' {
return int(c - '0')
}
return 1
}
// arm64ExtPredicate parses a predicate register operand: P0..P15 with an
// optional element-size suffix (P0.B) and an optional qualifier in either
// spelling the corpus and the wired forms use, P0/M and P0.Z.
func arm64ExtPredicate(text string) (arch.ExtOperand, bool) {
qual := arch.ExtQualNone
if base, suffix, found := strings.Cut(text, "/"); found {
switch suffix {
case "M":
qual = arch.ExtQualMerging
case "Z":
qual = arch.ExtQualZeroing
default:
return arch.ExtOperand{}, false
}
text = base
} else if base, suffix, found := strings.Cut(text, "."); found &&
(suffix == "Z" || suffix == "M") {
// The dot qualifier stands in place of an arrangement, the spelling
// the toolchain's corpus writes (P1.Z, P14.M).
qual = arch.ExtQualMerging
if suffix == "Z" {
qual = arch.ExtQualZeroing
}
text = base
}
reg, arr, ok := arm64ExtReg(text, 'P')
if !ok {
return arch.ExtOperand{}, false
}
return arch.ExtOperand{Kind: arch.ExtPReg, Reg: reg, Arr: arr, Qual: qual}, true
}
// arm64ExtCounter parses a predicate-as-counter register operand: PN8..PN15
// with an optional element-size suffix, PN14.S. The register range is the
// counter range the layer's convention carries; the encoding validates it.
func arm64ExtCounter(text string) (arch.ExtOperand, bool) {
rest, ok := strings.CutPrefix(text, "PN")
if !ok {
return arch.ExtOperand{}, false
}
reg, arr, ok := arm64ExtRegDigits(rest)
if !ok {
return arch.ExtOperand{}, false
}
return arch.ExtOperand{Kind: arch.ExtPNReg, Reg: reg, Arr: arr}, true
}
// arm64ExtGeneral parses a general register operand: R0..R30, the plain
// spelling the while-compare forms take, beside the ZR and RSP spellings of
// the thirty-first slot the conversion above reads. The register range is
// left to the encoding, whose diagnostics name it.
func arm64ExtGeneral(text string) (arch.ExtOperand, bool) {
rest, ok := strings.CutPrefix(text, "R")
if !ok {
return arch.ExtOperand{}, false
}
reg, arr, ok := arm64ExtRegDigits(rest)
if !ok || arr != arch.ExtArrNone {
return arch.ExtOperand{}, false
}
return arch.ExtOperand{Kind: arch.ExtGReg, Reg: reg}, true
}
// arm64ExtSIMD parses a 128-bit SIMD register operand: V0..V31, written
// bare, the scalar destination the reductions and the crypto read-back
// forms take, or with the counted quadword suffix of the SVE2.1 QV class,
// V5.S4 reading four 32-bit lanes (the spellings .B16, .H8, .S4 and .D2;
// no other suffix parses). The register range is left to the encoding.
func arm64ExtSIMD(text string) (arch.ExtOperand, bool) {
rest, ok := strings.CutPrefix(text, "V")
if !ok {
return arch.ExtOperand{}, false
}
arr := arch.ExtArrNone
for _, q := range []struct {
suffix string
arr arch.ExtArrangement
}{
{"B16", arch.ExtArrB},
{"H8", arch.ExtArrH},
{"S4", arch.ExtArrS},
{"D2", arch.ExtArrD},
} {
if s := "." + q.suffix; strings.HasSuffix(rest, s) {
arr = q.arr
rest = rest[:len(rest)-len(s)]
break
}
}
reg, bare, ok := arm64ExtRegDigits(rest)
if !ok || bare != arch.ExtArrNone {
return arch.ExtOperand{}, false
}
return arch.ExtOperand{Kind: arch.ExtVReg, Reg: reg, Arr: arr}, true
}
// arm64ExtSveMem parses the gather/scatter memory operand off a normalised
// operand text: the parenthesised pair (R6)(R14), (Z23.D<<1)(R24) and
// (Z4.S.UXTW)(R3), the immediate-offset base 6(Z7.S), and the lone vector
// base (Z5.D) of the stores. The second parenthesis accepts the
// stack-pointer spelling RSP; the ranges and the mode's own rules are left
// to the encoding, whose diagnostics name them.
func arm64ExtSveMem(text string) (arch.ExtOperand, bool) {
// The immediate-offset spelling: digits straight before the parenthesis.
if i := strings.IndexByte(text, '('); i > 0 && i == strings.LastIndexByte(text, '(') {
disp, err := strconv.ParseUint(text[:i], 10, 32)
if err == nil && strings.HasSuffix(text, ")") {
op, ok := arm64ExtSveMemGroup(text[i+1 : len(text)-1])
if !ok {
return arch.ExtOperand{}, false
}
if !op.BaseVec || op.Extend != 0 || op.Shift != 0 {
return arch.ExtOperand{}, false
}
op.Imm = int64(disp)
return op, true
}
}
// The parenthesised forms: one group or two.
rest, ok := strings.CutPrefix(text, "(")
if !ok || !strings.HasSuffix(text, ")") {
return arch.ExtOperand{}, false
}
rest = rest[:len(rest)-1]
first := rest
op := arch.ExtOperand{Off: -1}
if base, second, found := strings.Cut(rest, ")("); found {
first = base
off, ok := arm64ExtSveMemOffset(second)
if !ok {
return arch.ExtOperand{}, false
}
op = off
}
group, ok := arm64ExtSveMemGroup(first)
if !ok {
return arch.ExtOperand{}, false
}
group.Off = op.Off
group.Reg31 = op.Reg31
return group, true
}
// arm64ExtSveMemGroup parses one parenthesised memory register: R6, R6<<3,
// Z23.D, Z23.D<<1, Z4.S.UXTW or Z7.D.SXTW. The general registers run
// R0-R30 and the scalable vectors Z0-Z31 with an .S or .D element size and
// an optional UXTW or SXTW extension; the ranges are left to the encoding.
func arm64ExtSveMemGroup(text string) (arch.ExtOperand, bool) {
op := arch.ExtOperand{Kind: arch.ExtSveMem, Off: -1}
if base, shift, found := strings.Cut(text, "<<"); found {
n, err := strconv.Atoi(shift)
if err != nil || n < 0 {
return arch.ExtOperand{}, false
}
op.Shift = n
text = base
}
parts := strings.Split(text, ".")
switch parts[0][0] {
case 'R':
reg, err := strconv.Atoi(parts[0][1:])
if err != nil || len(parts) != 1 {
return arch.ExtOperand{}, false
}
op.Reg = reg
case 'Z':
reg, err := strconv.Atoi(parts[0][1:])
if err != nil || len(parts) < 2 || len(parts) > 3 {
return arch.ExtOperand{}, false
}
switch parts[1] {
case "S":
op.Arr = arch.ExtArrS
case "D":
op.Arr = arch.ExtArrD
default:
return arch.ExtOperand{}, false
}
op.Reg = reg
op.BaseVec = true
if len(parts) == 3 {
switch parts[2] {
case "UXTW":
op.Extend = 1
case "SXTW":
op.Extend = 2
default:
return arch.ExtOperand{}, false
}
}
default:
return arch.ExtOperand{}, false
}
return op, true
}
// arm64ExtSveMemOffset parses the second parenthesis of a gather/scatter
// memory operand: a plain R0-R30 or the stack-pointer spelling RSP.
func arm64ExtSveMemOffset(text string) (arch.ExtOperand, bool) {
if text == "RSP" {
return arch.ExtOperand{Off: 31, Reg31: 2}, true
}
if len(text) < 2 || text[0] != 'R' {
return arch.ExtOperand{}, false
}
reg, err := strconv.Atoi(text[1:])
if err != nil {
return arch.ExtOperand{}, false
}
return arch.ExtOperand{Off: reg}, true
}
// arm64ExtRegDigits parses the digits and optional arrangement suffix of a
// register spelling once the letter prefix is gone.
func arm64ExtRegDigits(text string) (reg int, arr arch.ExtArrangement, ok bool) {
if base, suffix, found := strings.Cut(text, "."); found {
switch suffix {
case "B":
arr = arch.ExtArrB
case "H":
arr = arch.ExtArrH
case "S":
arr = arch.ExtArrS
case "D":
arr = arch.ExtArrD
case "Q":
arr = arch.ExtArrQ
default:
return 0, 0, false
}
text = base
}
n, err := strconv.Atoi(text)
if err != nil || n < 0 {
return 0, 0, false
}
return n, arr, true
}
// arm64ExtReg parses Pn or Zn with an optional arrangement suffix off a
// normalised operand text. The register range is left to the encoding: the
// layer's own diagnostics name the range a form carries.
func arm64ExtReg(text string, letter byte) (reg int, arr arch.ExtArrangement, ok bool) {
if len(text) < 2 || text[0] != letter {
return 0, 0, false
}
digits := text[1:]
if base, suffix, found := strings.Cut(digits, "."); found {
switch suffix {
case "B":
arr = arch.ExtArrB
case "H":
arr = arch.ExtArrH
case "S":
arr = arch.ExtArrS
case "D":
arr = arch.ExtArrD
case "Q":
arr = arch.ExtArrQ
default:
return 0, 0, false
}
digits = base
}
n, err := strconv.Atoi(digits)
if err != nil || n < 0 {
return 0, 0, false
}
return n, arr, true
}
File diff suppressed because it is too large. Load diff
+106 -2
View File
@@ -53,7 +53,7 @@ package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// arm64FrameInfo holds the frame layout derived from a TEXT directive.
@@ -69,6 +69,12 @@ type arm64FrameInfo struct {
// autosize below StackSmall as NOSPLIT.
needSplit bool
splitClass int // 0: <=StackSmall, 1: <=StackBig, 2: >StackBig
// tls holds the names the file's own GLOBL declarations mark TLSBSS:
// the toolchain's aclass keys the TLS-LE load off the symbol's type
// (objabi.STLSBSS), which only the file's declarations reveal. A nil
// map answers no, the plain static-symbol load.
tls map[string]bool
}
// arm64ComputeFrame derives the frame layout for a TEXT function.
@@ -87,7 +93,10 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
if fi.frame != 0 || !fi.leaf {
fi.autosize = fi.frame + 8 // space for the saved LR
// The toolchain always adds an extrasize: 8 when the total leaves a
// 16-byte alignment gap, another 16 when already aligned.
// 16-byte alignment gap, another 16 when already aligned. An
// autosize of zero (the $-8 convention included) is frameless and
// takes neither.
if fi.autosize != 0 {
switch fi.autosize % 16 {
case 8:
fi.autosize += 8
@@ -99,6 +108,13 @@ func arm64ComputeFrame(t *ast.Text) arm64FrameInfo {
fi.autosize += 16 - (fi.autosize % 16)
}
}
}
if fi.autosize == 0 {
// The NOFRAME shape: the toolchain forces the leaf mark on any
// autosize-zero function (calls included), so nothing saves LR and
// no stack-split guard runs.
fi.leaf = true
}
switch {
case fi.noSplit:
case fi.autosize < stackSmall && fi.leaf:
@@ -301,6 +317,37 @@ func arm64Return(fi arm64FrameInfo) []byte {
return a64WordsLE(ws...)
}
// arm64RetInstr encodes a RET. The plain form runs the frame epilogue and
// branches to LR; RET Rn runs the epilogue and branches to the register
// (asm7.go case 78); RET sym(SB) runs the epilogue and branches to the
// symbol with the call relocation, the toolchain's retJMP tail call.
func arm64RetInstr(fi arm64FrameInfo, ops []*ast.Operand, relocs *[]Reloc) []byte {
out := arm64Return(fi)
if len(ops) != 1 {
return out
}
op := ops[0]
switch {
case op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Index == "":
if rn := arm64RegNum(op.Addr.Sym.Name); rn >= 0 {
// The operand form replaces the default BR LR word.
return append(out[:len(out)-4], a64wordLE(0xd65f0000|uint32(rn)<<5)...)
}
case op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" && op.Addr.Base == "" && op.Addr.Index == "":
if relocs != nil {
*relocs = append(*relocs, Reloc{
Off: 0,
After: 4,
Name: op.Addr.Sym.Name,
Addend: op.Addr.Sym.Offset,
Kind: RelArm64Branch,
})
}
return append(out[:len(out)-4], a64wordLE(0x14000000)...) // B sym(SB)
}
return out
}
// arm64PrologueSpadjPC returns the function-relative byte offset where the
// prologue has finished decrementing SP (the delta becomes autosize).
func arm64PrologueSpadjPC(fi arm64FrameInfo) int {
@@ -362,6 +409,59 @@ func arm64ResolvePseudo(sym *ast.Symbol, fi arm64FrameInfo) (base int, off int32
return -1, 0
}
// arm64FrameAddrValue returns the SP-relative displacement a $sym+off(FP)
// or $sym+off(SP) immediate-address operand stands for, the toolchain's
// aclass arithmetic (asm7.go): a parameter reference sits autosize+8 above
// the hardware SP, and a pseudo-SP reference sits frame+8 above it, the
// alignment padding cancelling out of the autosize.
func arm64FrameAddrValue(sym *ast.Symbol, fi arm64FrameInfo) int64 {
switch sym.Pseudo {
case "FP":
return sym.Offset + int64(fi.autosize) + 8
default: // SP
return sym.Offset + int64(fi.frame) + 8
}
}
// arm64IsAddcon reports whether v is an addcon value (asm7.go isaddcon): an
// unsigned imm12, or a multiple of 4096 whose shifted form fits imm12.
func arm64IsAddcon(v int64) bool {
if v < 0 {
return false
}
if v&0xFFF == 0 {
v >>= 12
}
return v <= 0xFFF
}
// arm64FrameAddrWords returns the word sequence of the toolchain's optab
// case 4 for a frame-relative address (the C_AACON and C_AACON2 rows): one
// ADD/SUB imm12 word inside the addcon band, SUB carrying a negative
// displacement, and otherwise the hi<<12 word from SP followed by the low
// word added in place. The 24-bit band never reaches the REGTMP
// materialisation the ADD/SUB immediate ladder uses: aclass classifies the
// address straight into C_AACON2.
func arm64FrameAddrWords(v int64, rd int) []uint32 {
word := func(v int64, rn int) uint32 {
op := uint32(0) // ADD
if v < 0 {
op = 1 // SUB
v = -v
}
sh := uint32(0)
if v&0xFFF000 != 0 { // asm7.go oaddi: the shift form when low 12 bits are clear
sh = 1
v >>= 12
}
return a64AddSub(1, op, 0, sh, uint32(v), uint32(rn), uint32(rd))
}
if arm64IsAddcon(v) || arm64IsAddcon(-v) {
return []uint32{word(v, 31)}
}
return []uint32{word(v&^int64(0xFFF), 31), word(v&0xFFF, rd)}
}
// arm64PreStoreImm encodes a pre-index store (STR with writeback):
// size<<30 | 7<<27 | V<<26 | opc<<22 | 1<<11 | 1<<10 | imm9<<12 | Rn<<5 | Rt.
func arm64PreStoreImm(size, V int, imm9 int32, rn, rt int) uint32 {
@@ -454,6 +554,10 @@ func arm64GuardBytes(fi arm64FrameInfo, blockStart int) []byte {
return a64WordsLE(ws...)
}
// arm64MoreStackBlockLen is the byte length of arm64MoreStackBlock: the
// saved LR, the BL and the branch back, three words whatever the target.
const arm64MoreStackBlockLen = 12
// arm64MoreStackBlock emits the trailing block: MOVD R30, R3 (save LR),
// BL runtime.morestack_noctxt, B back to the function start. The BL carries
// the R_CALLARM64 relocation.
+236
View File
@@ -0,0 +1,236 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// assembleArm64Source parses src and assembles it for arm64, returning the
// first function's words little-endian.
func assembleArm64Source(t *testing.T, src string) []uint32 {
t.Helper()
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("got %d funcs, want 1", len(img.Funcs))
}
return wordsOf(img.Code[img.Funcs[0].Offset : img.Funcs[0].Offset+img.Funcs[0].Size])
}
func TestArm64FrameAddrEncoding(t *testing.T) {
src := `#include "textflag.h"
TEXT ·fr(SB), NOSPLIT, $432-24
MOVD $argframe+0(FP), R3
MOVD $big+4096(FP), R4
MOVD $ret-8(FP), R2
MOVD $x-64(FP), R5
MOVD RSP, R19
MOVD R20, RSP
RET
`
ws := assembleArm64Source(t, src)
// Prologue (4: large frame) then the body at words 4..9, the toolchain's
// own encodings for the same statements:
// ADD $456, RSP, R3 (456 = 448 + 8 + 0)
// ADD $(1<<12), RSP, R4 (4552, the hi<<12 half)
// ADD $456, R4, R4 (then the lo half)
// ADD $448, RSP, R2 (448 - 8 + 8)
// ADD $392, RSP, R5 (448 - 64 + 8)
// ADD $0, RSP, R19 (the SP register move)
// ADD $0, R20, RSP
want := []uint32{
0xd10703f4, 0xa93ffa9d, 0x9100029f, 0xd10023fd,
0x910723e3, 0x914007e4, 0x91072084, 0x910703e2,
0x910623e5, 0x910003f3, 0x9100029f,
}
if len(ws) < len(want) {
t.Fatalf("got %d words, want at least %d", len(ws), len(want))
}
for i, w := range want {
if ws[i] != w {
t.Errorf("word %d: got %08x, want %08x", i, ws[i], w)
}
}
}
func TestArm64FrameAddrSizes(t *testing.T) {
// One imm12 word inside the addcon band, two inside the 24-bit band.
tests := []struct {
v int64
verb int
}{
{456, 1},
{0xFFF, 1},
{0x1000, 1}, // the shifted imm12 form
{0x1005, 2}, // the hi<<12 plus lo pair
{0xFFFFFF, 2},
}
for _, tt := range tests {
got := len(arm64FrameAddrWords(tt.v, 3))
if got != tt.verb {
t.Errorf("arm64FrameAddrWords(%d) took %d words, want %d", tt.v, got, tt.verb)
}
}
if ws := arm64FrameAddrWords(0x1000, 3); len(ws) != 1 || ws[0] != 0x914007e3 {
t.Errorf("arm64FrameAddrWords(0x1000) = %08x, want the shifted ADD 914007e3", ws[0])
}
if !arm64IsAddcon(0xFFF) || arm64IsAddcon(0x1001) || arm64IsAddcon(-1) {
t.Error("arm64IsAddcon misclassifies the band edges")
}
fp := &ast.Symbol{Name: "x", Pseudo: "FP", Offset: 8}
if v := arm64FrameAddrValue(fp, arm64FrameInfo{autosize: 448}); v != 464 {
t.Errorf("FP address: got %d, want 464", v)
}
sp := &ast.Symbol{Name: "x", Pseudo: "SP", Offset: 8}
if v := arm64FrameAddrValue(sp, arm64FrameInfo{frame: 432}); v != 448 {
t.Errorf("SP address: got %d, want 448", v)
}
}
func TestArm64FrameAddrRejects(t *testing.T) {
tests := []struct {
name string
src string
want string
}{
{
"width", `TEXT ·f(SB), NOSPLIT, $16-8
MOVW $x+0(FP), R7
RET
`, "illegal combination",
},
{
"pool band", `TEXT ·f(SB), NOSPLIT, $20000000-8
MOVD $x+20000000(FP), R7
RET
`, "literal pool",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
f, errs := parser.Parse("test_arm64.s", tt.src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
_, err := AssembleFileARM64(f)
if err == nil {
t.Fatalf("%s: no error, want one naming %q", tt.name, tt.want)
}
if !strings.Contains(err.Error(), tt.want) {
t.Errorf("%s: error %q, want it to name %q", tt.name, err, tt.want)
}
})
}
}
func TestArm64SPMoveEncoding(t *testing.T) {
src := `TEXT ·f(SB), NOSPLIT, $0-8
MOVD RSP, R19
MOVD R20, RSP
MOVD ZR, R4
MOVD RSP, RSP
RET
`
ws := assembleArm64Source(t, src)
// The SP register moves ride ADD $0; the zero move stays ORR.
want := []uint32{
0x910003f3, // ADD $0, RSP, R19
0x9100029f, // ADD $0, R20, RSP
0xaa1f03e4, // ORR R4, ZR, ZR
0x910003ff, // ADD $0, RSP, RSP
0xd65f03c0, // RET
}
if len(ws) != len(want) {
t.Fatalf("got %d words, want %d", len(ws), len(want))
}
for i, w := range want {
if ws[i] != w {
t.Errorf("word %d: got %08x, want %08x", i, ws[i], w)
}
}
rej := `TEXT ·f(SB), NOSPLIT, $0-8
MOVW RSP, R7
RET
`
f, errs := parser.Parse("test_arm64.s", rej)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil || !strings.Contains(err.Error(), "illegal combination") {
t.Errorf("MOVW RSP: error %v, want an illegal-combination refusal", err)
}
}
func TestArm64AliasLiveness(t *testing.T) {
// A define inside a dead conditional branch must not become an alias:
// go_tls.h's `#ifdef GOARCH_arm` block defines LR as R14, which would
// silently renumber the link register on arm64, where LR is R30.
src := `#define RARG R5
#ifdef GOARCH_arm
#define LR R14
#endif
TEXT ·f(SB), NOSPLIT, $0-8
MOVD LR, R0
MOVD RARG, R1
MOVD R14, R2
RET
`
ws := assembleArm64Source(t, src)
want := []uint32{
0xaa1e03e0, // ORR R0, ZR, R30: LR stayed the link register
0xaa0503e1, // ORR R1, ZR, R5: the live alias applied
0xaa0e03e2, // ORR R2, ZR, R14: R14 is R14
0xd65f03c0,
}
if len(ws) != len(want) {
t.Fatalf("got %d words, want %d", len(ws), len(want))
}
for i, w := range want {
if ws[i] != w {
t.Errorf("word %d: got %08x, want %08x", i, ws[i], w)
}
}
}
func TestArm64ADRNoChainChase(t *testing.T) {
// The toolchain's jump-to-jump collapse rewrites branch targets only:
// a B to a chain-leading label is redirected, an ADR to the same label
// resolves to the label itself.
src := `TEXT ·adrchain(SB), NOSPLIT, $0-0
B a
a:
B b
b:
ADR a, R0
RET
`
ws := assembleArm64Source(t, src)
want := []uint32{
0x14000002, // B +2: chased through a to b
0x14000001, // B +1: a's own jump to b
0x10ffffe0, // ADR a, R0: -4, the label itself, unchased
0xd65f03c0,
}
if len(ws) != len(want) {
t.Fatalf("got %d words, want %d", len(ws), len(want))
}
for i, w := range want {
if ws[i] != w {
t.Errorf("word %d: got %08x, want %08x", i, ws[i], w)
}
}
}
+269
View File
@@ -0,0 +1,269 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"os/exec"
"path/filepath"
"regexp"
"strconv"
"strings"
"testing"
"time"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
var le = binary.LittleEndian
// kindSTEXTFIPS is objabi's STEXTFIPS: the fips140 packages' text kind.
const kindSTEXTFIPS = 2
// gorootARM64Packages names the GOROOT packages whose arm64 assembly the
// parity harness pins: every *_arm64.s of each package, as the real build
// assembles it, the package's generated go_asm.h included. Together they
// carry the heavy real-world shapes: the runtime's TLS and stack plumbing,
// the cryptographic kernels, big-number arithmetic and the bytealg search
// loops.
var gorootARM64Packages = []string{
"runtime",
"internal/bytealg",
"internal/cpu",
"internal/chacha8rand",
"internal/runtime/maps",
"reflect",
"math/big",
"hash/crc32",
"crypto/md5",
"crypto/sha1",
"crypto/internal/fips140/aes",
"crypto/internal/fips140/aes/gcm",
"crypto/internal/fips140/bigmod",
"crypto/internal/fips140/nistec",
"crypto/internal/fips140/sha256",
"crypto/internal/fips140/sha512",
"crypto/internal/fips140/sha3",
"crypto/internal/fips140/subtle",
}
// gorootARM64GapFiles names the files kept out of the byte parity set by
// known, pre-existing gaps, each with the reason. A file here is skipped,
// not silently dropped: the gaps are findings, and closing one is a matter
// of removing its entry and watching the file pin itself.
var gorootARM64GapFiles = map[string]string{
"runtime/asm_arm64.s": "the unparenthesised NOSPLIT|NOFRAME flag list is not recognised, so the prologue and guard shapes diverge",
"runtime/sys_linux_arm64.s": "the unparenthesised NOSPLIT|NOFRAME flag list is not recognised, so the prologue and guard shapes diverge (cgoSigtramp, clone)",
"runtime/preempt_arm64.s": "the unparenthesised NOSPLIT|NOFRAME flag list is not recognised, so the prologue and guard shapes diverge (asyncPreempt)",
"runtime/race_arm64.s": "the unparenthesised NOSPLIT|NOFRAME flag list is not recognised, so the prologue shape diverges (racecallbackthunk)",
"runtime/rt0_linux_arm64.s": "the #ifdef GOOS selection diverges: gasm keeps a word the toolchain drops",
}
// gorootOtherGOOS matches the file names of the ports the linux build never
// assembles: the harness pins the linux arm64 set.
var gorootOtherGOOS = regexp.MustCompile(`_(darwin|ios|freebsd|netbsd|openbsd|windows|android|plan9|aix|js|wasip1)_`)
// TestGOROOTARM64Parity assembles each package's arm64 files with gasm and
// with the installed toolchain and holds the functions' bytes equal,
// relocation sites masked. The toolchain side needs the go_asm.h the build
// generates for the package, so the harness rebuilds it once with -work and
// harvests the header the compiler wrote; the -gcflags flag exists only to
// make that rebuild happen, the header's constants do not depend on it.
func TestGOROOTARM64Parity(t *testing.T) {
if testing.Short() {
t.Skip("live go tool asm oracle and per-package rebuild: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
out, err := exec.Command(goBin, "env", "GOROOT").Output()
if err != nil {
t.Fatalf("go env GOROOT: %v", err)
}
goroot := strings.TrimSpace(string(out))
include := filepath.Join(goroot, "pkg", "include")
totalFns, totalBytes := 0, 0
for _, pkg := range gorootARM64Packages {
t.Run(pkg, func(t *testing.T) {
dir := t.TempDir()
work := harvestGoAsm(t, goBin, pkg)
defer os.RemoveAll(work)
headers, err := filepath.Glob(filepath.Join(work, "b*", "go_asm.h"))
if err != nil || len(headers) == 0 {
t.Fatalf("no generated go_asm.h under %s", work)
}
if err := os.WriteFile(filepath.Join(dir, "go_asm.h"), mustRead(t, headers[0]), 0o644); err != nil {
t.Fatal(err)
}
files, err := filepath.Glob(filepath.Join(goroot, "src", pkg, "*_arm64.s"))
if err != nil || len(files) == 0 {
t.Fatalf("no arm64 assembly found for %s", pkg)
}
for _, f := range files {
base := filepath.Base(f)
if gorootOtherGOOS.MatchString(base) {
continue // another port's file: the linux build never assembles it
}
t.Run(base, func(t *testing.T) {
if reason, gap := gorootARM64GapFiles[pkg+"/"+base]; gap {
t.Skip(reason)
}
// The parser side: the file's own directory resolves the
// package's headers, the harvested directory carries
// go_asm.h, and the platform conditionals read the same
// predefines the go command drives go tool asm with.
src := string(mustRead(t, f))
af, errs := parser.ParseWithOptions(f, src, parser.Options{
Expand: true,
IncludeDirs: []string{dir, filepath.Join(goroot, "src", pkg), include},
Predefines: map[string]string{
"GOARCH_arm64": "1",
"GOOS_linux": "1",
},
})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs[0])
}
img, err := AssembleFileARM64(af)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
// The oracle side: the build's own invocation, the
// generated header directory first.
objPath := filepath.Join(t.TempDir(), "oracle.o")
cmd := exec.Command(goBin, "tool", "asm",
"-I", dir, "-I", filepath.Join(goroot, "src", pkg), "-I", include,
"-D", "GOOS_linux", "-D", "GOARCH_arm64", "-std",
"-p", pkg, "-o", objPath, f)
cmd.Env = append(os.Environ(), "GOOS=linux", "GOARCH=arm64")
if oout, err := cmd.CombinedOutput(); err != nil {
t.Fatalf("go tool asm %s: %v\n%s", base, err, oout)
}
byLocal := oracleFuncText(t, mustRead(t, objPath))
// The object's symdef order is the source order, gasm's
// function list too, so same-named functions pair up in
// definition order: a file-local kernel beside its
// package-level twin (runtime·racefuncenter and
// racefuncenter<>) carries the plain name twice in the
// object and the reader cannot see the locality.
seen := map[string]int{}
for _, fn := range img.Funcs {
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
bodies := byLocal[fn.Name]
idx := seen[fn.Name]
seen[fn.Name] = idx + 1
if idx >= len(bodies) {
keys := make([]string, 0, len(byLocal))
for name := range byLocal {
keys = append(keys, name)
}
t.Errorf("%s: not in the oracle output (%d functions: %s)",
fn.Name, len(byLocal), strings.Join(keys, ", "))
continue
}
goCode := maskCode(append([]byte(nil), bodies[idx]...), fn.Relocs)
cmpLen := min(len(goCode), len(gasmCode))
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
for w := 0; w < cmpLen/4; w++ {
g := le.Uint32(gasmCode[w*4:])
o := le.Uint32(goCode[w*4:])
if g != o {
t.Errorf("%s: word %d (offset %d) differs: gasm %08x go %08x", fn.Name, w, w*4, g, o)
break
}
}
continue
}
if len(goCode) > len(gasmCode) {
for _, b := range goCode[len(gasmCode):] {
if b != 0 {
t.Errorf("%s: non-zero trailing bytes in the oracle output", fn.Name)
break
}
}
}
totalFns++
totalBytes += len(gasmCode)
}
})
}
})
}
t.Logf("GOROOT arm64 parity: %d functions, %d bytes identical", totalFns, totalBytes)
}
// oracleFuncText extracts every TEXT function of a toolchain object, the
// non-package and the hashed (file-local) definitions both, keyed by the
// local name: GOROOT keeps several kernels file-local (cmpbody<>,
// encryptBlockAsm<>), and those ride the hashed definition blocks the
// non-package reader never sees. The value is the functions' bodies in
// symdef order: a file-local kernel beside its package-level twin carries
// the same plain name twice (the object reader cannot see the locality),
// and the encoder pairs them up in definition order.
func oracleFuncText(t *testing.T, obj []byte) map[string][][]byte {
t.Helper()
v := openGoobj(t, obj)
data := v.blk(blkData)
didx := v.blk(blkDataIdx)
out := make(map[string][][]byte)
di := 0
for _, bi := range []int{blkSymdef, blkHashed64def, blkHasheddef, blkNonpkgdef} {
for _, s := range v.syms(bi) {
// STEXT and STEXTFIPS both: the fips140 packages' text carries
// the FIPS kind in Go 1.27 and up.
if (s.typ == kindSTEXT || s.typ == kindSTEXTFIPS) && s.size > 0 && 4*di+8 <= len(didx) {
off := le.Uint32(didx[4*di:])
if int(off)+int(s.size) <= len(data) {
name := s.name
if _, after, ok := strings.Cut(name, "."); ok {
name = after
}
out[name] = append(out[name], data[off:int(off)+int(s.size)])
}
}
di++
}
}
return out
}
// harvestGoAsm rebuilds pkg once with -work and returns the work directory
// holding the compiler's generated go_asm.h. A fully cached build leaves
// the work directory empty, so the compile action is given a unique, inert
// flag value each run (the inlining level never touches the header's
// constants) and re-runs for the target package alone.
func harvestGoAsm(t *testing.T, goBin, pkg string) string {
t.Helper()
nonce := time.Now().UnixNano() % 1000000
cmd := exec.Command(goBin, "build", "-x", "-work",
"-gcflags", pkg+"=-N", "-gcflags", pkg+"=-l=7"+strconv.FormatInt(nonce, 10),
"-o", "/dev/null", pkg)
cmd.Env = append(os.Environ(), "GOOS=linux", "GOARCH=arm64")
out, err := cmd.CombinedOutput()
if err != nil {
t.Fatalf("rebuild %s: %v\n%s", pkg, err, out)
}
m := regexp.MustCompile(`WORK=(\S+)`).FindSubmatch(out)
if m == nil {
t.Fatalf("rebuild %s: no WORK directory in the build log", pkg)
}
return string(m[1])
}
// mustRead reads path, failing the test when it cannot.
func mustRead(t *testing.T, path string) []byte {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
return data
}
+279
View File
@@ -0,0 +1,279 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The mid-function literal pool flush, pinned against the toolchain. A
// kernel of distinct pooled stores walks the conservative displacement bound
// (asm7.go's maxPCDisp) inside one body, so the pool must drain exactly
// where cmd/internal/obj/arm64's checkpool drains it: the bytes, the flush
// points and the reach of every load literal are the toolchain's own.
// pooledKernel renders a kernel of n distinct pooled stores: each offset
// sits a step of 8 past the aligned split band (every fourth one aligned,
// but 0x2000000+8i+4 never is), so every statement pools its own four-byte
// word and expands to eight instruction bytes, the densest walk towards the
// distance bound an arm64 body can make. head carries the TEXT line for the
// splitting variant, tail the closing statements.
func pooledKernel(head string, n int, tail string) string {
var b strings.Builder
b.WriteString("#include \"textflag.h\"\n")
if head != "" {
b.WriteString(head)
b.WriteString("\n")
}
b.WriteString("TEXT \u00b7poolmid(SB), NOSPLIT, $0-0\n")
for i := range n {
b.WriteString("\tMOVD\tR1, ")
b.WriteString(decimal(0x2000000 + 8*i + 4))
b.WriteString("(R2)\n")
}
b.WriteString(tail)
return b.String()
}
// decimal formats v in decimal.
func decimal(v int) string {
if v == 0 {
return "0"
}
var buf [20]byte
i := len(buf)
for v > 0 {
i--
buf[i] = byte('0' + v%10)
v /= 10
}
return string(buf[i:])
}
// assemblePooled parses and assembles a generated kernel, returning the
// image and the single function's code words.
func assemblePooled(t *testing.T, head string, n int, tail string) (*Image, []uint32) {
t.Helper()
src := pooledKernel(head, n, tail)
dir := t.TempDir()
path := filepath.Join(dir, "poolmid_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("functions = %d, want 1", len(img.Funcs))
}
fn := img.Funcs[0]
return img, leWords(img.Code[fn.Offset : fn.Offset+fn.Size])
}
// a64BranchWord reports whether w is an unconditional B: op 000101 in bits
// 31..26, the encoding the flush guards ride.
func a64BranchWord(w uint32) bool {
return w>>26 == 0x05
}
// a64BranchTarget decodes a B word's target byte offset from its own pc.
func a64BranchTarget(w uint32, pc int) int {
d := int(w & 0x03FFFFFF)
if d&(1<<25) != 0 {
d |= ^0x03FFFFFF
}
return pc + 4*d
}
// a64LoadLiteral reports whether w is a LDR W/X literal (the pool's loads
// into REGTMP: word forms 0x18 and 0x58 in the top byte) and decodes its
// target byte displacement.
func a64LoadLiteral(w uint32) (int, bool) {
if w>>24 != 0x18 && w>>24 != 0x58 {
return 0, false
}
d := int(w>>5) & 0x7FFFF
if d&(1<<18) != 0 {
d |= ^0x7FFFF
}
return d * 4, true
}
// TestArm64PoolFlushStructure assembles the distance-bound kernel and pins
// the flush structure on gasm's own image: exactly one mid-body branch over
// a drained segment, no code words inside it, every load literal inside the
// conservative displacement bound, and the segment's byte range free of line
// rows: the pool words carry the flushing statement's source line, so the
// pc-line tables see no delta across them.
func TestArm64PoolFlushStructure(t *testing.T) {
const n = 44000 // one flush: the bound arrives at about 43690 statements
img, words := assemblePooled(t, "", n, "\tRET\n")
fn := img.Funcs[0]
var branches []int
for i, w := range words {
if a64BranchWord(w) {
branches = append(branches, i*4)
}
}
if len(branches) != 1 {
t.Fatalf("branch words = %d, want exactly the one flush guard", len(branches))
}
branchPC := branches[0]
target := a64BranchTarget(words[branchPC/4], branchPC)
segStart, segEnd := branchPC+4, target
if segEnd <= segStart || segEnd%4 != 0 {
t.Fatalf("flush branch target %d leaves no legal word range after %d", target, branchPC)
}
if segEnd+4 > len(words)*4 {
t.Fatalf("flush branch target %d runs past the image (%d words)", target, len(words))
}
// The drained segment holds pool words only: four-byte constants, no
// branch opcodes, no literal loads.
for off := segStart; off < segEnd; off += 4 {
w := words[off/4]
if a64BranchWord(w) {
t.Fatalf("word at %d inside the drained segment is a branch: %08x", off, w)
}
if _, lit := a64LoadLiteral(w); lit {
t.Fatalf("word at %d inside the drained segment is a literal load: %08x", off, w)
}
}
// Every load literal resolves inside the conservative bound and inside
// the image: the property the flush exists to keep.
for i, w := range words {
d, ok := a64LoadLiteral(w)
if !ok {
continue
}
pc := i * 4
if pc+d < 0 || pc+d >= len(words)*4 {
t.Fatalf("literal load at %d targets %d, outside the image", pc, pc+d)
}
if d >= a64MaxPCDisp || d <= -a64MaxPCDisp {
t.Fatalf("literal load at %d displaces %d, outside the conservative bound", pc, d)
}
}
// The drained words carry no line rows of their own: the toolchain gives
// them the flushing statement's Pos so the pc-line tables see no delta.
for _, l := range fn.Lines {
if l.Offset >= segStart && l.Offset < segEnd {
t.Fatalf("line row at %d sits inside the drained segment [%d, %d)", l.Offset, segStart, segEnd)
}
}
}
// TestArm64PoolFlushDifferential assembles the kernel family with gasm and
// with the installed toolchain and holds the bytes equal, the toolchain's
// tail alignment padding excepted: one flush, a fall-through end behind the
// UNDEF guard, two flushes, and a splitting function whose guard prefix and
// morestack block sit behind a shifted pool.
func TestArm64PoolFlushDifferential(t *testing.T) {
if testing.Short() {
t.Skip("live go tool asm oracle: skipped in -short mode")
}
for _, k := range []struct {
name string
head string
n int
tail string
}{
{"one flush, RET end", "", 44000, "\tRET\n"},
{"one flush, UNDEF end", "", 44000, ""},
{"two flushes", "", 90000, "\tRET\n"},
{"split function", "TEXT \u00b7poolmid(SB), $8-0", 44000, "\tRET\n"},
} {
t.Run(k.name, func(t *testing.T) {
if k.head != "" {
// The split variant spells its own TEXT: the generator's
// NOSPLIT line must give way to it.
runPoolFlushCase(t, pooledKernelFor(k.head, k.n, k.tail))
return
}
runPoolFlushCase(t, pooledKernel("", k.n, k.tail))
})
}
}
// pooledKernelFor renders the kernel with an explicit TEXT line, the split
// variant's shape: no NOSPLIT, so the frame forces the stack-split guard.
func pooledKernelFor(text string, n int, tail string) string {
var b strings.Builder
b.WriteString("#include \"textflag.h\"\n")
b.WriteString(text)
b.WriteString("\n")
for i := range n {
b.WriteString("\tMOVD\tR1, ")
b.WriteString(decimal(0x2000000 + 8*i + 4))
b.WriteString("(R2)\n")
}
b.WriteString(tail)
return b.String()
}
// runPoolFlushCase assembles one generated kernel both ways and compares
// the function's bytes, relocation sites masked, the toolchain's trailing
// alignment zeros excepted.
func runPoolFlushCase(t *testing.T, src string) {
t.Helper()
dir := t.TempDir()
path := filepath.Join(dir, "poolmid_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
f, errs := parser.Parse(path, src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
gt := oracleFuncCode(t, toolAsmObject(t, path, "arm64"))
byLocal := make(map[string][]byte, len(gt))
for name, code := range gt {
if _, after, ok := strings.Cut(name, "."); ok {
name = after
}
byLocal[name] = code
}
for _, fn := range img.Funcs {
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
goCode, ok := byLocal[fn.Name]
if !ok {
t.Fatalf("%s: not in the oracle output (%d functions)", fn.Name, len(gt))
}
goCode = maskCode(append([]byte(nil), goCode...), fn.Relocs)
cmpLen := min(len(goCode), len(gasmCode))
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
for w := 0; w < cmpLen/4; w++ {
g := binary.LittleEndian.Uint32(gasmCode[w*4:])
o := binary.LittleEndian.Uint32(goCode[w*4:])
if g != o {
t.Fatalf("%s: word %d (offset %d) differs: gasm %08x go %08x", fn.Name, w, w*4, g, o)
}
}
t.Fatalf("%s: prefixes equal but lengths differ (gasm %d, oracle %d)", fn.Name, len(gasmCode), len(goCode))
}
for _, b := range goCode[len(gasmCode):] {
if b != 0 {
t.Fatalf("%s: non-zero trailing bytes in the oracle output", fn.Name)
}
}
}
}
+57 -1
View File
@@ -7,7 +7,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// parseArm64File is a helper assembling one arm64 source file.
@@ -124,3 +124,59 @@ func TestArm64GOObjRelocTypes(t *testing.T) {
// The detailed layout is covered by the goobj tests; here we only pin
// that emission succeeds with the new relocation kinds in play.
}
// TestArm64TLSLoad pins the local-exec TLS load: a symbol the file's own
// GLOBL marks TLSBSS loads as one MOVZ word carrying the R_ARM64_TLS_LE
// relocation (asm7.go case 69), the shape `go tool asm` emits for the
// runtime's tls_g accesses. A non-TLS GLOBL keeps the ADRP+LDR pair.
func TestArm64TLSLoad(t *testing.T) {
img := parseArm64File(t, "#include \"textflag.h\"\n\n"+
"TEXT \u00b7f(SB), NOSPLIT, $0-0\n"+
"\tMOVD tlsvar(SB), R0\n"+
"\tMOVD plain(SB), R1\n"+
"\tRET\n"+
"GLOBL tlsvar(SB), TLSBSS, $8\n"+
"GLOBL plain(SB), NOPTR, $8\n")
fn := img.Funcs[0]
if fn.Size != 4+8+4 {
t.Fatalf("function size = %d, want 16", fn.Size)
}
if w := binary.LittleEndian.Uint32(img.Code[fn.Offset:]); w != 0xd2800000 {
t.Errorf("TLS load word = %08x, want MOVZ 0 (d2800000)", w)
}
var tlsSeen, plainSeen bool
for _, r := range fn.Relocs {
if r.Name != "tlsvar" {
continue
}
tlsSeen = true
if r.Kind != RelArm64TLSLE {
t.Errorf("tlsvar reloc kind = %v, want RelArm64TLSLE", r.Kind)
}
if r.Off != 0 {
t.Errorf("tlsvar reloc off = %d, want 0", r.Off)
}
}
for _, r := range fn.Relocs {
if r.Name == "plain" && r.Kind == RelArm64LDST64 {
plainSeen = true
}
}
if !tlsSeen {
t.Error("no tlsvar relocation recorded")
}
if !plainSeen {
t.Error("the plain GLOBL load lost its ADRP+LDR relocation")
}
// The other widths have no TLS row: the toolchain refuses them.
f, errs := parser.Parse("k_arm64.s", "TEXT \u00b7f(SB), NOSPLIT, $0-0\n"+
"\tMOVW tlsvar(SB), R0\n"+
"\tRET\n"+
"GLOBL tlsvar(SB), TLSBSS, $8\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Error("MOVW of a TLS symbol assembled, want an illegal combination")
}
}
+600
View File
@@ -0,0 +1,600 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import "strings"
// arm64 system registers and system-instruction aliases.
//
// The tables are transcribed from the data the Go toolchain itself carries
// (cmd/internal/obj/arm64/sysRegEnc.go and the sysInstFields map of asm7.go),
// which the ARM ARM defines: every system register is the packed field set
// op0<<19 | op1<<16 | CRn<<12 | CRm<<8 | op2<<5, and the read/write flags are
// the toolchain's own access classification. The encoding tables live here so
// the encoder stays testable against the GOROOT testdata word for word.
// a64SysReg is one system register: the packed encoding fields and the
// directions the register supports.
type a64SysReg struct {
v uint32
read bool
write bool
}
// a64SysRegs maps the system register names the toolchain knows to their
// encodings. MRS reads 0xd5300000 | v | Rd and MSR writes
// 0xd5100000 | v | Rt.
var a64SysRegs = map[string]a64SysReg{
"ACTLR_EL1": a64SysReg{0x181020, true, true},
"AFSR0_EL1": a64SysReg{0x185100, true, true},
"AFSR1_EL1": a64SysReg{0x185120, true, true},
"AIDR_EL1": a64SysReg{0x1900e0, true, false},
"AMAIR_EL1": a64SysReg{0x18a300, true, true},
"AMCFGR_EL0": a64SysReg{0x1bd220, true, false},
"AMCGCR_EL0": a64SysReg{0x1bd240, true, false},
"AMCNTENCLR0_EL0": a64SysReg{0x1bd280, true, true},
"AMCNTENCLR1_EL0": a64SysReg{0x1bd300, true, true},
"AMCNTENSET0_EL0": a64SysReg{0x1bd2a0, true, true},
"AMCNTENSET1_EL0": a64SysReg{0x1bd320, true, true},
"AMCR_EL0": a64SysReg{0x1bd200, true, true},
"AMEVCNTR00_EL0": a64SysReg{0x1bd400, true, true},
"AMEVCNTR01_EL0": a64SysReg{0x1bd420, true, true},
"AMEVCNTR02_EL0": a64SysReg{0x1bd440, true, true},
"AMEVCNTR03_EL0": a64SysReg{0x1bd460, true, true},
"AMEVCNTR04_EL0": a64SysReg{0x1bd480, true, true},
"AMEVCNTR05_EL0": a64SysReg{0x1bd4a0, true, true},
"AMEVCNTR06_EL0": a64SysReg{0x1bd4c0, true, true},
"AMEVCNTR07_EL0": a64SysReg{0x1bd4e0, true, true},
"AMEVCNTR08_EL0": a64SysReg{0x1bd500, true, true},
"AMEVCNTR09_EL0": a64SysReg{0x1bd520, true, true},
"AMEVCNTR010_EL0": a64SysReg{0x1bd540, true, true},
"AMEVCNTR011_EL0": a64SysReg{0x1bd560, true, true},
"AMEVCNTR012_EL0": a64SysReg{0x1bd580, true, true},
"AMEVCNTR013_EL0": a64SysReg{0x1bd5a0, true, true},
"AMEVCNTR014_EL0": a64SysReg{0x1bd5c0, true, true},
"AMEVCNTR015_EL0": a64SysReg{0x1bd5e0, true, true},
"AMEVCNTR10_EL0": a64SysReg{0x1bdc00, true, true},
"AMEVCNTR11_EL0": a64SysReg{0x1bdc20, true, true},
"AMEVCNTR12_EL0": a64SysReg{0x1bdc40, true, true},
"AMEVCNTR13_EL0": a64SysReg{0x1bdc60, true, true},
"AMEVCNTR14_EL0": a64SysReg{0x1bdc80, true, true},
"AMEVCNTR15_EL0": a64SysReg{0x1bdca0, true, true},
"AMEVCNTR16_EL0": a64SysReg{0x1bdcc0, true, true},
"AMEVCNTR17_EL0": a64SysReg{0x1bdce0, true, true},
"AMEVCNTR18_EL0": a64SysReg{0x1bdd00, true, true},
"AMEVCNTR19_EL0": a64SysReg{0x1bdd20, true, true},
"AMEVCNTR110_EL0": a64SysReg{0x1bdd40, true, true},
"AMEVCNTR111_EL0": a64SysReg{0x1bdd60, true, true},
"AMEVCNTR112_EL0": a64SysReg{0x1bdd80, true, true},
"AMEVCNTR113_EL0": a64SysReg{0x1bdda0, true, true},
"AMEVCNTR114_EL0": a64SysReg{0x1bddc0, true, true},
"AMEVCNTR115_EL0": a64SysReg{0x1bdde0, true, true},
"AMEVTYPER00_EL0": a64SysReg{0x1bd600, true, false},
"AMEVTYPER01_EL0": a64SysReg{0x1bd620, true, false},
"AMEVTYPER02_EL0": a64SysReg{0x1bd640, true, false},
"AMEVTYPER03_EL0": a64SysReg{0x1bd660, true, false},
"AMEVTYPER04_EL0": a64SysReg{0x1bd680, true, false},
"AMEVTYPER05_EL0": a64SysReg{0x1bd6a0, true, false},
"AMEVTYPER06_EL0": a64SysReg{0x1bd6c0, true, false},
"AMEVTYPER07_EL0": a64SysReg{0x1bd6e0, true, false},
"AMEVTYPER08_EL0": a64SysReg{0x1bd700, true, false},
"AMEVTYPER09_EL0": a64SysReg{0x1bd720, true, false},
"AMEVTYPER010_EL0": a64SysReg{0x1bd740, true, false},
"AMEVTYPER011_EL0": a64SysReg{0x1bd760, true, false},
"AMEVTYPER012_EL0": a64SysReg{0x1bd780, true, false},
"AMEVTYPER013_EL0": a64SysReg{0x1bd7a0, true, false},
"AMEVTYPER014_EL0": a64SysReg{0x1bd7c0, true, false},
"AMEVTYPER015_EL0": a64SysReg{0x1bd7e0, true, false},
"AMEVTYPER10_EL0": a64SysReg{0x1bde00, true, true},
"AMEVTYPER11_EL0": a64SysReg{0x1bde20, true, true},
"AMEVTYPER12_EL0": a64SysReg{0x1bde40, true, true},
"AMEVTYPER13_EL0": a64SysReg{0x1bde60, true, true},
"AMEVTYPER14_EL0": a64SysReg{0x1bde80, true, true},
"AMEVTYPER15_EL0": a64SysReg{0x1bdea0, true, true},
"AMEVTYPER16_EL0": a64SysReg{0x1bdec0, true, true},
"AMEVTYPER17_EL0": a64SysReg{0x1bdee0, true, true},
"AMEVTYPER18_EL0": a64SysReg{0x1bdf00, true, true},
"AMEVTYPER19_EL0": a64SysReg{0x1bdf20, true, true},
"AMEVTYPER110_EL0": a64SysReg{0x1bdf40, true, true},
"AMEVTYPER111_EL0": a64SysReg{0x1bdf60, true, true},
"AMEVTYPER112_EL0": a64SysReg{0x1bdf80, true, true},
"AMEVTYPER113_EL0": a64SysReg{0x1bdfa0, true, true},
"AMEVTYPER114_EL0": a64SysReg{0x1bdfc0, true, true},
"AMEVTYPER115_EL0": a64SysReg{0x1bdfe0, true, true},
"AMUSERENR_EL0": a64SysReg{0x1bd260, true, true},
"APDAKeyHi_EL1": a64SysReg{0x182220, true, true},
"APDAKeyLo_EL1": a64SysReg{0x182200, true, true},
"APDBKeyHi_EL1": a64SysReg{0x182260, true, true},
"APDBKeyLo_EL1": a64SysReg{0x182240, true, true},
"APGAKeyHi_EL1": a64SysReg{0x182320, true, true},
"APGAKeyLo_EL1": a64SysReg{0x182300, true, true},
"APIAKeyHi_EL1": a64SysReg{0x182120, true, true},
"APIAKeyLo_EL1": a64SysReg{0x182100, true, true},
"APIBKeyHi_EL1": a64SysReg{0x182160, true, true},
"APIBKeyLo_EL1": a64SysReg{0x182140, true, true},
"CCSIDR2_EL1": a64SysReg{0x190040, true, false},
"CCSIDR_EL1": a64SysReg{0x190000, true, false},
"CLIDR_EL1": a64SysReg{0x190020, true, false},
"CNTFRQ_EL0": a64SysReg{0x1be000, true, true},
"CNTKCTL_EL1": a64SysReg{0x18e100, true, true},
"CNTP_CTL_EL0": a64SysReg{0x1be220, true, true},
"CNTP_CVAL_EL0": a64SysReg{0x1be240, true, true},
"CNTP_TVAL_EL0": a64SysReg{0x1be200, true, true},
"CNTPCT_EL0": a64SysReg{0x1be020, true, false},
"CNTPS_CTL_EL1": a64SysReg{0x1fe220, true, true},
"CNTPS_CVAL_EL1": a64SysReg{0x1fe240, true, true},
"CNTPS_TVAL_EL1": a64SysReg{0x1fe200, true, true},
"CNTV_CTL_EL0": a64SysReg{0x1be320, true, true},
"CNTV_CVAL_EL0": a64SysReg{0x1be340, true, true},
"CNTV_TVAL_EL0": a64SysReg{0x1be300, true, true},
"CNTVCT_EL0": a64SysReg{0x1be040, true, false},
"CONTEXTIDR_EL1": a64SysReg{0x18d020, true, true},
"CPACR_EL1": a64SysReg{0x181040, true, true},
"CSSELR_EL1": a64SysReg{0x1a0000, true, true},
"CTR_EL0": a64SysReg{0x1b0020, true, false},
"CurrentEL": a64SysReg{0x184240, true, false},
"DAIF": a64SysReg{0x1b4220, true, true},
"DBGAUTHSTATUS_EL1": a64SysReg{0x107ec0, true, false},
"DBGBCR0_EL1": a64SysReg{0x1000a0, true, true},
"DBGBCR1_EL1": a64SysReg{0x1001a0, true, true},
"DBGBCR2_EL1": a64SysReg{0x1002a0, true, true},
"DBGBCR3_EL1": a64SysReg{0x1003a0, true, true},
"DBGBCR4_EL1": a64SysReg{0x1004a0, true, true},
"DBGBCR5_EL1": a64SysReg{0x1005a0, true, true},
"DBGBCR6_EL1": a64SysReg{0x1006a0, true, true},
"DBGBCR7_EL1": a64SysReg{0x1007a0, true, true},
"DBGBCR8_EL1": a64SysReg{0x1008a0, true, true},
"DBGBCR9_EL1": a64SysReg{0x1009a0, true, true},
"DBGBCR10_EL1": a64SysReg{0x100aa0, true, true},
"DBGBCR11_EL1": a64SysReg{0x100ba0, true, true},
"DBGBCR12_EL1": a64SysReg{0x100ca0, true, true},
"DBGBCR13_EL1": a64SysReg{0x100da0, true, true},
"DBGBCR14_EL1": a64SysReg{0x100ea0, true, true},
"DBGBCR15_EL1": a64SysReg{0x100fa0, true, true},
"DBGBVR0_EL1": a64SysReg{0x100080, true, true},
"DBGBVR1_EL1": a64SysReg{0x100180, true, true},
"DBGBVR2_EL1": a64SysReg{0x100280, true, true},
"DBGBVR3_EL1": a64SysReg{0x100380, true, true},
"DBGBVR4_EL1": a64SysReg{0x100480, true, true},
"DBGBVR5_EL1": a64SysReg{0x100580, true, true},
"DBGBVR6_EL1": a64SysReg{0x100680, true, true},
"DBGBVR7_EL1": a64SysReg{0x100780, true, true},
"DBGBVR8_EL1": a64SysReg{0x100880, true, true},
"DBGBVR9_EL1": a64SysReg{0x100980, true, true},
"DBGBVR10_EL1": a64SysReg{0x100a80, true, true},
"DBGBVR11_EL1": a64SysReg{0x100b80, true, true},
"DBGBVR12_EL1": a64SysReg{0x100c80, true, true},
"DBGBVR13_EL1": a64SysReg{0x100d80, true, true},
"DBGBVR14_EL1": a64SysReg{0x100e80, true, true},
"DBGBVR15_EL1": a64SysReg{0x100f80, true, true},
"DBGCLAIMCLR_EL1": a64SysReg{0x1079c0, true, true},
"DBGCLAIMSET_EL1": a64SysReg{0x1078c0, true, true},
"DBGDTR_EL0": a64SysReg{0x130400, true, true},
"DBGDTRRX_EL0": a64SysReg{0x130500, true, false},
"DBGDTRTX_EL0": a64SysReg{0x130500, false, true},
"DBGPRCR_EL1": a64SysReg{0x101480, true, true},
"DBGWCR0_EL1": a64SysReg{0x1000e0, true, true},
"DBGWCR1_EL1": a64SysReg{0x1001e0, true, true},
"DBGWCR2_EL1": a64SysReg{0x1002e0, true, true},
"DBGWCR3_EL1": a64SysReg{0x1003e0, true, true},
"DBGWCR4_EL1": a64SysReg{0x1004e0, true, true},
"DBGWCR5_EL1": a64SysReg{0x1005e0, true, true},
"DBGWCR6_EL1": a64SysReg{0x1006e0, true, true},
"DBGWCR7_EL1": a64SysReg{0x1007e0, true, true},
"DBGWCR8_EL1": a64SysReg{0x1008e0, true, true},
"DBGWCR9_EL1": a64SysReg{0x1009e0, true, true},
"DBGWCR10_EL1": a64SysReg{0x100ae0, true, true},
"DBGWCR11_EL1": a64SysReg{0x100be0, true, true},
"DBGWCR12_EL1": a64SysReg{0x100ce0, true, true},
"DBGWCR13_EL1": a64SysReg{0x100de0, true, true},
"DBGWCR14_EL1": a64SysReg{0x100ee0, true, true},
"DBGWCR15_EL1": a64SysReg{0x100fe0, true, true},
"DBGWVR0_EL1": a64SysReg{0x1000c0, true, true},
"DBGWVR1_EL1": a64SysReg{0x1001c0, true, true},
"DBGWVR2_EL1": a64SysReg{0x1002c0, true, true},
"DBGWVR3_EL1": a64SysReg{0x1003c0, true, true},
"DBGWVR4_EL1": a64SysReg{0x1004c0, true, true},
"DBGWVR5_EL1": a64SysReg{0x1005c0, true, true},
"DBGWVR6_EL1": a64SysReg{0x1006c0, true, true},
"DBGWVR7_EL1": a64SysReg{0x1007c0, true, true},
"DBGWVR8_EL1": a64SysReg{0x1008c0, true, true},
"DBGWVR9_EL1": a64SysReg{0x1009c0, true, true},
"DBGWVR10_EL1": a64SysReg{0x100ac0, true, true},
"DBGWVR11_EL1": a64SysReg{0x100bc0, true, true},
"DBGWVR12_EL1": a64SysReg{0x100cc0, true, true},
"DBGWVR13_EL1": a64SysReg{0x100dc0, true, true},
"DBGWVR14_EL1": a64SysReg{0x100ec0, true, true},
"DBGWVR15_EL1": a64SysReg{0x100fc0, true, true},
"DCZID_EL0": a64SysReg{0x1b00e0, true, false},
"DISR_EL1": a64SysReg{0x18c120, true, true},
"DIT": a64SysReg{0x1b42a0, true, true},
"DLR_EL0": a64SysReg{0x1b4520, true, true},
"DSPSR_EL0": a64SysReg{0x1b4500, true, true},
"ELR_EL1": a64SysReg{0x184020, true, true},
"ERRIDR_EL1": a64SysReg{0x185300, true, false},
"ERRSELR_EL1": a64SysReg{0x185320, true, true},
"ERXADDR_EL1": a64SysReg{0x185460, true, true},
"ERXCTLR_EL1": a64SysReg{0x185420, true, true},
"ERXFR_EL1": a64SysReg{0x185400, true, false},
"ERXMISC0_EL1": a64SysReg{0x185500, true, true},
"ERXMISC1_EL1": a64SysReg{0x185520, true, true},
"ERXMISC2_EL1": a64SysReg{0x185540, true, true},
"ERXMISC3_EL1": a64SysReg{0x185560, true, true},
"ERXPFGCDN_EL1": a64SysReg{0x1854c0, true, true},
"ERXPFGCTL_EL1": a64SysReg{0x1854a0, true, true},
"ERXPFGF_EL1": a64SysReg{0x185480, true, false},
"ERXSTATUS_EL1": a64SysReg{0x185440, true, true},
"ESR_EL1": a64SysReg{0x185200, true, true},
"FAR_EL1": a64SysReg{0x186000, true, true},
"FPCR": a64SysReg{0x1b4400, true, true},
"FPSR": a64SysReg{0x1b4420, true, true},
"GCR_EL1": a64SysReg{0x1810c0, true, true},
"GMID_EL1": a64SysReg{0x31400, true, false},
"ICC_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICC_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICC_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICC_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICC_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICC_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICC_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICC_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICC_ASGI1R_EL1": a64SysReg{0x18cbc0, false, true},
"ICC_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICC_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICC_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICC_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICC_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICC_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICC_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICC_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICC_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICC_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICC_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICC_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICC_PMR_EL1": a64SysReg{0x184600, true, true},
"ICC_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ICC_SGI0R_EL1": a64SysReg{0x18cbe0, false, true},
"ICC_SGI1R_EL1": a64SysReg{0x18cba0, false, true},
"ICC_SRE_EL1": a64SysReg{0x18cca0, true, true},
"ICV_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICV_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICV_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICV_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICV_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICV_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICV_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICV_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICV_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICV_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICV_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICV_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICV_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICV_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICV_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICV_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICV_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICV_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICV_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICV_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICV_PMR_EL1": a64SysReg{0x184600, true, true},
"ICV_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ID_AA64AFR0_EL1": a64SysReg{0x180580, true, false},
"ID_AA64AFR1_EL1": a64SysReg{0x1805a0, true, false},
"ID_AA64DFR0_EL1": a64SysReg{0x180500, true, false},
"ID_AA64DFR1_EL1": a64SysReg{0x180520, true, false},
"ID_AA64ISAR0_EL1": a64SysReg{0x180600, true, false},
"ID_AA64ISAR1_EL1": a64SysReg{0x180620, true, false},
"ID_AA64MMFR0_EL1": a64SysReg{0x180700, true, false},
"ID_AA64MMFR1_EL1": a64SysReg{0x180720, true, false},
"ID_AA64MMFR2_EL1": a64SysReg{0x180740, true, false},
"ID_AA64PFR0_EL1": a64SysReg{0x180400, true, false},
"ID_AA64PFR1_EL1": a64SysReg{0x180420, true, false},
"ID_AA64ZFR0_EL1": a64SysReg{0x180480, true, false},
"ID_AFR0_EL1": a64SysReg{0x180160, true, false},
"ID_DFR0_EL1": a64SysReg{0x180140, true, false},
"ID_ISAR0_EL1": a64SysReg{0x180200, true, false},
"ID_ISAR1_EL1": a64SysReg{0x180220, true, false},
"ID_ISAR2_EL1": a64SysReg{0x180240, true, false},
"ID_ISAR3_EL1": a64SysReg{0x180260, true, false},
"ID_ISAR4_EL1": a64SysReg{0x180280, true, false},
"ID_ISAR5_EL1": a64SysReg{0x1802a0, true, false},
"ID_ISAR6_EL1": a64SysReg{0x1802e0, true, false},
"ID_MMFR0_EL1": a64SysReg{0x180180, true, false},
"ID_MMFR1_EL1": a64SysReg{0x1801a0, true, false},
"ID_MMFR2_EL1": a64SysReg{0x1801c0, true, false},
"ID_MMFR3_EL1": a64SysReg{0x1801e0, true, false},
"ID_MMFR4_EL1": a64SysReg{0x1802c0, true, false},
"ID_PFR0_EL1": a64SysReg{0x180100, true, false},
"ID_PFR1_EL1": a64SysReg{0x180120, true, false},
"ID_PFR2_EL1": a64SysReg{0x180380, true, false},
"ISR_EL1": a64SysReg{0x18c100, true, false},
"LORC_EL1": a64SysReg{0x18a460, true, true},
"LOREA_EL1": a64SysReg{0x18a420, true, true},
"LORID_EL1": a64SysReg{0x18a4e0, true, false},
"LORN_EL1": a64SysReg{0x18a440, true, true},
"LORSA_EL1": a64SysReg{0x18a400, true, true},
"MAIR_EL1": a64SysReg{0x18a200, true, true},
"MDCCINT_EL1": a64SysReg{0x100200, true, true},
"MDCCSR_EL0": a64SysReg{0x130100, true, false},
"MDRAR_EL1": a64SysReg{0x101000, true, false},
"MDSCR_EL1": a64SysReg{0x100240, true, true},
"MIDR_EL1": a64SysReg{0x180000, true, false},
"MPAM0_EL1": a64SysReg{0x18a520, true, true},
"MPAM1_EL1": a64SysReg{0x18a500, true, true},
"MPAMIDR_EL1": a64SysReg{0x18a480, true, false},
"MPIDR_EL1": a64SysReg{0x1800a0, true, false},
"MVFR0_EL1": a64SysReg{0x180300, true, false},
"MVFR1_EL1": a64SysReg{0x180320, true, false},
"MVFR2_EL1": a64SysReg{0x180340, true, false},
"NZCV": a64SysReg{0x1b4200, true, true},
"OSDLR_EL1": a64SysReg{0x101380, true, true},
"OSDTRRX_EL1": a64SysReg{0x100040, true, true},
"OSDTRTX_EL1": a64SysReg{0x100340, true, true},
"OSECCR_EL1": a64SysReg{0x100640, true, true},
"OSLAR_EL1": a64SysReg{0x101080, false, true},
"OSLSR_EL1": a64SysReg{0x101180, true, false},
"PAN": a64SysReg{0x184260, true, true},
"PAR_EL1": a64SysReg{0x187400, true, true},
"PMBIDR_EL1": a64SysReg{0x189ae0, true, false},
"PMBLIMITR_EL1": a64SysReg{0x189a00, true, true},
"PMBPTR_EL1": a64SysReg{0x189a20, true, true},
"PMBSR_EL1": a64SysReg{0x189a60, true, true},
"PMCCFILTR_EL0": a64SysReg{0x1befe0, true, true},
"PMCCNTR_EL0": a64SysReg{0x1b9d00, true, true},
"PMCEID0_EL0": a64SysReg{0x1b9cc0, true, false},
"PMCEID1_EL0": a64SysReg{0x1b9ce0, true, false},
"PMCNTENCLR_EL0": a64SysReg{0x1b9c40, true, true},
"PMCNTENSET_EL0": a64SysReg{0x1b9c20, true, true},
"PMCR_EL0": a64SysReg{0x1b9c00, true, true},
"PMEVCNTR0_EL0": a64SysReg{0x1be800, true, true},
"PMEVCNTR1_EL0": a64SysReg{0x1be820, true, true},
"PMEVCNTR2_EL0": a64SysReg{0x1be840, true, true},
"PMEVCNTR3_EL0": a64SysReg{0x1be860, true, true},
"PMEVCNTR4_EL0": a64SysReg{0x1be880, true, true},
"PMEVCNTR5_EL0": a64SysReg{0x1be8a0, true, true},
"PMEVCNTR6_EL0": a64SysReg{0x1be8c0, true, true},
"PMEVCNTR7_EL0": a64SysReg{0x1be8e0, true, true},
"PMEVCNTR8_EL0": a64SysReg{0x1be900, true, true},
"PMEVCNTR9_EL0": a64SysReg{0x1be920, true, true},
"PMEVCNTR10_EL0": a64SysReg{0x1be940, true, true},
"PMEVCNTR11_EL0": a64SysReg{0x1be960, true, true},
"PMEVCNTR12_EL0": a64SysReg{0x1be980, true, true},
"PMEVCNTR13_EL0": a64SysReg{0x1be9a0, true, true},
"PMEVCNTR14_EL0": a64SysReg{0x1be9c0, true, true},
"PMEVCNTR15_EL0": a64SysReg{0x1be9e0, true, true},
"PMEVCNTR16_EL0": a64SysReg{0x1bea00, true, true},
"PMEVCNTR17_EL0": a64SysReg{0x1bea20, true, true},
"PMEVCNTR18_EL0": a64SysReg{0x1bea40, true, true},
"PMEVCNTR19_EL0": a64SysReg{0x1bea60, true, true},
"PMEVCNTR20_EL0": a64SysReg{0x1bea80, true, true},
"PMEVCNTR21_EL0": a64SysReg{0x1beaa0, true, true},
"PMEVCNTR22_EL0": a64SysReg{0x1beac0, true, true},
"PMEVCNTR23_EL0": a64SysReg{0x1beae0, true, true},
"PMEVCNTR24_EL0": a64SysReg{0x1beb00, true, true},
"PMEVCNTR25_EL0": a64SysReg{0x1beb20, true, true},
"PMEVCNTR26_EL0": a64SysReg{0x1beb40, true, true},
"PMEVCNTR27_EL0": a64SysReg{0x1beb60, true, true},
"PMEVCNTR28_EL0": a64SysReg{0x1beb80, true, true},
"PMEVCNTR29_EL0": a64SysReg{0x1beba0, true, true},
"PMEVCNTR30_EL0": a64SysReg{0x1bebc0, true, true},
"PMEVTYPER0_EL0": a64SysReg{0x1bec00, true, true},
"PMEVTYPER1_EL0": a64SysReg{0x1bec20, true, true},
"PMEVTYPER2_EL0": a64SysReg{0x1bec40, true, true},
"PMEVTYPER3_EL0": a64SysReg{0x1bec60, true, true},
"PMEVTYPER4_EL0": a64SysReg{0x1bec80, true, true},
"PMEVTYPER5_EL0": a64SysReg{0x1beca0, true, true},
"PMEVTYPER6_EL0": a64SysReg{0x1becc0, true, true},
"PMEVTYPER7_EL0": a64SysReg{0x1bece0, true, true},
"PMEVTYPER8_EL0": a64SysReg{0x1bed00, true, true},
"PMEVTYPER9_EL0": a64SysReg{0x1bed20, true, true},
"PMEVTYPER10_EL0": a64SysReg{0x1bed40, true, true},
"PMEVTYPER11_EL0": a64SysReg{0x1bed60, true, true},
"PMEVTYPER12_EL0": a64SysReg{0x1bed80, true, true},
"PMEVTYPER13_EL0": a64SysReg{0x1beda0, true, true},
"PMEVTYPER14_EL0": a64SysReg{0x1bedc0, true, true},
"PMEVTYPER15_EL0": a64SysReg{0x1bede0, true, true},
"PMEVTYPER16_EL0": a64SysReg{0x1bee00, true, true},
"PMEVTYPER17_EL0": a64SysReg{0x1bee20, true, true},
"PMEVTYPER18_EL0": a64SysReg{0x1bee40, true, true},
"PMEVTYPER19_EL0": a64SysReg{0x1bee60, true, true},
"PMEVTYPER20_EL0": a64SysReg{0x1bee80, true, true},
"PMEVTYPER21_EL0": a64SysReg{0x1beea0, true, true},
"PMEVTYPER22_EL0": a64SysReg{0x1beec0, true, true},
"PMEVTYPER23_EL0": a64SysReg{0x1beee0, true, true},
"PMEVTYPER24_EL0": a64SysReg{0x1bef00, true, true},
"PMEVTYPER25_EL0": a64SysReg{0x1bef20, true, true},
"PMEVTYPER26_EL0": a64SysReg{0x1bef40, true, true},
"PMEVTYPER27_EL0": a64SysReg{0x1bef60, true, true},
"PMEVTYPER28_EL0": a64SysReg{0x1bef80, true, true},
"PMEVTYPER29_EL0": a64SysReg{0x1befa0, true, true},
"PMEVTYPER30_EL0": a64SysReg{0x1befc0, true, true},
"PMINTENCLR_EL1": a64SysReg{0x189e40, true, true},
"PMINTENSET_EL1": a64SysReg{0x189e20, true, true},
"PMMIR_EL1": a64SysReg{0x189ec0, true, false},
"PMOVSCLR_EL0": a64SysReg{0x1b9c60, true, true},
"PMOVSSET_EL0": a64SysReg{0x1b9e60, true, true},
"PMSCR_EL1": a64SysReg{0x189900, true, true},
"PMSELR_EL0": a64SysReg{0x1b9ca0, true, true},
"PMSEVFR_EL1": a64SysReg{0x1899a0, true, true},
"PMSFCR_EL1": a64SysReg{0x189980, true, true},
"PMSICR_EL1": a64SysReg{0x189940, true, true},
"PMSIDR_EL1": a64SysReg{0x1899e0, true, false},
"PMSIRR_EL1": a64SysReg{0x189960, true, true},
"PMSLATFR_EL1": a64SysReg{0x1899c0, true, true},
"PMSWINC_EL0": a64SysReg{0x1b9c80, false, true},
"PMUSERENR_EL0": a64SysReg{0x1b9e00, true, true},
"PMXEVCNTR_EL0": a64SysReg{0x1b9d40, true, true},
"PMXEVTYPER_EL0": a64SysReg{0x1b9d20, true, true},
"REVIDR_EL1": a64SysReg{0x1800c0, true, false},
"RGSR_EL1": a64SysReg{0x1810a0, true, true},
"RMR_EL1": a64SysReg{0x18c040, true, true},
"RNDR": a64SysReg{0x1b2400, true, false},
"RNDRRS": a64SysReg{0x1b2420, true, false},
"RVBAR_EL1": a64SysReg{0x18c020, true, false},
"SCTLR_EL1": a64SysReg{0x181000, true, true},
"SCXTNUM_EL0": a64SysReg{0x1bd0e0, true, true},
"SCXTNUM_EL1": a64SysReg{0x18d0e0, true, true},
"SP_EL0": a64SysReg{0x184100, true, true},
"SP_EL1": a64SysReg{0x1c4100, true, true},
"SPSel": a64SysReg{0x184200, true, true},
"SPSR_abt": a64SysReg{0x1c4320, true, true},
"SPSR_EL1": a64SysReg{0x184000, true, true},
"SPSR_fiq": a64SysReg{0x1c4360, true, true},
"SPSR_irq": a64SysReg{0x1c4300, true, true},
"SPSR_und": a64SysReg{0x1c4340, true, true},
"SSBS": a64SysReg{0x1b42c0, true, true},
"TCO": a64SysReg{0x1b42e0, true, true},
"TCR_EL1": a64SysReg{0x182040, true, true},
"TFSR_EL1": a64SysReg{0x185600, true, true},
"TFSRE0_EL1": a64SysReg{0x185620, true, true},
"TPIDR_EL0": a64SysReg{0x1bd040, true, true},
"TPIDR_EL1": a64SysReg{0x18d080, true, true},
"TPIDRRO_EL0": a64SysReg{0x1bd060, true, true},
"TRFCR_EL1": a64SysReg{0x181220, true, true},
"TTBR0_EL1": a64SysReg{0x182000, true, true},
"TTBR1_EL1": a64SysReg{0x182020, true, true},
"UAO": a64SysReg{0x184280, true, true},
"VBAR_EL1": a64SysReg{0x18c000, true, true},
"ZCR_EL1": a64SysReg{0x181200, true, true},
}
// a64SysInst is one TLBI alias: the fields the SYS encoding carries beside
// the fixed op0 = 01 and CRn = 8.
type a64SysInst struct {
op1, cm, op2 uint32
}
// a64TLBIOps maps the TLBI operation names to their fields; the register
// operand is optional and defaults to ZR.
var a64TLBIOps = map[string]a64SysInst{
"ALLE1": {0x4, 0x7, 0x4},
"ALLE1IS": {0x4, 0x3, 0x4},
"ALLE1OS": {0x4, 0x1, 0x4},
"ALLE2": {0x4, 0x7, 0x0},
"ALLE2IS": {0x4, 0x3, 0x0},
"ALLE2OS": {0x4, 0x1, 0x0},
"ALLE3": {0x6, 0x7, 0x0},
"ALLE3IS": {0x6, 0x3, 0x0},
"ALLE3OS": {0x6, 0x1, 0x0},
"ASIDE1": {0x0, 0x7, 0x2},
"ASIDE1IS": {0x0, 0x3, 0x2},
"ASIDE1OS": {0x0, 0x1, 0x2},
"IPAS2E1": {0x4, 0x4, 0x1},
"IPAS2E1IS": {0x4, 0x0, 0x1},
"IPAS2E1OS": {0x4, 0x4, 0x0},
"IPAS2LE1": {0x4, 0x4, 0x5},
"IPAS2LE1IS": {0x4, 0x0, 0x5},
"IPAS2LE1OS": {0x4, 0x4, 0x4},
"RIPAS2E1": {0x4, 0x4, 0x2},
"RIPAS2E1IS": {0x4, 0x0, 0x2},
"RIPAS2E1OS": {0x4, 0x4, 0x3},
"RIPAS2LE1": {0x4, 0x4, 0x6},
"RIPAS2LE1IS": {0x4, 0x0, 0x6},
"RIPAS2LE1OS": {0x4, 0x4, 0x7},
"RVAAE1": {0x0, 0x6, 0x3},
"RVAAE1IS": {0x0, 0x2, 0x3},
"RVAAE1OS": {0x0, 0x5, 0x3},
"RVAALE1": {0x0, 0x6, 0x7},
"RVAALE1IS": {0x0, 0x2, 0x7},
"RVAALE1OS": {0x0, 0x5, 0x7},
"RVAE1": {0x0, 0x6, 0x1},
"RVAE1IS": {0x0, 0x2, 0x1},
"RVAE1OS": {0x0, 0x5, 0x1},
"RVAE2": {0x4, 0x6, 0x1},
"RVAE2IS": {0x4, 0x2, 0x1},
"RVAE2OS": {0x4, 0x5, 0x1},
"RVAE3": {0x6, 0x6, 0x1},
"RVAE3IS": {0x6, 0x2, 0x1},
"RVAE3OS": {0x6, 0x5, 0x1},
"RVALE1": {0x0, 0x6, 0x5},
"RVALE1IS": {0x0, 0x2, 0x5},
"RVALE1OS": {0x0, 0x5, 0x5},
"RVALE2": {0x4, 0x6, 0x5},
"RVALE2IS": {0x4, 0x2, 0x5},
"RVALE2OS": {0x4, 0x5, 0x5},
"RVALE3": {0x6, 0x6, 0x5},
"RVALE3IS": {0x6, 0x2, 0x5},
"RVALE3OS": {0x6, 0x5, 0x5},
"VAAE1": {0x0, 0x7, 0x3},
"VAAE1IS": {0x0, 0x3, 0x3},
"VAAE1OS": {0x0, 0x1, 0x3},
"VAALE1": {0x0, 0x7, 0x7},
"VAALE1IS": {0x0, 0x3, 0x7},
"VAALE1OS": {0x0, 0x1, 0x7},
"VAE1": {0x0, 0x7, 0x1},
"VAE1IS": {0x0, 0x3, 0x1},
"VAE1OS": {0x0, 0x1, 0x1},
"VAE2": {0x4, 0x7, 0x1},
"VAE2IS": {0x4, 0x3, 0x1},
"VAE2OS": {0x4, 0x1, 0x1},
"VAE3": {0x6, 0x7, 0x1},
"VAE3IS": {0x6, 0x3, 0x1},
"VAE3OS": {0x6, 0x1, 0x1},
"VALE1": {0x0, 0x7, 0x5},
"VALE1IS": {0x0, 0x3, 0x5},
"VALE1OS": {0x0, 0x1, 0x5},
"VALE2": {0x4, 0x7, 0x5},
"VALE2IS": {0x4, 0x3, 0x5},
"VALE2OS": {0x4, 0x1, 0x5},
"VALE3": {0x6, 0x7, 0x5},
"VALE3IS": {0x6, 0x3, 0x5},
"VALE3OS": {0x6, 0x1, 0x5},
"VMALLE1": {0x0, 0x7, 0x0},
"VMALLE1IS": {0x0, 0x3, 0x0},
"VMALLE1OS": {0x0, 0x1, 0x0},
"VMALLS12E1": {0x4, 0x7, 0x6},
"VMALLS12E1IS": {0x4, 0x3, 0x6},
"VMALLS12E1OS": {0x4, 0x1, 0x6},
}
// arm64TLBITakesReg reports whether a TLBI operation spells a by-address
// invalidation that carries the address in its optional second register
// (asm7.go's sysInstFields hasOperand2). The whole-entry spellings are
// exactly the VMALL-prefixed and the ALL-prefixed ones; everything else
// (VAE*, VAAE*, VALE*, RVAA*, RVAE*, ASIDE1*, IPAS2*, RIPAS2*) is
// by-address and takes the register.
func arm64TLBITakesReg(name string) bool {
return !strings.HasPrefix(name, "VMALL") && !strings.HasPrefix(name, "ALL")
}
// a64DCOps2 maps the DC operation names to their fields; the register
// operand is mandatory.
var a64DCOps2 = map[string]a64SysInst{
"CGDSW": {0x0, 0xa, 0x6},
"CGDVAC": {0x3, 0xa, 0x5},
"CGDVADP": {0x3, 0xd, 0x5},
"CGDVAP": {0x3, 0xc, 0x5},
"CGSW": {0x0, 0xa, 0x4},
"CGVAC": {0x3, 0xa, 0x3},
"CGVADP": {0x3, 0xd, 0x3},
"CGVAP": {0x3, 0xc, 0x3},
"CIGDSW": {0x0, 0xe, 0x6},
"CIGDVAC": {0x3, 0xe, 0x5},
"CIGSW": {0x0, 0xe, 0x4},
"CIGVAC": {0x3, 0xe, 0x3},
"CISW": {0x0, 0xe, 0x2},
"CIVAC": {0x3, 0xe, 0x1},
"CSW": {0x0, 0xa, 0x2},
"CVAC": {0x3, 0xa, 0x1},
"CVADP": {0x3, 0xd, 0x1},
"CVAP": {0x3, 0xc, 0x1},
"CVAU": {0x3, 0xb, 0x1},
"GVA": {0x3, 0x4, 0x3},
"GZVA": {0x3, 0x4, 0x4},
"IGDSW": {0x0, 0x6, 0x6},
"IGDVAC": {0x0, 0x6, 0x5},
"IGSW": {0x0, 0x6, 0x4},
"IGVAC": {0x0, 0x6, 0x3},
"ISW": {0x0, 0x6, 0x2},
"IVAC": {0x0, 0x6, 0x1},
"ZVA": {0x3, 0x4, 0x1},
}
// a64RPRFOps maps the range-prefetch operation names to their 6-bit values.
var a64RPRFOps = map[string]uint32{
"PLDKEEP": 0,
"PLDSTRM": 4,
"PSTKEEP": 1,
"PSTSTRM": 5,
}
+164
View File
@@ -0,0 +1,164 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestARM64SysRegsDifferential proves the whole system-register table against
// the toolchain at once: one TEXT whose body reads every register the table
// carries (and writes every writable one), assembled by gasm and by
// go tool asm, must agree byte for byte. A single wrong op0/op1/CRn/CRm/op2
// packing names its register through the first differing word.
func TestARM64SysRegsDifferential(t *testing.T) {
names := make([]string, 0, len(a64SysRegs))
for name := range a64SysRegs {
names = append(names, name)
}
slices.Sort(names)
var body strings.Builder
for i, name := range names {
// R18 is the arm64 platform register and R29-R31 carry dedicated
// meanings; a plain read/write destination keeps to R0-R17.
reg := fmt.Sprintf("R%d", i%18)
if a64SysRegs[name].read {
body.WriteString(fmt.Sprintf("\tMRS %s, %s\n", name, reg))
}
if a64SysRegs[name].write {
body.WriteString(fmt.Sprintf("\tMSR %s, %s\n", reg, name))
}
}
src := "#include \"textflag.h\"\n\nTEXT ·sysregs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n"
dir := t.TempDir()
path := filepath.Join(dir, "sysregs_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "sysregs")
}
// TestARM64FamiliesDifferential pins the non-sysreg families the arm64
// campaign added: the LSE compare-and-swap pairs, the VMOVI immediate, the
// SIMD narrow/long shift pairs, the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// structure accesses with their post-index and replicate forms, LDPSW, the
// pointer-authentication hint and the DC maintenance operation. Every
// spelling is the toolchain's own, taken from its arm64 testdata, and the
// bytes must agree word for word.
func TestARM64FamiliesDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·families(SB), NOSPLIT, $0
CASPD (R2, R3), (R2), (R8, R9)
CASPW (R6, R7), (R8), (R4, R5)
VMOVI $82, V0.B16
VMOVI $146, V22.B16
VSSHLL $0, V1.B8, V2.H8
VSSHLL $7, V1.B8, V2.H8
VSSHLL2 $0, V1.B16, V2.H8
VSHRN $7, V1.H8, V0.B8
VSHRN2 $31, V1.D2, V0.S4
VLD2 (R29), [V23.H8, V24.H8]
VLD2.P 16(R0), [V18.B8, V19.B8]
VLD2.P (R1)(R2), [V15.S2, V16.S2]
VLD3 (R27), [V11.S4, V12.S4, V13.S4]
VLD3.P 48(RSP), [V11.S4, V12.S4, V13.S4]
VLD4 (R15), [V10.H4, V11.H4, V12.H4, V13.H4]
VLD4.P 32(R24), [V31.B8, V0.B8, V1.B8, V2.B8]
VLD1R (R1), [V9.B8]
VLD1R.P (R0), [V0.B16]
VLD1R.P 2(R1), [V2.H4]
VLD2R (R15), [V15.H4, V16.H4]
VLD2R.P 16(R0), [V0.D2, V1.D2]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VLD4R.P 16(RSP), [V31.S4, V0.S4, V1.S4, V2.S4]
VST2 [V22.H8, V23.H8], (R23)
VST2.P [V14.H4, V15.H4], 16(R17)
VST2.P [V14.H4, V15.H4], (R3)(R17)
VST3 [V1.D2, V2.D2, V3.D2], (R11)
VST3.P [V18.S4, V19.S4, V20.S4], 48(R25)
VST4 [V22.D2, V23.D2, V24.D2, V25.D2], (R3)
VST4.P [V14.D2, V15.D2, V16.D2, V17.D2], 64(R15)
LDPSW (R0), (R1, R2)
LDPSW 4(R0), (R1, R2)
LDPSW -4(R0), (R1, R2)
PACIASP
DC IVAC, R1
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "families_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "families")
}
// assertARM64Differential assembles the same source with gasm and with the
// toolchain for arm64 and requires the named function's code bytes to agree.
// The live oracle is a deliberate-run comparison, so -short skips it (the
// push pipeline's mode); the golden bytes of the individual encoders are
// pinned separately in every mode.
func assertARM64Differential(t *testing.T, path, src, fn string) {
t.Helper()
oracle := oracleFuncCode(t, toolAsmObject(t, path, "arm64"))
// The oracle keys its functions by the qualified object name
// (pkg.name); match on the local part.
want := map[string][]byte{}
for name, code := range oracle {
if _, after, ok := strings.Cut(name, "."); ok {
want[after] = code
} else {
want[name] = code
}
}
if want[fn] == nil {
t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want))
}
f, perrs := parser.Parse(path, src)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs[0])
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := trimTrailingZeroWords(img.Code)
wantB := trimTrailingZeroWords(want[fn])
if len(got) != len(wantB) {
t.Fatalf("gasm %d bytes, oracle %d bytes", len(got), len(wantB))
}
for i := range wantB {
if got[i] != wantB[i] {
t.Fatalf("word %d differs: gasm %08x, oracle %08x", i/4,
binary.LittleEndian.Uint32(got[i:i+4]), binary.LittleEndian.Uint32(wantB[i:i+4]))
}
}
}
// trimTrailingZeroWords drops whole zero words off the end of a code span:
// an object pads a function to its alignment, and the raw image does not.
// A difference in the middle survives the trim untouched.
func trimTrailingZeroWords(b []byte) []byte {
for len(b) >= 4 {
last := b[len(b)-4:]
if last[0]|last[1]|last[2]|last[3] != 0 {
break
}
b = b[:len(b)-4]
}
return b
}
+252 -19
View File
@@ -8,7 +8,8 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Assemble encodes the body of a TEXT function into x86-64 machine code,
@@ -543,11 +544,27 @@ func pcJumpOffset(op *ast.Operand) (int, bool) {
// pcJumpTarget resolves a numeric jump at statement index j: N counts the
// instruction statements after the jump itself (N = 0 is the jump's own
// address, the classic park loop), and the target is the start of the Nth
// one. It reports false when the count runs past the end of the function.
// one. A negative N counts the same way backwards, before the jump: the
// exit loops write JMP -3(PC) to land three instructions earlier. Labels
// count not, in either direction. It reports false when the count runs
// past the end of the function, or before its first instruction.
func pcJumpTarget(t *ast.Text, j, n int, pcs []int) (int, bool) {
if n == 0 {
return pcs[j], true
}
if n < 0 {
seen := 0
for k := j - 1; k >= 0; k-- {
if _, ok := t.Body[k].(*ast.Instr); !ok {
continue
}
seen--
if seen == n {
return pcs[k], true
}
}
return 0, false
}
seen := 0
for k := j + 1; k < len(t.Body); k++ {
if _, ok := t.Body[k].(*ast.Instr); !ok {
@@ -792,17 +809,43 @@ func isJumpMnemonic(mnem string) bool {
if mnem == "JMP" || mnem == "CALL" {
return true
}
if isLoopMnemonic(mnem) {
return true
}
_, ok := condCode(mnem)
return ok
}
// isLoopMnemonic reports the LOOP family, rel8 alone (E0-E2).
func isLoopMnemonic(mnem string) bool {
switch mnem {
case "LOOP", "LOOPE", "LOOPNE":
return true
}
return false
}
// loopOpcode maps the LOOP family to its E0-E2 opcode.
func loopOpcode(mnem string) byte {
switch mnem {
case "LOOPE":
return 0xE1
case "LOOPNE":
return 0xE0
}
return 0xE2
}
// jumpSize returns the length of a jump instruction in the requested form:
// short (rel8) where available, otherwise the rel32 form. CALL is always
// rel32.
// rel32; the LOOP family is rel8 alone.
func jumpSize(mnem string, long bool) int {
if mnem == "CALL" {
return 5 // opcode + rel32
}
if isLoopMnemonic(mnem) {
return 2 // opcode + rel8, the only form
}
if !long {
return 2 // opcode + rel8
}
@@ -887,6 +930,25 @@ func encodeInstr(s *ast.Instr, pc int, offsets map[string]int, fi frameInfo, lon
func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, []floatPoolEntry, error) {
mnemUpper := strings.ToUpper(s.Mnemonic.Text)
// The extended-instruction layer: a statement whose mnemonic is
// registered in the extension registry and that the scalar paths cannot
// encode goes through the registry, before operandFromAST would reject
// operands only the layer reads (asm/amd64_ext.go). The registry's
// Feature field stays metadata at assembly time: the assembler has no
// CPU, and the toolchain does not gate assembly on CPU features, so
// every registered feature assembles.
if extops, pinned, convErr := amd64ExtStatement(mnemUpper, s.Operands); pinned {
if convErr != nil {
return nil, nil, nil, convErr
}
code, encErr := EncodeExtension(arch.AMD64, mnemUpper, extops...)
if encErr != nil {
return nil, nil, nil, encErr
}
return code, nil, nil, nil
}
if mnemUpper == "FUNCDATA" || mnemUpper == "PCDATA" {
code, err := encodeBookkeeping(mnemUpper, s)
if err != nil {
@@ -929,6 +991,16 @@ func encodeNormal(s *ast.Instr, fi frameInfo, link *linkInfo) ([]byte, []sbPatch
if (mnemUpper == "MOVQ" || mnemUpper == "MOVL") && len(s.Operands) == 2 && isBareTLS(s.Operands[0]) {
return encodeTLSBaseLoad(s, fi, link)
}
// The old paired-register shift spelling, SHLL CX, R11:AX (a colon
// between the two registers), is the toolchain's SHLD family: SHLDL CL,
// AX, R11 with the count register first, the paired source in the reg
// field and the pair's head in r/m.
if code, ps, err := encodeColonShift(s, mnemUpper, fi, link); code != nil || err != nil {
if err != nil {
return nil, nil, nil, err
}
return code, ps, nil, nil
}
_, size := splitSize(mnemUpper)
if size == 0 {
size = 8
@@ -1035,6 +1107,66 @@ func encodeBookkeeping(upper string, s *ast.Instr) ([]byte, error) {
return nil, nil
}
// encodeColonShift encodes the paired-register shift spellings, SHLx CX,
// dst:src: the toolchain reads them as the SHLD family (double-precision
// shift by CL), reg = the paired source, r/m = the pair's head. The second
// operand's raw text carries the colon; ok reports the spelling was found.
func encodeColonShift(s *ast.Instr, mnemUpper string, fi frameInfo, link *linkInfo) ([]byte, []sbPatch, error) {
base, _ := strings.CutPrefix(mnemUpper, "SHL")
if base == mnemUpper || len(s.Operands) != 2 {
return nil, nil, nil
}
_, size := splitSize(mnemUpper)
raw := strings.ReplaceAll(s.Operands[1].Raw, " ", "")
head, tail, ok := strings.Cut(raw, ":")
if !ok || head == "" || tail == "" {
return nil, nil, nil
}
headReg, ok1 := ParseReg(head)
srcReg, ok2 := ParseReg(tail)
if !ok1 || !ok2 {
return nil, nil, fmt.Errorf("%s: invalid paired register %q", mnemUpper, s.Operands[1].Raw)
}
cnt, err := operandFromAST(mnemUpper, s.Operands[0], size, fi, link)
if err != nil {
return nil, nil, err
}
cntReg, ok := cnt.(Reg)
if !ok || cntReg.idx != 1 {
return nil, nil, fmt.Errorf("%s: the paired-register form counts in CL", mnemUpper)
}
// SHLD r/m, reg, CL: 0F A5 (REX.W for the 64-bit width).
e := &enc{}
i := &instr{rexW: size == 8, opcode: []byte{0x0F, 0xA5}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, headReg, size); err != nil {
return nil, nil, err
}
if err := e.emit(i); err != nil {
return nil, nil, err
}
ps := make([]sbPatch, len(e.patches))
for j, p := range e.patches {
ps[j] = sbPatch{off: p.off, name: p.name, addend: p.addend}
}
return e.out, ps, nil
}
// trailingIndexGroup recovers a trailing "(index*scale)" or "(index)" group
// from an operand's raw text: the symbol-pseudo parse returns before the
// index group, so foo(SP)(AX*1) keeps its index only in the spelling.
func trailingIndexGroup(raw string) (string, int, bool) {
compact := strings.ReplaceAll(raw, " ", "")
if !strings.HasSuffix(compact, ")") {
return "", 0, false
}
open := strings.LastIndex(compact, "(")
if open < 2 || !strings.Contains(compact[:open], ")") {
return "", 0, false // one group alone: no trailing index
}
name, scale, _, ok := cutParenGroup(compact[open:])
return name, scale, ok
}
// encodeJump encodes a JMP/CALL/Jcc with a relative offset resolved from the
// target label or from a numeric ±N(PC) instruction count, in the short
// (rel8) or long (rel32) form. numTarget is the resolved byte offset of a
@@ -1069,9 +1201,15 @@ func encodeJump(s *ast.Instr, mnem string, pc int, offsets map[string]int, long
if mnem == "JMP" {
return []byte{0xEB, byte(int8(rel))}, nil
}
if isLoopMnemonic(mnem) {
return []byte{loopOpcode(mnem), byte(int8(rel))}, nil
}
cc, _ := condCode(mnem)
return []byte{0x70 + byte(cc), byte(int8(rel))}, nil
}
if isLoopMnemonic(mnem) {
return nil, fmt.Errorf("%s has no long form", mnem)
}
switch mnem {
case "JMP":
return append([]byte{0xE9}, le32(rel)...), nil
@@ -1123,6 +1261,72 @@ func labelName(op *ast.Operand) (string, bool) {
return "", false
}
// jumpOperand returns the branch-target operand of a JMP/CALL, rewriting the
// `*`-prefixed indirect spellings (JMP *(R12), JMP *4(SP)) into their plain
// memory form. The star marks an indirect target and changes no bytes; the
// address parser leaves the operand's address empty because of the leading
// star, so the fields are rebuilt from the raw text onto a copy of the
// operand, never on the shared syntax tree.
func jumpOperand(s *ast.Instr) *ast.Operand {
if len(s.Operands) != 1 {
return nil
}
op := s.Operands[0]
compact := strings.ReplaceAll(op.Raw, " ", "")
inner, ok := strings.CutPrefix(compact, "*")
if !ok {
return op
}
var addr ast.Address
if i := strings.IndexByte(inner, '('); i > 0 {
v, err := strconv.ParseInt(inner[:i], 0, 64)
if err != nil {
return op
}
addr.Offset, addr.HasOff = v, true
inner = inner[i:]
}
base, _, rest, ok := cutParenGroup(inner)
if !ok {
return op
}
if base != "" {
addr.Base = base
}
if rest != "" {
idx, scale, _, ok := cutParenGroup(rest)
if ok && idx != "" {
addr.Index = idx
addr.Scale = scale
}
}
c := *op
c.Addr = addr
return &c
}
// cutParenGroup splits a leading "(name)" or "(name*n)" off s, returning the
// inner text, the scale it names (1 when the group spells no multiplier) and
// the remainder.
func cutParenGroup(s string) (name string, scale int, rest string, ok bool) {
if !strings.HasPrefix(s, "(") {
return "", 0, "", false
}
i := strings.IndexByte(s, ')')
if i < 0 {
return "", 0, "", false
}
inner, rest := s[1:i], s[i+1:]
if before, after, ok := strings.Cut(inner, "*"); ok {
n, err := strconv.Atoi(after)
if err != nil {
return "", 0, "", false
}
return before, n, rest, true
}
return inner, 1, rest, true
}
// indirectJumpTarget reports whether the JMP/CALL operand addresses a
// register or a memory location rather than a label or a static symbol.
// A bare identifier is a register when the register table knows the name and
@@ -1131,7 +1335,14 @@ func indirectJumpTarget(s *ast.Instr) bool {
if len(s.Operands) != 1 || s.Operands[0].Kind != ast.OpAddr {
return false
}
a := s.Operands[0].Addr
op := jumpOperand(s)
if op == nil {
return false
}
if op != s.Operands[0] {
return true // the star marker spells an indirect target
}
a := op.Addr
// ±N(PC) is the numeric relative form, the PC counts instructions from
// the branch: relative, not indirect.
if a.Base == "PC" || a.Index == "PC" {
@@ -1149,18 +1360,20 @@ func indirectJumpTarget(s *ast.Instr) bool {
}
// encodeIndirectJump assembles a JMP/CALL through a register or memory
// operand, which carries no relocation and no label to resolve.
// operand, which carries no relocation and no label to resolve. The
// `*`-prefixed spellings go through jumpOperand first, their star rebuilt
// into a plain memory operand.
func encodeIndirectJump(s *ast.Instr, mnem string) ([]byte, error) {
ops := make([]Operand, len(s.Operands))
for i, op := range s.Operands {
op := s.Operands[0]
if cleaned := jumpOperand(s); cleaned != nil {
op = cleaned
}
o, err := operandFromAST(mnem, op, 8, frameInfo{}, nil)
if err != nil {
return nil, err
}
ops[i] = o
}
e := &enc{}
if err := e.encodeIndirectBranch(mnem, ops); err != nil {
if err := e.encodeIndirectBranch(mnem, []Operand{o}); err != nil {
return nil, err
}
return e.out, nil
@@ -1229,31 +1442,51 @@ func operandFromAST(mnemUpper string, op *ast.Operand, size int, fi frameInfo, l
off := a.Sym.Offset + fi.fpAdjust
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
}
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP).
// SP-relative local: x-N(SP) → (spAdjust + offset)(SP), keeping a scaled
// index beside the virtual stack pointer (foo(SP)(AX*1)). The
// symbol-pseudo parse returns before the index group, so the index
// is recovered from the raw text when the address lacks it.
if a.Sym != nil && a.Sym.Pseudo == "SP" && a.Base == "" {
off := fi.spAdjust + a.Sym.Offset
return Mem{Base: spReg, Disp: off, HasBase: true, Size: size}, nil
m := Mem{Base: spReg, Disp: off, HasBase: true, Size: size}
if name, scale, ok := trailingIndexGroup(op.Raw); ok {
idx, ok := ParseReg(name)
if !ok {
return nil, fmt.Errorf("unknown index register %q", name)
}
m.Index = idx
m.Scale = scale
m.HasIndex = true
}
return m, nil
}
// SB (global symbol): a symbol defined in the same file (GLOBL) is
// encoded RIP-relative and resolved by the file-level layout;
// anything not defined here needs object-file emission.
// anything not defined here needs object-file emission. A static
// (file-local) spelling of an undefined symbol defers the same way
// the toolchain does: the relocation names it and the linker decides.
if a.Sym != nil && a.Sym.Pseudo == "SB" {
if link == nil || link.symbols == nil {
return nil, fmt.Errorf("symbol %q needs file-level assembly (AssembleFile)", a.Sym.Name)
}
if !link.symbols[a.Sym.Name] {
if a.Sym.Static {
return nil, fmt.Errorf("undefined symbol %q", a.Sym.Name)
}
if !link.allowExternal {
if !link.symbols[a.Sym.Name] && !link.allowExternal {
return nil, fmt.Errorf("external symbol %q needs object-file emission", a.Sym.Name)
}
}
return sbMem{size: size, name: a.Sym.Name, addend: a.Sym.Offset}, nil
}
// Memory with a real base register: (base), off(base), (base)(index*scale).
if a.Base != "" {
// The TLS pseudo-base, off(TLS): the segment-prefixed absolute
// the thread-local access lowers to, 64 8B 04 25 with its
// R_TLS_LE patch site on the disp32.
if a.Base == "TLS" {
seg := byte(0x64) // FS on linux, freebsd, plan9
if link != nil && link.goos == "windows" {
seg = 0x65 // GS
}
return TLSMem{Disp: a.Offset, Size: size, Seg: seg}, nil
}
// Segment-absolute: 0x30(GS) and 0x28(FS), the windows TLS
// spellings. The segment override prefixes a disp32 absolute
// reference with no relocation.
+54 -2
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstText parses src and returns its first TEXT function.
@@ -370,6 +370,58 @@ end:
}
}
// TestAssembleNumericPCJumps pins the numeric ±N(PC) branch operands: N
// counts instruction statements, skipping labels, in both directions (the
// runtime's exit loops write JMP -3(PC)), N = 0 parks on the jump itself.
func TestAssembleNumericPCJumps(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
TEXT ·exit(SB), NOSPLIT, $0
MOVB $1, AL
lab:
MOVB $2, AL
MOVB $3, AL
JMP -3(PC)
MOVB $4, AL
park:
JMP 0(PC)
MOVB $5, AL
JMP 2(PC)
MOVB $6, AL
RET
`)
code, _, err := Assemble(fn)
if err != nil {
t.Fatalf("Assemble: %v", err)
}
// From the Go-assembled function:
// MOVB $1, AL b001
// MOVB $2, AL b002
// MOVB $3, AL b003
// JMP -3(PC) ebf8 (three instructions back, past lab:)
// MOVB $4, AL b004
// JMP 0(PC) ebfe (the park loop)
// MOVB $5, AL b005
// JMP 2(PC) eb02 (over MOVB $6 to the RET)
// MOVB $6, AL b006
// RET c3
want := []byte{
0xb0, 0x01,
0xb0, 0x02,
0xb0, 0x03,
0xeb, 0xf8,
0xb0, 0x04,
0xeb, 0xfe,
0xb0, 0x05,
0xeb, 0x02,
0xb0, 0x06,
0xc3,
}
if hexBytes(code) != hexBytes(want) {
t.Errorf("numeric-PC mismatch:\n got: %s\n want: %s", hexBytes(code), hexBytes(want))
}
}
func TestAssemblePrefetch(t *testing.T) {
fn := firstText(t, `
#include "textflag.h"
+288
View File
@@ -0,0 +1,288 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// fuzzIncludeDirs points the expansion path at the package's testdata include
// directory, so a seed's #include resolves the way the CLI's -I list does.
var fuzzIncludeDirs = []string{filepath.Join("testdata", "include")}
// corpusSeeds seeds every fuzz target with the repository's kernels, so a
// plain `go test` run replays each seed as a regression case and CI exercises
// them without any fuzzing budget. Kernels of a foreign architecture
// exercise the rejection path (the fixed target reports them as diagnostics);
// kernels of the target's own architecture reach its encoder.
func corpusSeeds(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
}
// archCorpusSeeds seeds a target with every assembly file of its architecture
// seed directory, testdata/seeds/<dir>. The files hold the target-specific
// corpus: the instruction families GOROOT's own assembler corpus exercises for
// the architecture, minimalised, plus the boundary shapes the encoder's range
// gates live on. They are committed assembly, so `gasm fmt` and `gasm lint`
// gate them the way they gate every other .s file. A plain `go test` run
// replays each one as a regression case.
func archCorpusSeeds(f *testing.F, dir string) {
files, _ := filepath.Glob(filepath.Join("testdata", "seeds", dir, "*.s"))
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
// fuzzAssemble is the whole fuzz body, shared by every target and one line
// apart between them: parse with macro and include expansion, assemble
// through the target's file-level entry, and hold the invariants. The
// contract:
//
// - no panic, however malformed the source (a crash fails the target);
// - a rejected file yields a diagnostic and never a partial emission:
// the assembler returns a nil image beside its error, and the
// diagnostic is not empty;
// - the output is deterministic: the same source, parsed and assembled
// again from scratch, produces the same bytes;
// - no unbounded memory: the fence around every test run kills a run that
// amplifies its input, and the timebox turns a hang into a campaign
// failure to bisect.
//
// A file the parser rejects still reaches the assembler: the parser is
// line-oriented and tolerant, so it hands back a usable file either way, and
// the assembler's own contract is to answer any file it is given with bytes
// or with a diagnostic, never with a panic.
func fuzzAssemble(t *testing.T, name string, src string, assemble func(*ast.File) (*Image, error)) {
file, _ := parser.ParseWithOptions(name, src,
parser.Options{Expand: true, IncludeDirs: fuzzIncludeDirs})
if file == nil {
t.Fatal("ParseWithOptions returned a nil file")
}
img, err := assemble(file)
if err != nil {
if img != nil {
t.Fatal("the assembler returned an image beside its error: a rejected file must not emit")
}
if strings.TrimSpace(err.Error()) == "" {
t.Fatal("rejection carries an empty diagnostic")
}
return
}
// Determinism: a second assembly of the same file must produce the same
// bytes. One parse serves both runs, so any mutation the assembler makes
// to the syntax tree it was handed shows up as differing bytes, and the
// workers' footprint under the shared memory fence stays that of a single
// parse.
img2, err2 := assemble(file)
if err2 != nil {
t.Fatalf("the second assembly failed where the first succeeded: %v", err2)
}
if !bytes.Equal(img.Bytes(), img2.Bytes()) {
t.Fatal("the same source assembled to different bytes")
}
}
// FuzzAssembleAMD64 hammers the full parse-and-assemble pipeline for the
// fixed amd64 target with arbitrary source: expansion (macros and includes)
// included, matching the CLI's own pipeline.
func FuzzAssembleAMD64(f *testing.F) {
corpusSeeds(f)
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("TEXT ·f(SB), $16-8\n\tMOVQ x+0(FP), AX\n\tMOVQ AX, ret+8(FP)\n\tRET\n")
f.Add("TEXT ·f(SB), $256-0\n\tCALL ·helper(SB)\n\tRET\nTEXT ·helper(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#define L(n) MOVQ $n, AX\nTEXT ·f(SB), NOSPLIT, $0\n\tL(7)\n\tRET\n")
f.Add("#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#include \"fuzzdefs.h\"\nTEXT ·f(SB), $16-8\n\tMOVQ KONST, AX\n\tMOVQ ARG(x), BX\n\tRET\n")
f.Add("DATA d<>+0(SB)/8, $0xf4f8fcff\nDATA d<>+4(SB)/4, $1\nGLOBL d<>(SB), RODATA, $8\n" +
"TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ d<>(SB), AX\n\tRET\n")
f.Add("DATA s+0(SB)/8, $\"hi there\"\nGLOBL s(SB), $8\nDATA p+0(SB)/8, $s(SB)\nGLOBL p(SB), $8\n")
f.Add("DATA d+0(SB)/8, $0xFFFFFFFFFFFFFFFF\nGLOBL d(SB), $8\n" +
"TEXT ·f(SB), $0-8\n\tMOVQ $0xFFFFFFFFFFFFFFFF, AX\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\nL1:\n\tMOVQ AX, BX\n\tJMP L1\n\tJMP -3(PC)\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tLOOP L1\nL1:\n\tLOOPE L1\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tJMP *AX\n\tCALL (BX)\n\tJMP (R12)(R8*4)\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ TLS, AX\n\tMOVQ 8(AX)(TLS*1), BX\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tPCALIGN $16\n\tMOVQ AX, BX\n\tPCALIGN $32\n\tMOVQ AX, BX\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADJSP $16\n\tMOVQ AX, -8(SP)\n\tADJSP $-16\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADDSD $1.5, X0\n\tMULSD $(-1.0), X1\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tSHLL CX, R11:AX\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tPCDATA $0, $1\n\tFUNCDATA $0, ·meta(SB)\n\tRET\n")
f.Add("TEXT ·f(SB), $0\n\tCALL runtime·morestack_noctxt(SB)\n\tRET\n")
f.Add("TEXT ·f(SB), $32-0\n\tMOVQ AX, x-8(SP)\n\tMOVQ BX, x-16(SP)(CX*1)\n\tRET\n")
// Shapes that must be rejected: each pins a diagnostic path the seeds
// above never reach.
f.Add("TEXT ·f(SB), $0\n\tBOGUSINSTR AX, BX\n\tRET\n")
f.Add("GLOBL d(SB), $-8\n")
f.Add("GLOBL d(SB), $-1\n")
f.Add("GLOBL d(SB), $0x4000001\n")
f.Add("GLOBL d(SB), $0x7FFFFFFFFFFFFFFF\n")
f.Add("DATA d+0(SB)/9, $1\nGLOBL d(SB), $8\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADJSP $16\n\tRET\n")
f.Add("#define A A\nA\n")
f.Fuzz(func(t *testing.T, src string) {
fuzzAssemble(t, "fuzz_amd64.s", src, func(f *ast.File) (*Image, error) {
return AssembleFile(f)
})
})
}
// FuzzAssembleARM64 hammers the same pipeline for the fixed arm64 target,
// whose encoder carries its own immediate classification, memory-offset
// gates and literal pool. The seeds pin the recent encoder families: the
// system registers and barriers, the LSE atomics and exclusive pairs, the
// NEON structure loads and stores, the ADDCON2 offset split, the pooled
// vector constants, and the macro and include expansion over arm64
// spellings. The rejection shapes pin the diagnostic paths the accepted
// seeds never reach.
func FuzzAssembleARM64(f *testing.F) {
corpusSeeds(f)
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("TEXT ·f(SB), $16-8\n\tMOVW x+0(FP), R0\n\tMOVW R0, ret+8(FP)\n\tRET\n")
f.Add("TEXT ·f(SB), $256-0\n\tCALL ·helper(SB)\n\tRET\nTEXT ·helper(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#define L(n) MOVD $n, R0\nTEXT ·f(SB), NOSPLIT, $0\n\tL(7)\n\tRET\n")
f.Add("#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#include \"fuzzdefs.h\"\nTEXT ·f(SB), $16-8\n\tMOVD KONST, R0\n\tMOVD ARG(x), R1\n\tRET\n")
f.Add("DATA d<>+0(SB)/8, $0xf4f8fcff\nDATA d<>+4(SB)/4, $1\nGLOBL d<>(SB), RODATA, $8\n" +
"TEXT ·f(SB), NOSPLIT, $0\n\tMOVD d<>(SB), R0\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMRS DCZID_EL0, R3\n\tMRS CNTVCT_EL0, R0\n\tMSR $3, SPSel\n" +
"\tMSR $9, DAIFSet\n\tDMB $15\n\tDSB $4\n\tISB $1\n\tDC ZVA, R4\n\tSVC $0\n\tBRK $35943\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tLDADDB R2, (R1), R3\n\tLDADDD R2, (R1), ZR\n\tCASW R2, (R1), R3\n" +
"\tSWPD R2, (R1), ZR\n\tLDXR (R1), R2\n\tLDAXRW (R1), R5\n\tSTXR R2, (R1), R6\n\tSTLXRB R3, (R1), R6\n" +
"\tLDXP (R1), (R2, R3)\n\tSTXP (R2, R3), (R1), R6\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVLD1 (R2), [V21.B16]\n\tVLD1.P 32(R1), [V2.B16, V3.B16]\n" +
"\tVLD1R (R0), [V0.B16]\n\tVST1 [V2.S4, V3.S4], (R14)\n\tVST1.P [V2.B16], (R1)\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADD $0xaaaaaa, R2, R3\n\tSUB $0x186a0, R2, R3\n\tADDW $0x60060, R2\n\tCMP $40960, R0\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVMOVD $0x123456789ABCDEF0, V0\n\tVMOVQ $0x12345678, $0x9ABCDEF0, V1\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0-8\n\tMOVW $-1, R0\n\tB after1\n\tMOVD $0x0001000200030004, R1\n" +
"after1:\n\tMOVD R1, ret+0(FP)\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\nL1:\n\tCBZ R1, L1\n\tTBZ $3, R2, L1\n\tBEQ L1\n\tJMP L1\n\tCALL (R5)\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tPCALIGN $16\n\tMOVD R0, R1\n\tPCALIGN $32\n\tRET\n")
f.Add("TEXT ·f(SB), $0\n\tPCDATA $0, $1\n\tFUNCDATA $0, ·meta(SB)\n\tRET\n")
f.Add("TEXT ·f(SB), $32-0\n\tMOVD R0, x-8(SP)\n\tRET\n")
// Shapes that must be rejected: each pins a diagnostic path the seeds
// above never reach.
f.Add("TEXT ·f(SB), $0\n\tBOGUSINSTR R0, R1\n\tRET\n")
f.Add("GLOBL d(SB), $-8\n")
f.Add("GLOBL d(SB), $0x4000001\n")
f.Add("DATA d+0(SB)/9, $1\nGLOBL d(SB), $8\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADD R1, X99\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVD $1, R1\n\tMOVD 0x1000000(R1), R2\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVLD1 (R2), [V21.B17]\n\tRET\n")
f.Add("#define A A\nA\n")
f.Fuzz(func(t *testing.T, src string) {
fuzzAssemble(t, "fuzz_arm64.s", src, AssembleFileARM64)
})
}
// FuzzAssembleRISCV64 hammers the same pipeline for the fixed riscv64 target.
// The seed corpus lives in testdata/seeds/riscv64: the instruction families
// GOROOT's riscv64 assembler corpus exercises (the immediate-range ladder of
// the I-type arithmetic, the load/store and branch offsets, the atomics, the
// FP conversions and the fused multiply-adds), the RVV configuration and
// arithmetic classes with their mask forms, the RVC-compressible shapes, the
// CSR instructions and the Zbb/Zba/Zbs bit-manipulation set, minimalised.
// The inline seeds below pin the shared file-level surface and the rejection
// shapes, one diagnostic path each.
func FuzzAssembleRISCV64(f *testing.F) {
corpusSeeds(f)
archCorpusSeeds(f, "riscv64")
// Minimal seeds for the shared file-level surface, spelled the riscv64
// way: the guard classes, the macro and include expansion, and the data
// section with its symbol-valued fields.
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("TEXT ·f(SB), $16-8\n\tMOV x+0(FP), X10\n\tMOV X10, ret+8(FP)\n\tRET\n")
f.Add("TEXT ·f(SB), $256-0\n\tCALL ·helper(SB)\n\tRET\nTEXT ·helper(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#define L(n) ADDI $n, X10, X10\nTEXT ·f(SB), NOSPLIT, $0\n\tL(7)\n\tRET\n")
f.Add("#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#include \"fuzzdefs.h\"\nTEXT ·f(SB), $16-8\n\tMOV KONST, X10\n\tMOV ARG(x), X11\n\tRET\n")
f.Add("DATA d<>+0(SB)/8, $0xf4f8fcff\nDATA d<>+4(SB)/4, $1\nGLOBL d<>(SB), RODATA, $8\n" +
"TEXT ·f(SB), NOSPLIT, $0\n\tMOV d<>(SB), X10\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tPCALIGN $16\n\tADD X11, X10, X10\n\tPCALIGN $2048\n\tRET\n")
f.Add("TEXT ·f(SB), $0\n\tPCDATA $0, $1\n\tFUNCDATA $0, ·meta(SB)\n\tRET\n")
// Shapes that must be rejected: each pins a diagnostic path the accepted
// seeds never reach.
f.Add("TEXT ·f(SB), $0\n\tBOGUSINSTR X10, X11\n\tRET\n")
f.Add("GLOBL d(SB), $-8\n")
f.Add("GLOBL d(SB), $0x4000001\n")
f.Add("DATA d+0(SB)/9, $1\nGLOBL d(SB), $8\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADD X11, X32\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADDI $2048, X5, X6\n\tADD X11, X40, X6\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tCSRRW $0x1000, X5, X6\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tSLLI $64, X5, X6\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVLE8V (X10), V32\n\tRET\n")
// Operand-starved spellings that used to panic the layout and encode
// passes; each must come back as a diagnostic.
f.Add("TEXT ·f(SB), $0\n\tJALR\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tROR $3\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVLE8V (X10)\n\tRET\n")
f.Add("#define A A\nA\n")
f.Fuzz(func(t *testing.T, src string) {
fuzzAssemble(t, "fuzz_riscv64.s", src, AssembleFileRISCV)
})
}
// FuzzAssembleLOONG64 hammers the same pipeline for the fixed loong64 target.
// The seed corpus lives in testdata/seeds/loong64: the LSX and LASX register
// banks with their immediate forms and range gates (the si5 compares, the
// biased shifts, the VSHUF4I/VPERMI/VEXTRINS immediates), the ll/sc offset
// ladder with its three encoding spans, the pointer loads and stores, the
// atomics with their dbar forms, the branches, the bit-field instructions and
// the register-class moves, minimalised. The inline seeds below pin the
// shared file-level surface and the rejection shapes, one diagnostic path
// each.
func FuzzAssembleLOONG64(f *testing.F) {
corpusSeeds(f)
archCorpusSeeds(f, "loong64")
// Minimal seeds for the shared file-level surface, spelled the loong64
// way.
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("TEXT ·f(SB), $16-8\n\tMOVV x+0(FP), R4\n\tMOVV R4, ret+8(FP)\n\tRET\n")
f.Add("TEXT ·f(SB), $256-0\n\tCALL ·helper(SB)\n\tRET\nTEXT ·helper(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#define L(n) ADDV $n, R4, R4\nTEXT ·f(SB), NOSPLIT, $0\n\tL(7)\n\tRET\n")
f.Add("#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("#include \"fuzzdefs.h\"\nTEXT ·f(SB), $16-8\n\tMOVV KONST, R4\n\tMOVV ARG(x), R5\n\tRET\n")
f.Add("DATA d<>+0(SB)/8, $0xf4f8fcff\nDATA d<>+4(SB)/4, $1\nGLOBL d<>(SB), RODATA, $8\n" +
"TEXT ·f(SB), NOSPLIT, $0\n\tMOVV d<>(SB), R4\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tPCALIGN $16\n\tADDV R5, R4, R4\n\tPCALIGN $2048\n\tRET\n")
f.Add("TEXT ·f(SB), $0\n\tPCDATA $0, $1\n\tFUNCDATA $0, ·meta(SB)\n\tRET\n")
// Shapes that must be rejected: each pins a diagnostic path the accepted
// seeds never reach.
f.Add("TEXT ·f(SB), $0\n\tBOGUSINSTR R4, R5\n\tRET\n")
f.Add("GLOBL d(SB), $-8\n")
f.Add("GLOBL d(SB), $0x4000001\n")
f.Add("DATA d+0(SB)/9, $1\nGLOBL d(SB), $8\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tADD R5, R32\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tBEQZ V0, L1\nL1:\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVADDV V1, V2, X3\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVSEQV $32, V2, V3\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tVSHUF4IV $16, V2, V1\n\tRET\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tBSTRPICKV $64, R4, $5, R6\n\tRET\n")
f.Add("#define A A\nA\n")
f.Fuzz(func(t *testing.T, src string) {
fuzzAssemble(t, "fuzz_loong64.s", src, AssembleFileLOONG64)
})
}
+1 -1
View File
@@ -8,7 +8,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
+2 -2
View File
@@ -12,8 +12,8 @@ import (
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The object-file tests share one source: two exported functions, one
+8
View File
@@ -18,6 +18,9 @@ const (
rArm64AddAbsLo12NC = 277 // R_AARCH64_ADD_ABS_LO12_NC (ADD page offset)
rArm64Call26 = 283 // R_AARCH64_CALL26 (BL instruction)
rArm64Ldst64Lo12NC = 286 // R_AARCH64_LDST64_ABS_LO12_NC (64-bit LDR/STR page offset)
// R_AARCH64_TLSLE_MOVW_TPREL_G0 (debug/elf 547): the local-exec TLS
// load's MOVZ field, the module offset at bits [15:0].
rArm64TLSLEMovwTprelG0 = 547
// R_AARCH64_ABS32 (debug/elf 258): the absolute 32-bit address of a
// symbol, the R_ADDR shape a 4-byte DATA field carries. ABS64 (257)
// lives with the DWARF fixup constants as rAARCH64Abs64.
@@ -131,6 +134,11 @@ func (img *Image) ELFAARCH64Object() ([]byte, error) {
elfRela{off: uint64(fn.Offset + r.Off), typ: rArm64PrelPgHi21, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rArm64Ldst64Lo12NC, sym: idx, addend: r.Addend},
)
case RelArm64TLSLE:
// The local-exec MOVZ: one relocation over the imm16 field.
relas = append(relas, elfRela{
off: uint64(fn.Offset + r.Off), typ: rArm64TLSLEMovwTprelG0, sym: idx, addend: r.Addend,
})
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFAARCH64Object checks the structure of the emitted AArch64 ELF64
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64
+11
View File
@@ -24,6 +24,8 @@ const (
rRISCVPCRELHI20 = 23 // R_RISCV_PCREL_HI20
rRISCVPCRELLO12I = 24 // R_RISCV_PCREL_LO12_I
rRISCVPCRELLO12S = 25 // R_RISCV_PCREL_LO12_S
rRISCVTPRELHI20 = 29 // R_RISCV_TPREL_HI20
rRISCVTPRELLO12I = 30 // R_RISCV_TPREL_LO12_I
// R_RISCV_32 (debug/elf 1): the absolute 32-bit address of a symbol,
// the R_ADDR shape a 4-byte DATA field carries. R_RISCV_64 (2) lives
// with the DWARF fixup constants as rRISCVAbs64.
@@ -128,6 +130,15 @@ func (img *Image) ELFRISCVObject() ([]byte, error) {
)
case RelRISCVJal:
relas = append(relas, elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVJAL, sym: idx, addend: r.Addend})
case RelRISCVTLSLE:
// The local-exec pair splits into the TPREL HI20 on the LUI
// and the TPREL LO12_I on the ADDIW, both against the symbol
// (a thread offset, not PC-relative, so the LO12 needs no
// label indirection).
relas = append(relas,
elfRela{off: uint64(fn.Offset + r.Off), typ: rRISCVTPRELHI20, sym: idx, addend: r.Addend},
elfRela{off: uint64(fn.Offset + r.Off + 4), typ: rRISCVTPRELLO12I, sym: idx, addend: r.Addend},
)
default:
return nil, fmt.Errorf("relocation kind %v unsupported in ELF emission", r.Kind)
}
+1 -1
View File
@@ -9,7 +9,7 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestELFRISCVObjectDataRelocation checks that a symbol-valued DATA field
+27 -9
View File
@@ -16,15 +16,35 @@ import "strings"
func Encodable(mnemonic string) bool {
upper := strings.ToUpper(mnemonic)
// The corpus families first, mirroring encode()'s dispatch order: a
// family owning the name decides encodability whatever the suffix split
// would make of it.
if amd64FamilyEncodable(upper) {
return true
}
// Fixed-name instructions (no size suffix).
switch upper {
case "RET", "NOP", "CALL", "JMP",
"POPFQ", "PUSHFQ", "INT", "LDMXCSR", "STMXCSR", "CMPSD", "SHA256RNDS2",
// The SSE compare family sharing CMPSD's predicate-last shape, the
// far return with its stack pop, the loop family, the bank-crossing
// MMX moves and the one-operand system controls.
"CMPSS", "CMPPS", "CMPPD", "RETFL",
"LOOP", "LOOPE", "LOOPNE",
"MOVDQ2Q", "MOVQ2DQ",
"ENDBR64", "CLWB", "TPAUSE", "UMONITOR", "UMWAIT", "RDPID", "CLDEMOTE",
// The literal-data pseudo-ops, the accepted-and-ignored END and
// bookkeeping statements, and the SP adjust.
"BYTE", "WORD", "LONG", "QUAD", "END", "ADJSP", "FUNCDATA", "PCDATA":
return true
}
if _, ok := sysUnaryTable[upper]; ok {
return true
}
if _, ok := sseStoreOnly[upper]; ok {
return true
}
if _, ok := noOperandTable[upper]; ok {
return true
}
@@ -42,18 +62,16 @@ func Encodable(mnemonic string) bool {
return true
}
// CMOV carries size then condition (CMOVLGT); SET carries the condition
// alone (SETNE). The size letter is checked exactly as encodeCmov does,
// so a spelling like CMOVBGT is not reported encodable when Encode
// would reject it.
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok && len(rest) >= 2 {
switch rest[0] {
case 'W', 'L', 'Q':
if _, ok := jccMap[rest[1:]]; ok {
// CMOV carries size then condition (CMOVLGT), or the renderer's
// condition alone (CMOVLE) with the width from the operand; SET carries
// the condition alone (SETNE). cmovCondition checks the suffix exactly
// as encodeCmov does, so a spelling like CMOVBGT is not reported
// encodable when Encode would reject it.
if rest, ok := strings.CutPrefix(upper, "CMOV"); ok {
if _, _, ok := cmovCondition(rest); ok {
return true
}
}
}
if rest, ok := strings.CutPrefix(upper, "SET"); ok {
if _, ok := jccMap[rest]; ok {
return true
+107 -4
View File
@@ -70,12 +70,33 @@ type encPatch struct {
func (e *enc) encode(mnem string, ops []Operand) error {
upper := strings.ToUpper(mnem)
// The corpus families (x87, the system and string controls, XSAVE)
// dispatch on the full name from the amd64 family files before the
// fixed-name switch, the way their tables spell the mnemonics out.
if handled, err := e.encodeAmd64Family(upper, ops); handled {
return err
}
// Fixed-name instructions (no size suffix).
switch {
case upper == "RET":
// RET sym(SB), the absolute return: the toolchain encodes it as a
// tail jump, E9 rel32 with a call relocation against the symbol.
if len(ops) == 1 {
if m, ok := ops[0].(sbMem); ok {
return e.emit(&instr{opcode: []byte{0xE9}, modrm: -1, sib: -1, disp: le32(0), sb: &sbRef{name: m.name, addend: m.addend}})
}
return fmt.Errorf("RET: unsupported operand")
}
if len(ops) != 0 {
return fmt.Errorf("RET expects no operands, got %d", len(ops))
}
return e.encodeRet()
case upper == "NOP":
return e.emit(&instr{opcode: []byte{0x90}, modrm: -1, sib: -1})
// The toolchain consumes every NOP statement as a pseudo and emits
// nothing for it, operands included (a bare NOP, NOP AX and
// NOP sym(SB) all vanish from the object).
return nil
case upper == "CALL" || upper == "JMP":
// Through a register or memory: FF /2 (CALL) or FF /4 (JMP).
// Anything else is a rel32 against a label resolved by the assembler.
@@ -102,6 +123,15 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
return e.emit(&instr{opcode: op, modrm: -1, sib: -1})
}
// One-operand system instructions whose reg field is a fixed digit:
// the cache and wait controls under 0F AE/0F 1C and the RDPID read.
if m, ok := sysUnaryTable[upper]; ok {
return e.encodeSysUnary(upper, m, ops)
}
// The store-only SSE moves (the non-temporal store).
if m, ok := sseStoreOnly[upper]; ok {
return e.encodeSSEStoreOnly(upper, m, ops)
}
// POPFQ/PUSHFQ are exact names: the bare POPF/PUSHF and the L spellings
// are rejected by go tool asm in 64-bit mode, so they stay unsupported.
switch upper {
@@ -117,14 +147,70 @@ func (e *enc) encode(mnem string, ops []Operand) error {
return e.emit(&instr{opcode: []byte{0x9C}, modrm: -1, sib: -1})
case "INT":
return e.encodeInt(ops)
// The LOOP family outside the assembler's label settlement: the operand
// is the already-computed rel8 (E0-E2).
case "LOOP", "LOOPE", "LOOPNE":
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", upper, len(ops))
}
imm, ok := ops[0].(Imm)
if !ok || !fits8(int64(imm)) {
return fmt.Errorf("%s: relative offset must be a signed byte", upper)
}
return e.emit(&instr{opcode: []byte{loopOpcode(upper)}, modrm: -1, sib: -1, imm: []byte{byte(int8(imm))}})
case "LDMXCSR":
return e.encodeMxcsr(2, ops)
case "STMXCSR":
return e.encodeMxcsr(3, ops)
// CMPSD is the scalar double compare, whose predicate immediate comes
// LAST in Plan 9 order (src, dst, $imm).
// LAST in Plan 9 order (src, dst, $imm); the family shares the shape.
case "CMPSD":
return e.encodeCmpsd(ops)
return e.encodeSSECmp("CMPSD", 0xF2, ops)
case "CMPSS":
return e.encodeSSECmp("CMPSS", 0xF3, ops)
case "CMPPS":
return e.encodeSSECmp("CMPPS", 0x00, ops)
case "CMPPD":
return e.encodeSSECmp("CMPPD", 0x66, ops)
// RETFL pops the immediate's worth of bytes after the far return
// (LRET iw: CA imm16), the toolchain's RETF spelling with a stack
// adjustment.
case "RETFL":
if len(ops) != 1 {
return fmt.Errorf("RETFL expects 1 operand, got %d", len(ops))
}
imm, ok := ops[0].(Imm)
if !ok {
return fmt.Errorf("RETFL expects an immediate")
}
return e.emit(&instr{opcode: []byte{0xCA}, modrm: -1, sib: -1, imm: le16(int64(imm))})
// MOVDQ2Q/MOVQ2DQ cross the MMX and XMM banks, and each direction
// carries its own mandatory prefix: the toolchain renders F3 0F D6 as
// MOVQ2DQ (the MMX source, XMM destination) and F2 0F D6 as MOVDQ2Q
// (the XMM source, MMX destination), the register in the reg field, the
// other bank's in r/m.
case "MOVDQ2Q", "MOVQ2DQ":
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", upper, len(ops))
}
srcReg, ok1 := ops[0].(Reg)
dstReg, ok2 := ops[1].(Reg)
if !ok1 || !ok2 {
return fmt.Errorf("%s takes register operands alone", upper)
}
if upper == "MOVDQ2Q" && (!srcReg.isVec() || !dstReg.mmx) ||
upper == "MOVQ2DQ" && (!srcReg.mmx || !dstReg.isVec()) {
return fmt.Errorf("%s crosses the XMM and MMX banks in that order", upper)
}
prefix := byte(0xF2)
if upper == "MOVQ2DQ" {
prefix = 0xF3
}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, srcReg, 8); err != nil {
return err
}
return e.emit(i)
// SHA256RNDS2 carries the round constant in a literal X0 first operand.
case "SHA256RNDS2":
return e.encodeSha256rnds2(ops)
@@ -171,6 +257,19 @@ func (e *enc) encode(mnem string, ops []Operand) error {
}
base, size := splitSize(upper)
// The scalar families that own byte forms settle the width against the
// register operands here, before the unsuffixed 64-bit default applies,
// so a literal Q suffix stays distinguishable from no suffix at all
// (CRC32 DL, R11 is the byte form; CRC32Q DL, R11 is a conflict).
if byteFormBase[base] {
w, err := operandWidth(mnem, size, widthOperands(base, ops))
if err != nil {
return err
}
if w != 0 {
size = w
}
}
if size == 0 {
size = 8 // default operand size in 64-bit mode (e.g. PUSHQ)
}
@@ -220,8 +319,12 @@ func (e *enc) encode(mnem string, ops []Operand) error {
case "MOV":
return e.encodeMov(ops, size)
// MOVD is the Go assembler's alias of MOVQ: the same byte forms, 64-bit
// REX.W and all.
// REX.W and all. The alias takes no byte register either, the same
// conflict rule the Q-suffixed spelling answers to.
case "MOVD":
if _, err := operandWidth(mnem, 8, ops); err != nil {
return err
}
return e.encodeMov(ops, 8)
case "ADD", "SUB", "AND", "OR", "XOR", "CMP", "ADC", "SBB":
return e.encodeALU(aluOp[base], ops, size)
+172 -7
View File
@@ -10,8 +10,8 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// decode encodes an instruction and decodes it back, returning the decoded
@@ -265,7 +265,11 @@ func TestImul(t *testing.T) {
func TestControl(t *testing.T) {
checkSyntax(t, "ret", "RET")
checkSyntax(t, "nop", "NOP")
// NOP contributes nothing on amd64, consumed whole by the toolchain as
// a pseudo; only the bytes pin it (no decodable instruction remains).
if code, err := Encode("NOP"); err != nil || len(code) != 0 {
t.Errorf("NOP: bytes %x (err %v), want empty", code, err)
}
checkOp(t, x86asm.JMP, "JMP", Imm(0))
checkOp(t, x86asm.CALL, "CALL", Imm(0))
checkOp(t, x86asm.JGE, "JGE", Imm(0))
@@ -407,6 +411,16 @@ func TestScalarGroundTruth(t *testing.T) {
{"CMOVLEQ CX,AX", "CMOVLEQ", []Operand{CX, AX}, "0f44c1", "CMOVE"},
{"CMOVQGT R9,R8", "CMOVQGT", []Operand{r9, r8}, "4d0f4fc1", "CMOVG"},
{"CMOVWLS R9W,R8W", "CMOVWLS", []Operand{r9w, r8w}, "66450f46c1", "CMOVBE"},
// The renderer's condition spellings carry no size letter; the width
// rides the operand registers and the bytes match the toolchain's own
// size-prefixed spellings (CMOVQLE/CMOVLLE/CMOVWLE pinned from go tool
// asm). CMOVLE with the 16-bit registers reproduces the disasm
// fixture's 660f4e13 row byte for byte.
{"CMOVLE (BX),DX", "CMOVLE", []Operand{Ptr(BX, 0, 2), DX}, "660f4e13", "CMOVLE"},
{"CMOVQLE AX,BX", "CMOVQLE", []Operand{Reg{idx: 0, size: 8}, Reg{idx: 3, size: 8}}, "480f4ed8", "CMOVLE"},
{"CMOVLLE AX,BX", "CMOVLLE", []Operand{Reg{idx: 0, size: 4}, Reg{idx: 3, size: 4}}, "0f4ed8", "CMOVLE"},
{"CMOVWLE AX,BX", "CMOVWLE", []Operand{AX, BX}, "660f4ed8", "CMOVLE"},
{"CMOVB AL,CL", "CMOVB", []Operand{AL, CL}, "0f42c8", "CMOVB"},
{"SETNE AL", "SETNE", []Operand{AL}, "0f95c0", "SETNE"},
{"SETNE (AX)", "SETNE", []Operand{Ptr(AX, 0, 1)}, "0f9500", "SETNE"},
{"MOVBLZX AL,CX", "MOVBLZX", []Operand{AL, CX}, "0fb6c8", "MOVZX"},
@@ -591,14 +605,15 @@ func TestImmediateTruncation(t *testing.T) {
// TestEncodableCmovSize pins the linter contract for CMOVcc: Encodable must
// reject the spellings Encode rejects, so a mnemonic like CMOVBGT (no size
// letter) is not reported as encodable.
// letter) is not reported as encodable. The condition-name spellings the
// renderer prints (CMOVB, CMOVLE) encode with the width from the operands.
func TestEncodableCmovSize(t *testing.T) {
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOVB", "CMOV", "CMOVWXX"} {
for _, m := range []string{"CMOVBGT", "CMOVXEQ", "CMOV", "CMOVWXX"} {
if Encodable(m) {
t.Errorf("Encodable(%q) = true, want false", m)
}
}
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ"} {
for _, m := range []string{"CMOVLGT", "CMOVQGT", "CMOVWLS", "CMOVLEQ", "CMOVB", "CMOVLE"} {
if !Encodable(m) {
t.Errorf("Encodable(%q) = false, want true", m)
}
@@ -1193,7 +1208,7 @@ func TestBookkeepingGroundTruth(t *testing.T) {
if err != nil {
t.Fatalf("assemble: %v", err)
}
want := "90c3"
want := "c3"
if got := hexCompact(img.Code); got != want {
t.Errorf("body %s, want %s (the bookkeeping lines contribute nothing)", got, want)
}
@@ -1221,3 +1236,153 @@ func mustParse(t *testing.T, src string) *ast.File {
}
return f
}
// TestCorpusTailSystem pins the system and control forms the toolchain's own
// amd64 testdata carries, byte for byte: the one-operand IMUL, the compare
// family, the far return, the loop, the MMX moves, the CR/DR and segment
// register moves, the TLS pseudo-base and the 0F AE/1C/C7 controls.
func TestCorpusTailSystem(t *testing.T) {
regBPT := Reg{idx: 5, size: 8}
X0, X1, X2 := vreg(t, "X0"), vreg(t, "X1"), vreg(t, "X2")
Y1, Y2, Y7 := vreg(t, "Y1"), vreg(t, "Y2"), vreg(t, "Y7")
X5, X20 := vreg(t, "X5"), vreg(t, "X20")
cases := []struct {
name string
mnem string
ops []Operand
want string
}{
{"IMUL one-op byte", "IMULB", []Operand{DX}, "f6ea"},
{"IMUL one-op long", "IMULL", []Operand{AX}, "f7e8"},
{"CMPPD", "CMPPD", []Operand{X1, X2, Imm(4)}, "660fc2d104"},
{"CMPSS", "CMPSS", []Operand{X1, X2, Imm(4)}, "f30fc2d104"},
{"CMPPS", "CMPPS", []Operand{X1, X2, Imm(4)}, "0fc2d104"},
{"RETFL", "RETFL", []Operand{Imm(4)}, "ca0400"},
{"LOOP", "LOOP", []Operand{Imm(-2)}, "e2fe"},
{"LOOPE", "LOOPE", []Operand{Imm(-2)}, "e1fe"},
{"LOOPNE", "LOOPNE", []Operand{Imm(-2)}, "e0fe"},
{"PADDD MMX", "PADDD", []Operand{Reg{idx: 2, size: 8, mmx: true}, Reg{idx: 1, size: 8, mmx: true}}, "0ffeca"},
{"MOVDQ2Q", "MOVDQ2Q", []Operand{X1, Reg{idx: 1, size: 8, mmx: true}}, "f20fd6c9"},
{"MOVNTDQ", "MOVNTDQ", []Operand{X1, Ptr(AX, 0, 16)}, "660fe708"},
{"MOVQ mmx load", "MOVQ", []Operand{Ptr(AX, 0, 8), Reg{idx: 0, size: 8, mmx: true}}, "0f6f00"},
{"MOVQ mmx store", "MOVQ", []Operand{Reg{idx: 0, size: 8, mmx: true}, Ptr(SI, 0, 8)}, "0f7f06"},
{"MOVQ CR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 1}, AX}, "0f20c0"},
{"MOVQ CR4 load", "MOVQ", []Operand{Reg{idx: 4, size: 8, ctl: 1}, DI}, "0f20e7"},
{"MOVQ CR0 store", "MOVQ", []Operand{AX, Reg{idx: 0, size: 8, ctl: 1}}, "0f22c0"},
{"MOVQ DR0 load", "MOVQ", []Operand{Reg{idx: 0, size: 8, ctl: 2}, AX}, "0f21c0"},
{"MOVQ DR7 load", "MOVQ", []Operand{Reg{idx: 7, size: 8, ctl: 2}, SI}, "0f21fe"},
{"PUSHQ FS", "PUSHQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa0"},
{"PUSHQ GS", "PUSHQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa8"},
{"POPQ FS", "POPQ", []Operand{Reg{idx: 4, size: 2, seg: 5}}, "0fa1"},
{"POPQ GS", "POPQ", []Operand{Reg{idx: 5, size: 2, seg: 6}}, "0fa9"},
{"ENDBR64", "ENDBR64", nil, "f30f1efa"},
{"CLWB", "CLWB", []Operand{Ptr(BX, 0, 8)}, "660fae33"},
{"CLDEMOTE", "CLDEMOTE", []Operand{Ptr(BX, 0, 8)}, "0f1c03"},
{"TPAUSE", "TPAUSE", []Operand{BX}, "660faef3"},
{"UMONITOR", "UMONITOR", []Operand{BX}, "f30faef3"},
{"UMWAIT", "UMWAIT", []Operand{BX}, "f20faef3"},
{"RDPID", "RDPID", []Operand{DX}, "f30fc7fa"},
{"RDPID r11", "RDPID", []Operand{Reg{idx: 11, size: 8}}, "f3410fc7fb"},
{"LEAL wide disp", "LEAL", []Operand{Idx(regBPT, Reg{idx: 10, size: 8}, 1, 0x8f1bbcdc, 8), regBPT}, "428dac15dcbc1b8f"},
{"VPERMPD", "VPERMPD", []Operand{Imm(0xd8), Y7, Y7}, "c4e3fd01ffd8"},
{"VPERMILPD", "VPERMILPD", []Operand{Imm(0xff), X1, X2}, "c4e37905d1ff"},
{"VPERMILPS", "VPERMILPS", []Operand{Imm(0xff), X1, X2}, "c4e37904d1ff"},
{"VROUNDPD", "VROUNDPD", []Operand{Imm(-1), X1, X2}, "c4e37909d1ff"},
{"VROUNDPS", "VROUNDPS", []Operand{Imm(-1), Y1, Y2}, "c4e37d08d1ff"},
{"VAESKEYGENASSIST", "VAESKEYGENASSIST", []Operand{Imm(-1), X1, X2}, "c4e379dfd1ff"},
{"VPCMPESTRI", "VPCMPESTRI", []Operand{Imm(-1), X1, X2}, "c4e37961d1ff"},
{"VPCMPESTRM", "VPCMPESTRM", []Operand{Imm(-1), X1, X2}, "c4e37960d1ff"},
{"VPCMPISTRI", "VPCMPISTRI", []Operand{Imm(-1), X1, X2}, "c4e37963d1ff"},
{"VPCMPISTRM", "VPCMPISTRM", []Operand{Imm(-1), X1, X2}, "c4e37962d1ff"},
{"VEXTRACTPS", "VEXTRACTPS", []Operand{Imm(-1), X1, AX}, "c4e37917c8ff"},
{"VPEXTRW", "VPEXTRW", []Operand{Imm(0xff), X1, AX}, "c4e37915c8ff"},
{"VPBLENDVB", "VPBLENDVB", []Operand{X0, Ptr(BX, 0, 16), X1, X2}, "c4e3714c1300"},
{"VMOVHPD load", "VMOVHPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11628"},
{"VMOVHPD load disp", "VMOVHPD", []Operand{Ptr(DX, 7, 8), X5, X5}, "c5d1166a07"},
{"VMOVHPD store", "VMOVHPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91728"},
{"VMOVLPD load", "VMOVLPD", []Operand{Ptr(AX, 0, 8), X5, X5}, "c5d11228"},
{"VMOVLPD store", "VMOVLPD", []Operand{X5, Ptr(AX, 0, 8)}, "c5f91328"},
{"VMOVQ EVEX gpr load", "VMOVQ", []Operand{Reg{idx: 4, size: 8}, X20}, "62e1fd086ee4"},
{"VMOVQ EVEX mem store", "VMOVQ", []Operand{X20, Ptr(AX, 0, 8)}, "62e1fd087e20"},
{"VMOVQ EVEX mem load", "VMOVQ", []Operand{Ptr(AX, 0, 8), X20}, "62e1fd086e20"},
}
for _, c := range cases {
code, err := Encode(c.mnem, c.ops...)
if err != nil {
t.Errorf("%s: Encode: %v", c.name, err)
continue
}
if got := hexCompact(code); got != c.want {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
}
// TestCorpusTailFileForms pins the file-level forms the toolchain's amd64
// testdata carries: the star-marked indirect jumps, the TLS pseudo-base, the
// paired-register shift spelling, the absolute RET and the jump to an
// undefined static symbol (its displacement and the TLS slot offsets are
// relocation sites, zeroed here as the kernel parity suites do).
func TestCorpusTailFileForms(t *testing.T) {
mask32 := func(b []byte, at int) { b[at], b[at+1], b[at+2], b[at+3] = 0, 0, 0, 0 }
cases := []struct {
name string
src string
want string // hex, with X marking a masked 32-bit relocation site
}{
{"star reg jump", "\tJMP *(R12)\n\tRET\n", "41ff2424c3"},
{"star sp jump", "\tJMP *4(SP)\n\tRET\n", "ff642404c3"},
{"star indexed jump", "\tJMP *(R12)(R13*4)\n\tRET\n", "43ff24acc3"},
{"TLS load", "\tMOVQ (TLS), AX\n\tRET\n", "64488b0425XXXXXXXXc3"},
{"TLS load offset", "\tMOVQ 8(TLS), DX\n\tRET\n", "64488b1425XXXXXXXXc3"},
{"colon shift", "\tSHLL CX, R11:AX\n\tRET\n", "410fa5c3c3"},
{"SP indexed local", "\tMOVQ foo(SP)(AX*1), BX\n\tRET\n", "488b1c04c3"},
}
for _, c := range cases {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...)
if at := strings.Index(c.want, "XXXXXXXX"); at >= 0 {
mask32(code, at/2) // the masked relocation site
}
if got := hexCompact(code); got != strings.ReplaceAll(c.want, "X", "0") {
t.Errorf("%s: bytes %s, want %s", c.name, got, c.want)
}
}
// JMP to an undefined static symbol and the absolute RET: their rel32
// carries a call relocation against the symbol, masked to zero here.
for _, c := range []struct{ name, src string }{
{"static external jump", "\tJMP bar<>+4(SB)\n\tRET\n"},
{"static external indexed jump", "\tJMP bar<>+4(SB)(R11*4)\n\tRET\n"},
{"absolute ret", "\tRET\n\tRET foo(SB)\n"},
} {
f, errs := parser.Parse("t_amd64.s", "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n"+c.src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", c.name, errs)
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", c.name, err)
continue
}
fn := img.Funcs[0]
code := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
want := "e900000000c3"
if c.name == "absolute ret" {
want = "c3e900000000"
}
if got := hexCompact(code); got != want {
t.Errorf("%s: bytes %s, want %s", c.name, got, want)
}
}
}
+66 -17
View File
@@ -856,37 +856,41 @@ type evexMoveSpec struct {
vecOK bool // the non-memory operand may be a vector register
xmmOnly bool // wider than XMM registers are rejected
nds3 bool // a three-operand register form exists (VMOVSD/VMOVSS)
gprOK bool // the r/m side may be a general-purpose register (VMOVQ)
}
// evexMoveTable maps an upper-case EVEX move mnemonic to its encoding.
var evexMoveTable = map[string]evexMoveSpec{
// EVEX.128/256/512.F3.0F.W0, unaligned integer move.
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU32": {1, 2, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F3.0F.W1, unaligned qword move.
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU64": {1, 2, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W0, unaligned byte move (byte/word moves use the
// F2 prefix, dword/qword moves F3; the element size only changes the tuple
// semantics).
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU8": {1, 3, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.F2.0F.W1, unaligned word move (shares the qword
// encoding).
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVDQU16": {1, 3, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F.W1, unaligned packed double move.
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVUPD": {1, 1, 0x10, 0x11, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512, aligned packed moves.
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVAPS": {1, 0, 0x28, 0x29, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVAPD": {1, 1, 0x28, 0x29, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128/256/512.66.0F, aligned integer moves.
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false},
"VMOVDQA32": {1, 1, 0x6F, 0x7F, 0, [3]int{16, 32, 64}, true, false, false, false},
"VMOVDQA64": {1, 1, 0x6F, 0x7F, 1, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.F3.0F.W0, scalar single move, memory operands (the
// three-operand register form is not supported).
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true},
"VMOVSS": {1, 2, 0x10, 0x11, 0, [3]int{4, 4, 4}, false, true, true, false},
// EVEX.128.F2.0F.W1, scalar double move: memory operands and the
// three-operand register form (VMOVSD dst, src1, src2).
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true},
"VMOVSD": {1, 3, 0x10, 0x11, 1, [3]int{8, 8, 8}, false, true, true, false},
// EVEX.128/256/512.0F.W0, unaligned packed single move.
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false},
"VMOVUPS": {1, 0, 0x10, 0x11, 0, [3]int{16, 32, 64}, true, false, false, false},
// EVEX.128.66.0F.W1, the 64-bit GPR/memory ↔ XMM move (VMOVQ RSP, X20
// and friends, the EVEX spelling the high registers demand).
"VMOVQ": {1, 1, 0x6E, 0x7E, 1, [3]int{8, 8, 8}, true, true, false, true},
}
// isEvex reports whether the mnemonic has an EVEX encoding we handle.
@@ -900,6 +904,9 @@ func isEvex(mnemUpper string) bool {
if _, ok := evexMoveTable[mnemUpper]; ok {
return true
}
if _, ok := evexHptrTable[mnemUpper]; ok {
return true
}
return isEvexQuad(mnemUpper)
}
@@ -911,8 +918,14 @@ func evexRequired(upper string, ops []Operand) bool {
_, inVex := vexTable[upper]
_, inVexMove := vexMoveTable[upper]
if !inVex && !inVexMove {
// The dual-shape moves pick their VEX form by operand count, so
// they are not EVEX-only either.
switch upper {
case "VMOVHPD", "VMOVLPD", "VMOVHPS", "VMOVLPS":
default:
return true // EVEX-only mnemonic
}
}
// The byte-quad shifts have VEX register forms but EVEX-only memory
// forms: a memory count source forces the EVEX encoding.
if upper == "VPSLLDQ" || upper == "VPSRLDQ" {
@@ -1018,6 +1031,8 @@ var evexRound = map[string]bool{
"VCVTTSD2USIL": true, "VCVTTSD2USIQ": true, "VCVTTSS2USIL": true, "VCVTTSS2USIQ": true,
"VCVTSI2SDQ": true, "VCVTSI2SSL": true, "VCVTSI2SSQ": true,
"VCVTUSI2SDQ": true, "VCVTUSI2SSL": true, "VCVTUSI2SSQ": true,
// The scalar compares suppress exceptions on their LIG encoding.
"VCMPSD": true, "VCMPSS": true,
}
// evexBcstN maps an instruction accepting .BCST to the broadcast element
@@ -1082,6 +1097,14 @@ func (e *enc) encodeEvex(mnemUpper string, ops []Operand, sfx evexSuffix) error
return e.encodeEvexRM(spec, ops, 0, sfx)
}
spec, inTable := evexTable[mnemUpper]
// A high/low half move that lives in the hptr table alone (the packed
// double twins) reaches the same inTable block below, which completes
// its spec from the hptr entry.
if !inTable {
if _, ok := evexHptrTable[mnemUpper]; ok {
inTable = true
}
}
if q, ok := evexQuadTable[mnemUpper]; ok {
// The quad-register family carries no rounding, SAE or broadcast;
// only masking and zeroing apply.
@@ -1446,6 +1469,19 @@ func (e *enc) encodeEvexExtractGPR(spec evexSpec, ops []Operand, mask int, sfx e
// assembler. The scalar moves also carry a three-operand register form
// (VMOVSD dst, src1, src2: the load opcode with vvvv = src1), which ms.nds3
// opens.
// validEvexMoveOther reports whether the non-vector side of an EVEX move may
// take the operand: memory always, a general-purpose register when gprOK.
func validEvexMoveOther(ms evexMoveSpec, op Operand) bool {
if memOperand(op) {
return true
}
if !ms.gprOK {
return false
}
r, ok := op.(Reg)
return ok && !r.isVec() && !r.mask && r.ctl == 0 && !r.mmx && !r.fp
}
func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask int, sfx evexSuffix) error {
if len(ops) == 3 {
if !ms.nds3 {
@@ -1453,7 +1489,7 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
}
// The masked scalar register form keeps the Go assembler's own
// layout: the store opcode with reg = op0, vvvv = op1 and the
// destination in r/m (op2) — the bytes go tool asm emits, not
// destination in r/m (op2), the bytes go tool asm emits, not
// the manual's NDS reading.
src, src1, dst := ops[0], ops[1], ops[2]
reg, ok := src.(Reg)
@@ -1494,12 +1530,12 @@ func (e *enc) encodeEvexMove(mnem string, ms evexMoveSpec, ops []Operand, mask i
}
reg, rm = srcReg, dst
case srcIsVec:
if !memOperand(dst) {
if !validEvexMoveOther(ms, dst) {
return fmt.Errorf("%s: invalid destination operand", mnem)
}
reg, rm = srcReg, dst
case dstIsVec:
if !memOperand(src) {
if !validEvexMoveOther(ms, src) {
return fmt.Errorf("%s: invalid source operand", mnem)
}
op = ms.load
@@ -1890,6 +1926,15 @@ var evexHptrTable = map[string]evexHptrSpec{
"VMOVLHPS": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 0, pp: 0, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
},
// The packed-double twins, 66-prefixed.
"VMOVHPD": {
insert: evexSpec{mapSel: 1, opcode: 0x16, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x17, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
"VMOVLPD": {
insert: evexSpec{mapSel: 1, opcode: 0x12, w: 1, pp: 1, opdigit: -1, form: vexNDS3, n: [3]int{8, 0, 0}},
store: evexSpec{mapSel: 1, opcode: 0x13, w: 1, pp: 1, opdigit: -1, form: vexRMRev, n: [3]int{8, 0, 0}},
},
}
// encodeEvexPrefGather encodes a gather/scatter prefetch hint: OP K, vsib.
@@ -1954,7 +1999,7 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !maskReg.isVec() {
return fmt.Errorf("%s: mask must be a vector register", upper)
}
vsib, _, err := vsibLen(rest[1], upper)
vsib, idxLen, err := vsibLen(rest[1], upper)
if err != nil {
return err
}
@@ -1962,12 +2007,16 @@ func (e *enc) encodeGather(upper string, gs gatherSpec, ops []Operand, sfx evexS
if !ok || !dst.isVec() {
return fmt.Errorf("%s: destination must be a vector register", upper)
}
// The L bit is the wider of the data register and the VSIB index
// lengths (a YMM index under an XMM destination selects 256-bit, the
// bytes go tool asm emits).
ll := max(idxLen, dst.vecLenBit())
spec := vexSpec{mapSel: 2, opcode: gs.opcode, w: gs.w, pp: 1, opdigit: -1}
rBit := 0
if dst.idx >= 8 {
rBit = 1
}
return e.emitVexFields(spec, dst.vecLenBit(), dst.idx&7, rBit, 15-maskReg.idx, vsib)
return e.emitVexFields(spec, ll, dst.idx&7, rBit, 15-maskReg.idx, vsib)
}
// encodeScatter encodes a scatter (EVEX only): OP src, K, vsib, reg = src,
+159
View File
@@ -0,0 +1,159 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// The extended-instruction registry: the lookup over and above the generated
// architecture tables. The generated tables (arch/*_gen.go) list the
// mnemonics the Go toolchain knows; the extension layer carries the
// instructions it does not, and this file indexes them per architecture so
// the assembler and the linter can consult the layer without touching the
// generated lists or the main encoders. A later hook wires
// ExtensionEncodable into the Encodable mirror and EncodeExtension into the
// per-architecture assembly paths; nothing existing changes until then.
package asm
import (
"fmt"
"slices"
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// extensionIndex is the per-architecture index of the extension layer, keyed
// by upper-case mnemonic. One mnemonic registers several forms (the SVE ADD
// carries unpredicated, predicated and immediate shapes), so the value is the
// full candidate list in table order.
type extensionIndex struct {
byName map[string][]arch.ExtInstr
}
// extensionIndexes builds one index per known architecture. Architectures
// whose extension layer is not built yet get an empty index, which keeps the
// queries answering false rather than failing on a missing entry.
var extensionIndexes = buildExtensionIndexes()
func buildExtensionIndexes() map[arch.Arch]*extensionIndex {
m := make(map[arch.Arch]*extensionIndex)
for _, a := range []arch.Arch{arch.AMD64, arch.ARM64, arch.RISCV, arch.LOONG64} {
idx := &extensionIndex{byName: make(map[string][]arch.ExtInstr)}
for _, in := range arch.Extensions(a) {
key := strings.ToUpper(in.Name)
idx.byName[key] = append(idx.byName[key], in)
}
m[a] = idx
}
return m
}
// LookupExtension returns the extended instructions registered for the
// mnemonic on a, outside the generated architecture table. It reports false
// when a carries no extended layer or the mnemonic is not in it; a mnemonic
// the base table knows is not thereby covered, the layers stay independent.
func LookupExtension(a arch.Arch, mnemonic string) ([]arch.ExtInstr, bool) {
idx, ok := extensionIndexes[a]
if !ok || idx == nil {
return nil, false
}
cands, ok := idx.byName[strings.ToUpper(mnemonic)]
return cands, ok && len(cands) > 0
}
// ExtensionNames returns the mnemonics the extension layer of a registers,
// in table order, without duplicates.
func ExtensionNames(a arch.Arch) []string {
var names []string
seen := make(map[string]bool)
for _, in := range arch.Extensions(a) {
key := strings.ToUpper(in.Name)
if !seen[key] {
seen[key] = true
names = append(names, in.Name)
}
}
return names
}
// EncodeExtension encodes one extended instruction on a: it resolves the
// mnemonic through the extension registry, picks the registered form whose
// arity matches the operands and encodes against it. The first form that
// encodes wins. When every matching form rejects the operands, the error
// comes from the form whose operand kinds the list points at (the one with
// the most matching positions), so a mis-spelled predicate qualifier is
// diagnosed as one, not as the unpredicated form's register complaint.
func EncodeExtension(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) ([]byte, error) {
cands, ok := LookupExtension(a, mnemonic)
if !ok {
return nil, fmt.Errorf("%s registers no extended instruction %q", a, mnemonic)
}
var bestErr error
var bestScore int
var tried int
for _, in := range cands {
if in.Form.Arity() != len(ops) {
continue
}
tried++
b, err := in.Encode(ops)
if err == nil {
return b, nil
}
if score := kindScore(in.Form, ops); bestErr == nil || score > bestScore {
bestErr, bestScore = err, score
}
}
if tried == 0 {
return nil, fmt.Errorf("%s: extended %q takes %s, got %d operands",
a, mnemonic, extensionAritySummary(cands), len(ops))
}
return nil, bestErr
}
// kindScore counts the positions whose operand kind matches what the form
// wants, the tie-break that picks the most specific rejection.
func kindScore(form arch.ExtForm, ops []arch.ExtOperand) int {
kinds := form.Kinds()
score := 0
for i, op := range ops {
if i < len(kinds) && op.Kind == kinds[i] {
score++
}
}
return score
}
// ExtensionEncodable reports whether the extension layer of a encodes the
// mnemonic with these operands. It mirrors asm.Encodable for the extension
// layer: the predicate the linter consults once the hook wires it in.
func ExtensionEncodable(a arch.Arch, mnemonic string, ops ...arch.ExtOperand) bool {
_, err := EncodeExtension(a, mnemonic, ops...)
return err == nil
}
// extensionAritySummary describes the operand counts the candidate forms
// take, "2 or 3" style, for the arity error.
func extensionAritySummary(cands []arch.ExtInstr) string {
counts := make([]int, 0, len(cands))
seen := make(map[int]bool)
for _, in := range cands {
n := in.Form.Arity()
if !seen[n] {
seen[n] = true
counts = append(counts, n)
}
}
slices.Sort(counts)
var b strings.Builder
for i, n := range counts {
if i > 0 {
if i == len(counts)-1 {
b.WriteString(" or ")
} else {
b.WriteString(", ")
}
}
fmt.Fprintf(&b, "%d", n)
}
b.WriteString(" operands")
return b.String()
}
+269
View File
@@ -0,0 +1,269 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// TestAmd64ExtensionRegistry checks the mnemonic lookup for the amd64 layer:
// one mnemonic across several vector lengths or W bits resolves to every
// entry, the lookup is case-insensitive, and the counts match the registered
// families.
func TestAmd64ExtensionRegistry(t *testing.T) {
for _, tt := range []struct {
mnem string
forms int
}{
{"VCVTNE2PS2BF16", 3},
{"VCVTNEPS2BF16", 3},
{"VDPBF16PS", 3},
{"VP2INTERSECTD", 3},
{"VP2INTERSECTQ", 3},
{"VMOVSH", 3},
{"VMOVW", 4},
{"VADDSH", 1},
{"VSQRTSH", 1},
{"VCOMISH", 1},
{"VCVTSH2SS", 1},
{"VCVTSI2SH", 2},
{"VCVTSH2SI", 2},
{"VADDPH", 3},
{"VSQRTPH", 3},
{"VSCALEFSH", 1},
{"VGETEXPSH", 1},
{"VCMPSH", 1},
{"VGETMANTSH", 1},
{"VREDUCESH", 1},
{"VRNDSCALESH", 1},
{"VCVTPH2W", 3},
{"VCVTPH2UW", 3},
{"VCVTW2PH", 3},
{"VCVTUW2PH", 3},
{"VCVTPH2DQ", 3},
{"VCVTPH2UDQ", 3},
{"VCVTDQ2PH", 3},
{"VCVTUDQ2PH", 3},
{"VCVTPH2QQ", 3},
{"VCVTPH2UQQ", 3},
{"VCVTQQ2PH", 3},
{"VCVTUQQ2PH", 3},
{"VCVTPH2PD", 3},
{"VCVTPD2PH", 3},
{"VRNDSCALEPH", 3},
{"VREDUCEPH", 3},
{"VGETMANTPH", 3},
{"VFMADD132PH", 3},
{"VFMADD213PH", 3},
{"VFMADD231PH", 3},
{"VFMSUB132PH", 3},
{"VFMSUB213PH", 3},
{"VFMSUB231PH", 3},
{"VFMADDSUB132PH", 3},
{"VFMADDSUB213PH", 3},
{"VFMADDSUB231PH", 3},
{"VFMSUBADD132PH", 3},
{"VFMSUBADD213PH", 3},
{"VFMSUBADD231PH", 3},
{"VFMADD132SH", 1},
{"VFMADD213SH", 1},
{"VFMADD231SH", 1},
{"VFMSUB132SH", 1},
{"VFMSUB213SH", 1},
{"VFMSUB231SH", 1},
{"VFMULCPH", 3},
{"VFCMULCPH", 3},
{"VFMULCSH", 1},
{"VFCMULCSH", 1},
{"VFMADDCPH", 3},
{"VFCMADDCPH", 3},
{"VFMADDCSH", 1},
{"VFCMADDCSH", 1},
{"VMINMAXPH", 3},
{"VMINMAXSH", 1},
{"VPDPWSUD", 2},
{"VPDPWSUDS", 2},
{"VPDPWUSD", 2},
{"VPDPWUSDS", 2},
} {
cands, ok := LookupExtension(arch.AMD64, tt.mnem)
if !ok {
t.Fatalf("LookupExtension(AMD64, %s) found nothing", tt.mnem)
}
if len(cands) != tt.forms {
t.Errorf("%s registers %d forms, want %d", tt.mnem, len(cands), tt.forms)
}
lower, ok := LookupExtension(arch.AMD64, strings.ToLower(tt.mnem))
if !ok || len(lower) != tt.forms {
t.Errorf("the %s lookup is not case-insensitive", tt.mnem)
}
}
if got := arch.Extensions(arch.AMD64); len(got) != 191 {
t.Errorf("the amd64 layer registers %d instructions, want 191", len(got))
}
if _, ok := LookupExtension(arch.AMD64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")
}
// VPOPCNTD and VPOPCNTQ are toolchain instructions today: they stay in
// the generated table and out of the extension layer.
if _, ok := LookupExtension(arch.AMD64, "VPOPCNTD"); ok {
t.Error("VPOPCNTD is an extension, want it in the generated table alone")
}
}
// TestAmd64ExtensionAboveGeneratedTable pins the layering twice over: no
// registered mnemonic sits in the generated amd64 table, and the encoder
// mirror asm.Encodable answers false for every one of them, so the layer
// stays out of the main encoders by test and not by promise.
func TestAmd64ExtensionAboveGeneratedTable(t *testing.T) {
for _, mnem := range ExtensionNames(arch.AMD64) {
if _, found := arch.ForArch(arch.AMD64).Lookup(mnem); found {
t.Errorf("%s leaked into the generated amd64 table", mnem)
}
if Encodable(mnem) {
t.Errorf("%s is encodable through the main encoder, the layer is not sealed", mnem)
}
}
}
// TestEncodeExtensionAmd64 encodes through the registry and pins the same
// golden words the arch table tests pin, proving the registry resolves to the
// right encoding.
func TestEncodeExtensionAmd64(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
ops []arch.ExtOperand
want string
}{
{"bf16 convert", "VCVTNE2PS2BF16",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtZmm(6)},
"62f2574872f4"},
{"bf16 narrow convert", "VCVTNEPS2BF16",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtYmm(6)},
"62f27e4872f5"},
{"dot product", "VDPBF16PS",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtXmm(6)},
"62f2560852f4"},
{"intersect into a mask", "VP2INTERSECTD",
[]arch.ExtOperand{arch.ExtYmm(2), arch.ExtYmm(1), arch.ExtMask(2)},
"62f26f2868d1"},
{"scalar fp16 add, high registers", "VADDSH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)},
"6205160058f4"},
{"scalar compare", "VCOMISH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtXmm(30)},
"62057c082ff5"},
{"integer into a scalar fp16", "VCVTSI2SH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtGpr64(12), arch.ExtXmm(30)},
"624596002af4"},
{"scalar fp16 into an integer", "VCVTSH2SI",
[]arch.ExtOperand{arch.ExtXmm(30), arch.ExtGpr32(2)},
"62957e082dd6"},
{"word move into an xmm", "VMOVW",
[]arch.ExtOperand{arch.ExtGpr64(12), arch.ExtXmm(30)},
"62457d086ef4"},
{"scalar compare into a mask", "VCMPSH",
[]arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtMask(5)},
"62931600c2ec7b"},
{"mantissa extract with a control byte", "VGETMANTSH",
[]arch.ExtOperand{arch.ExtImmediate(0x0b), arch.ExtXmm(29), arch.ExtXmm(28), arch.ExtXmm(30)},
"6203140027f40b"},
{"packed fp16 add under embedded rounding", "VADDPH",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundTruncate)},
"62f5547858f4"},
{"scalar fp16 minimum with exceptions suppressed", "VMINSH",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundSAE)},
"62f556185df4"},
{"fp16 to signed words", "VCVTPH2W",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(6)},
"62f57d487df5"},
{"dwords to fp16 over a broadcast source", "VCVTDQ2PH",
[]arch.ExtOperand{arch.ExtBroadcast(9, 0), arch.ExtYmm(30)},
"62457c585b31"},
{"fp16 to signed qwords, rounded", "VCVTPH2QQ",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundTruncate)},
"62f57d787bf5"},
{"fp16 scalar out of memory into an integer", "VCVTSH2SI",
[]arch.ExtOperand{arch.ExtMemory(9, 0), arch.ExtGpr64(12)},
"6255fe082d21"},
{"fp16 to double-precision, widened", "VCVTPH2PD",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtZmm(6)},
"62f57c485af5"},
{"packed fp16 rounding to fraction bits", "VRNDSCALEPH",
[]arch.ExtOperand{arch.ExtImmediate(0x7b), arch.ExtZmm(5), arch.ExtZmm(6)},
"62f37c4808f57b"},
{"packed fused multiply-add, high registers", "VFMADD132PH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)},
"6206154098f4"},
{"fused multiply-add out of a broadcast source", "VFMADD231PH",
[]arch.ExtOperand{arch.ExtYmm(5), arch.ExtBroadcast(1, 0), arch.ExtYmm(6)},
"62f65538b831"},
{"scalar multiply-subtract under embedded rounding", "VFMSUB231SH",
[]arch.ExtOperand{arch.ExtXmm(5), arch.ExtXmm(4), arch.ExtRounded(arch.ExtXmm(6), arch.ExtRoundTruncate)},
"62f65578bbf4"},
{"complex multiply, conjugating the first source", "VFCMULCPH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)},
"62061740d6f4"},
{"complex multiply-add, conjugating the second source", "VFMADDCPH",
[]arch.ExtOperand{arch.ExtZmm(29), arch.ExtZmm(28), arch.ExtZmm(30)},
"6206164056f4"},
{"complex multiply-add, conjugated first source, rounded", "VFCMADDCPH",
[]arch.ExtOperand{arch.ExtZmm(5), arch.ExtZmm(4), arch.ExtRounded(arch.ExtZmm(6), arch.ExtRoundNearest)},
"62f6571856f4"},
{"scalar complex multiply-add out of memory", "VFMADDCSH",
[]arch.ExtOperand{arch.ExtXmm(29), arch.ExtMemory(9, 0), arch.ExtXmm(30)},
"624616005731"},
{"minimum or maximum under a control byte", "VMINMAXPH",
[]arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtZmm(29), arch.ExtMemory(9, 0), arch.ExtZmm(30)},
"62431440523188"},
{"scalar minimum or maximum out of memory", "VMINMAXSH",
[]arch.ExtOperand{arch.ExtImmediate(0x88), arch.ExtXmm(28), arch.ExtMemory(9, 0), arch.ExtXmm(29)},
"62431c00532988"},
{"vnni dot product through the VEX word", "VPDPWSUD",
[]arch.ExtOperand{arch.ExtXmm(2), arch.ExtXmm(1), arch.ExtXmm(3)},
"c4e26ad2d9"},
{"saturating dot product, high registers", "VPDPWUSDS",
[]arch.ExtOperand{arch.ExtYmm(10), arch.ExtYmm(15), arch.ExtYmm(8)},
"c4422dd3c7"},
{"dot product out of memory", "VPDPWSUD",
[]arch.ExtOperand{arch.ExtXmm(2), arch.ExtMemory(1, 127), arch.ExtXmm(1)},
"c4e26ad2497f"},
} {
got, err := EncodeExtension(arch.AMD64, tt.mnem, tt.ops...)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if hex.EncodeToString(got) != tt.want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, tt.want)
}
}
}
// TestEncodeExtensionAmd64Errors checks the registry's diagnostics on the
// amd64 side: a wrong arity names the form's count and a mis-classed operand
// surfaces the entry's own message.
func TestEncodeExtensionAmd64Errors(t *testing.T) {
if _, err := EncodeExtension(arch.AMD64, "VP2INTERSECTD", arch.ExtZmm(1)); err == nil {
t.Error("one operand encoded, want an arity error")
} else if !strings.Contains(err.Error(), "3 operands") {
t.Errorf("arity error %q does not name the count", err)
}
_, err := EncodeExtension(arch.AMD64, "VCVTNEPS2BF16", arch.ExtZmm(1), arch.ExtZmm(2))
if err == nil {
t.Fatal("a ZMM destination encoded on the narrow convert, want an error")
}
if !strings.Contains(err.Error(), "YMM register") {
t.Errorf("error %q does not name the YMM destination", err)
}
if _, err := EncodeExtension(arch.AMD64, "VCVTNE2PS2BF16"); err == nil ||
!strings.Contains(err.Error(), "takes 3 operands, got 0") {
t.Errorf("zero-operand error = %v, want the operand-count diagnostic", err)
}
}
+534
View File
@@ -0,0 +1,534 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
)
// TestExtensionRegistryARM64 checks the mnemonic lookup over and above the
// generated arm64 table: one mnemonic, several forms, case-insensitive, and
// nothing offered for a spelling the layer does not carry.
func TestExtensionRegistryARM64(t *testing.T) {
add, ok := LookupExtension(arch.ARM64, "ADD")
if !ok {
t.Fatal("LookupExtension(ARM64, ADD) found nothing")
}
var forms []arch.ExtForm
for _, in := range add {
if in.Name != "ADD" {
t.Errorf("candidate %q leaked into the ADD lookup", in.Name)
}
forms = append(forms, in.Form)
}
if len(forms) != 3 ||
forms[0] != arch.ExtFormVectors ||
forms[1] != arch.ExtFormPredicated ||
forms[2] != arch.ExtFormImmediate {
t.Errorf("ADD registers forms %v, want unpredicated, predicated and immediate", forms)
}
if _, ok := LookupExtension(arch.ARM64, "add"); !ok {
t.Error("the lookup is case-sensitive")
}
if _, ok := LookupExtension(arch.ARM64, "NOSUCHINSTR"); ok {
t.Error("a non-extended mnemonic resolved")
}
sqadd, ok := LookupExtension(arch.ARM64, "SQADD")
if !ok || len(sqadd) != 2 {
t.Errorf("SQADD registers %d forms, want the unpredicated and immediate pair", len(sqadd))
}
}
// TestExtensionAboveGeneratedTable pins the layering: SQADD is nowhere in the
// generated arm64 table (the toolchain knows only the NEON spelling VSQADD)
// yet the extension layer carries it, while ADD sits in both layers
// independently.
func TestExtensionAboveGeneratedTable(t *testing.T) {
if _, found := arch.ForArch(arch.ARM64).Lookup("SQADD"); found {
t.Error("SQADD is in the generated table, the layering assumption broke")
}
if _, ok := LookupExtension(arch.ARM64, "SQADD"); !ok {
t.Error("SQADD is missing from the extension layer")
}
if _, found := arch.ForArch(arch.ARM64).Lookup("ADD"); !found {
t.Error("ADD vanished from the generated table")
}
if add, ok := LookupExtension(arch.ARM64, "ADD"); !ok || len(add) != 3 {
t.Errorf("ADD carries %d extension forms, want 3", len(add))
}
}
// TestEncodeExtensionGolden encodes through the registry and pins the same
// golden words the arch table tests pin, proving the registry resolves to the
// right encoding.
func TestEncodeExtensionGolden(t *testing.T) {
for _, tt := range []struct {
name string
mnem string
ops []arch.ExtOperand
want uint32
}{
{"unpredicated add", "ADD",
[]arch.ExtOperand{
arch.ExtVector(2, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
},
0x04200040},
{"predicated mul", "MUL",
[]arch.ExtOperand{
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(2, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrB),
},
0x04100800},
{"immediate add with derived shift", "ADD",
[]arch.ExtOperand{arch.ExtImmediate(32512), arch.ExtVector(0, arch.ExtArrH)},
0x2560efe0},
{"signed immediate mul", "MUL",
[]arch.ExtOperand{arch.ExtImmediate(-1), arch.ExtVector(0, arch.ExtArrB)},
0x2530dfe0},
} {
got, err := EncodeExtension(arch.ARM64, tt.mnem, tt.ops...)
if err != nil {
t.Errorf("%s: encode: %v", tt.name, err)
continue
}
if want := hex.EncodeToString([]byte{
byte(tt.want), byte(tt.want >> 8), byte(tt.want >> 16), byte(tt.want >> 24),
}); hex.EncodeToString(got) != want {
t.Errorf("%s:\n got %x\n want %s", tt.name, got, want)
}
}
}
// TestEncodeExtensionErrors checks the registry's diagnostics: a wrong arity
// names every form's count, an operand the first candidate rejects surfaces
// its own message once a later form takes over.
func TestEncodeExtensionErrors(t *testing.T) {
if _, err := EncodeExtension(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)); err == nil {
t.Error("one operand encoded, want an arity error")
} else if !strings.Contains(err.Error(), "2 or 3 operands") {
t.Errorf("arity error %q does not name the counts", err)
}
// The predicated candidate must answer for its own operands: the /Z
// qualifier is rejected with the merging message, not the unpredicated
// form's register-kind complaint.
_, err := EncodeExtension(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrB), arch.ExtPredicate(0, arch.ExtQualZeroing), arch.ExtVector(0, arch.ExtArrB))
if err == nil {
t.Fatal("/Z encoded, want an error")
}
if !strings.Contains(err.Error(), "/M") {
t.Errorf("error %q does not name the merging qualifier", err)
}
if _, err := EncodeExtension(arch.ARM64, "NOSUCHINSTR", arch.ExtVector(0, arch.ExtArrB)); err == nil ||
!strings.Contains(err.Error(), "registers no extended instruction") {
t.Errorf("unknown mnemonic error = %v", err)
}
}
// TestExtensionEncodable checks the predicate the later Encodable hook will
// call: true exactly when the registry encodes the operand list.
func TestExtensionEncodable(t *testing.T) {
if !ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(1, arch.ExtArrS), arch.ExtVector(2, arch.ExtArrS)) {
t.Error("an encodable unpredicated add reported false")
}
if !ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(1, arch.ExtArrS), arch.ExtPredicate(0, arch.ExtQualMerging), arch.ExtVector(0, arch.ExtArrS)) {
t.Error("an encodable predicated add reported false")
}
if ExtensionEncodable(arch.ARM64, "ADD",
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrS), arch.ExtVector(0, arch.ExtArrB)) {
t.Error("mismatched arrangements reported encodable")
}
if ExtensionEncodable(arch.ARM64, "ADD", arch.ExtVector(0, arch.ExtArrB)) {
t.Error("a one-operand add reported encodable")
}
if ExtensionEncodable(arch.ARM64, "NOSUCHINSTR") {
t.Error("an unregistered mnemonic reported encodable")
}
}
// TestExtensionArchIsolation is the architecture-binding negative case: no
// architecture answers the arm64 mnemonics but arm64, and the amd64 layer
// answers nothing of the arm64 family either (its own mnemonics live in
// extension_amd64_test.go).
func TestExtensionArchIsolation(t *testing.T) {
ops := []arch.ExtOperand{
arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB), arch.ExtVector(0, arch.ExtArrB),
}
for _, a := range []arch.Arch{arch.RISCV, arch.LOONG64, arch.Unknown} {
if cands, ok := LookupExtension(a, "ADD"); ok || cands != nil {
t.Errorf("LookupExtension(%s, ADD) offered %d candidates", a, len(cands))
}
if cands, ok := LookupExtension(a, "MUL"); ok || cands != nil {
t.Errorf("LookupExtension(%s, MUL) offered %d candidates", a, len(cands))
}
if got, err := EncodeExtension(a, "ADD", ops...); err == nil {
t.Errorf("EncodeExtension(%s, ADD) encoded %x, want a refusal", a, got)
} else if !strings.Contains(err.Error(), string(a)) {
t.Errorf("EncodeExtension(%s) error %q does not name the architecture", a, err)
}
if ExtensionEncodable(a, "ADD", ops...) {
t.Errorf("ExtensionEncodable(%s, ADD) reported true", a)
}
if names := ExtensionNames(a); len(names) != 0 {
t.Errorf("ExtensionNames(%s) = %v, want none", a, names)
}
if got := arch.Extensions(a); len(got) != 0 {
t.Errorf("arch.Extensions(%s) carries %d instructions", a, len(got))
}
}
// The amd64 layer exists but stays silent about the arm64 family.
if cands, ok := LookupExtension(arch.AMD64, "ADD"); ok || cands != nil {
t.Errorf("LookupExtension(AMD64, ADD) offered %d candidates", len(cands))
}
if got, err := EncodeExtension(arch.AMD64, "ADD", ops...); err == nil {
t.Errorf("EncodeExtension(AMD64, ADD) encoded %x, want a refusal", got)
} else if !strings.Contains(err.Error(), string(arch.AMD64)) {
t.Errorf("EncodeExtension(AMD64) error %q does not name the architecture", err)
}
if ExtensionEncodable(arch.AMD64, "ADD", ops...) {
t.Error("ExtensionEncodable(AMD64, ADD) reported true")
}
if names := ExtensionNames(arch.AMD64); len(names) == 0 {
t.Error("the amd64 layer registers no names")
}
if got := arch.Extensions(arch.AMD64); len(got) == 0 {
t.Error("arch.Extensions(AMD64) is empty")
}
}
// TestExtensionNamesARM64 checks the completion-facing name list: every
// distinct mnemonic of the family, first-occurrence order, no duplicates.
func TestExtensionNamesARM64(t *testing.T) {
want := []string{
// The SVE integer add/subtract/multiply family.
"ADD", "SUB", "SQADD", "UQADD", "SQSUB", "UQSUB", "MUL", "SMULH", "UMULH", "SUBR",
// The SVE and SVE2.1 predicate family: the logical operations, the
// breaks, the permutations, the singles, the first-fault group and
// the while compares.
"PAND", "PANDS", "PBIC", "PBICS", "PEOR", "PEORS",
"PNAND", "PNANDS", "PNOR", "PNORS", "PORN", "PORNS", "PORR", "PORRS",
"PSEL",
"PBRKA", "PBRKAS", "PBRKB", "PBRKBS", "PBRKN", "PBRKNS",
"PBRKPA", "PBRKPAS", "PBRKPB", "PBRKPBS",
"PTRN1", "PTRN2", "PUZP1", "PUZP2", "PZIP1", "PZIP2",
"PPFALSE", "PPFIRST", "PPNEXT", "PPTEST", "PPTRUE", "PPUNPKHI", "PPUNPKLO",
"PRDFFR", "PRDFFRS", "PWRFFR", "PREV", "SETFFR",
"PWHILEGE", "PWHILEGT", "PWHILEHI", "PWHILEHS",
"PWHILELE", "PWHILELO", "PWHILELS", "PWHILELT", "PWHILERW", "PWHILEWR",
// The SVE2.1 Z-alias families, in first-occurrence order.
"ZABS",
"ZREVB",
"ZREVH",
"ZSXTB",
"ZSXTH",
"ZUXTB",
"ZUXTH",
"ZADD",
"ZAND",
"ZBIC",
"ZCLS",
"ZCLZ",
"ZCNOT",
"ZCNT",
"ZCOMPACT",
"ZDECP",
"ZDUP",
"ZEOR",
"ZEXPAND",
"ZINSR",
"ZLASTA",
"ZLASTB",
"ZMOVPRFX",
"ZREVD",
"ZREVW",
"ZSXTW",
"ZUXTW",
"ZNEG",
"ZNOT",
"ZORR",
"ZRBIT",
"ZREV",
"ZSEL",
"ZSQABS",
"ZSQNEG",
"ZSUB",
"ZSUBR",
"ZSUNPKHI",
"ZSUNPKLO",
"ZTBX",
"ZTBXQ",
"ZTRN1",
"ZTRN2",
"ZUUNPKHI",
"ZUUNPKLO",
"ZUZP1",
"ZUZP2",
"ZUZPQ1",
"ZUZPQ2",
"ZZIP1",
"ZZIP2",
"ZZIPQ1",
"ZZIPQ2",
// The stage-four families: the SVE2.1 narrowing two-to-one set, the
// pairwise forms and the quadword reductions.
"ZADDHNB",
"ZADDHNT",
"ZRADDHNB",
"ZRADDHNT",
"ZRSUBHNB",
"ZRSUBHNT",
"ZSUBHNB",
"ZSUBHNT",
"ZADDP",
"ZADDPT",
"ZADDQV",
"ZANDQV",
"ZEORQV",
"ZFADDQV",
"ZFMAXNMQV",
"ZFMAXQV",
"ZFMINNMQV",
"ZFMINQV",
"ZORQV",
"ZSMAXQV",
"ZSMINQV",
"ZUMAXQV",
"ZUMINQV",
"ZASR",
"ZASRR",
"ZLSL",
"ZLSLR",
"ZLSR",
"ZLSRR",
"ZBCAX",
"ZBDEP",
"ZBEXT",
"ZBGRP",
"ZBSL",
"ZBSL1N",
"ZBSL2N",
"ZEOR3",
"ZEORBT",
"ZEORTB",
"ZNBSL",
"ZBF1CVT",
"ZBF1CVTLT",
"ZBF2CVT",
"ZBF2CVTLT",
"ZF1CVT",
"ZF1CVTLT",
"ZF2CVT",
"ZF2CVTLT",
"ZBFADD",
"ZBFCLAMP",
"ZBFMAX",
"ZBFMAXNM",
"ZBFMIN",
"ZBFMINNM",
"ZBFMUL",
"ZBFSUB",
"ZBFCVT",
"ZBFCVTNT",
"ZBFDOT",
"ZBFMLA",
"ZBFMLALB",
"ZBFMLALT",
"ZBFMLS",
"ZBFMLSLB",
"ZBFMLSLT",
"ZBFMMLA",
"ZBFSCALE",
"ZCLASTA",
"ZCLASTB",
"ZCMPEQ",
"ZCMPGE",
"ZCMPGT",
"ZCMPHI",
"ZCMPHS",
"ZCMPNE",
"ADDPL",
"ADDVL",
"RDVL",
"ZLD2B",
"ZLD2D",
"ZLD2H",
"ZLD2Q",
"ZLD2W",
"ZLD3B",
"ZLD3D",
"ZLD3H",
"ZLD3Q",
"ZLD3W",
"ZLD4B",
"ZLD4D",
"ZLD4H",
"ZLD4Q",
"ZLD4W",
"ZST2B",
"ZST2D",
"ZST2H",
"ZST2Q",
"ZST2W",
"ZST3B",
"ZST3D",
"ZST3H",
"ZST3Q",
"ZST3W",
"ZST4B",
"ZST4D",
"ZST4H",
"ZST4Q",
"ZST4W",
// The stage-three families: the SVE2 crypto group, the predicate
// counters and loop terminators with the 32-bit while compares, and
// the reductions into a SIMD register.
"ZADCLB",
"ZADCLT",
"ZSBCLB",
"ZSBCLT",
"ZRAX1",
"ZSM4EKEY",
"ZSM4E",
"ZAESD",
"ZAESE",
"ZAESIMC",
"ZAESMC",
"CTERMEQ",
"CTERMEQW",
"CTERMNE",
"CTERMNEW",
"PCNTP",
"PFIRSTP",
"PLASTP",
"PDECP",
"PINCP",
"PSQDECP",
"PSQINCP",
"PUQDECP",
"PUQINCP",
"PUQDECPW",
"PUQINCPW",
"PSQDECPW",
"PSQINCPW",
"PWHILEGEW",
"PWHILEGTW",
"PWHILEHIW",
"PWHILEHSW",
"PWHILELEW",
"PWHILELOW",
"PWHILELSW",
"PWHILELTW",
"ZSADDVD",
"ZUADDVD",
"ZANDVB",
"ZANDVH",
"ZANDVS",
"ZANDVD",
"ZEORVB",
"ZEORVH",
"ZEORVS",
"ZEORVD",
"ZORVB",
"ZORVH",
"ZORVS",
"ZORVD",
"ZSMAXVB",
"ZSMAXVH",
"ZSMAXVS",
"ZSMAXVD",
"ZSMINVB",
"ZSMINVH",
"ZSMINVS",
"ZSMINVD",
"ZUMAXVB",
"ZUMAXVH",
"ZUMAXVS",
"ZUMAXVD",
"ZUMINVB",
"ZUMINVH",
"ZUMINVS",
"ZUMINVD",
"ZFADDVH",
"ZFADDVS",
"ZFADDVD",
"ZFMAXNMVH",
"ZFMAXNMVS",
"ZFMAXNMVD",
"ZFMAXVH",
"ZFMAXVS",
"ZFMAXVD",
"ZFMINNMVH",
"ZFMINNMVS",
"ZFMINNMVD",
"ZFMINVH",
"ZFMINVS",
"ZFMINVD",
"ZFADDAH",
"ZFADDAS",
"ZFADDAD",
// The gather loads and scatter stores: plain, sign-extended and
// first-fault loads, and the stores, in first-occurrence order.
"ZLD1B",
"ZLD1D",
"ZLD1H",
"ZLD1SB",
"ZLD1SH",
"ZLD1SW",
"ZLD1W",
"ZLDFF1B",
"ZLDFF1D",
"ZLDFF1H",
"ZLDFF1SB",
"ZLDFF1SH",
"ZLDFF1SW",
"ZLDFF1W",
"ZST1B",
"ZST1D",
"ZST1H",
"ZST1W",
// The shift-immediate classes: the narrowing and widening
// three-vector shifts and the predicated saturating left shifts,
// in first-occurrence order.
"ZSQSHRUNB",
"ZSQSHRUNT",
"ZSHRNB",
"ZSHRNT",
"ZRSHRNB",
"ZRSHRNT",
"ZSQSHRNB",
"ZSQSHRNT",
"ZSQRSHRNB",
"ZSQRSHRNT",
"ZUQSHRNB",
"ZUQSHRNT",
"ZUQRSHRNB",
"ZUQRSHRNT",
"ZSSHLLB",
"ZSSHLLT",
"ZUSHLLB",
"ZUSHLLT",
"ZSQSHL",
"ZSQSHLU",
"ZUQSHL",
"ZSRI",
"ZSSRA",
"ZUSRA",
"ZSRSRA",
"ZURSRA",
"ZASRD",
"ZXAR",
}
got := ExtensionNames(arch.ARM64)
if strings.Join(got, ",") != strings.Join(want, ",") {
t.Errorf("ExtensionNames(ARM64) = %v, want %v", got, want)
}
if n := len(arch.Extensions(arch.ARM64)); n != 537 {
t.Errorf("the family registers %d instructions, want 537", n)
}
}
+130
View File
@@ -0,0 +1,130 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"os"
"os/exec"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// flagListShapes are the TEXT shapes the flags operand decides: joined
// names, the legacy numeric spellings, the arithmetic and immediate forms,
// each with a frame size beside it. A flag the operand carries suppresses
// the stack-split guard; the toolchain reads one integer from the operand,
// and after the front end folds it the bytes must not tell the difference.
// The bodies are leaf-shaped on purpose: the frame engine's open divergences
// (the NOFRAME prologue for a non-leaf, the big-frame guard's shape) sit
// outside the flags operand and are somebody else's gap list.
var flagListShapes = []struct {
name string // the subtest's name
flags string // the flags operand as written
frame string // the frame operand as written
body string // the function body
}{
{"name", "NOSPLIT", "$4096-0", "\tMOVD R0, R1\n\tRET\n"},
{"numeric", "4", "$4096-0", "\tMOVD R0, R1\n\tRET\n"},
{"numericOR", "2|4", "$4096-0", "\tMOVD R0, R1\n\tRET\n"},
{"joinedFrame", "DUPOK|NOSPLIT", "$4096-0", "\tMOVD R0, R1\n\tRET\n"},
{"joinedFrameless", "NOSPLIT|TOPFRAME", "$0-0", "\tRET\n"},
{"parenthesised", "(NOSPLIT|NOFRAME)", "$0-0", "\tMOVD R0, R1\n\tRET\n"},
}
// TestFlagListAssemblyPathRejectsUnknown holds the front end's rejections
// that mirror the toolchain's: an identifier outside the flag table
// ("unexpected TYPO evaluating expression") and the immediate spelling of
// the operand ("TEXT: expected integer constant"), each refused on the
// assembly path before the encoder ever sees a tree.
func TestFlagListAssemblyPathRejectsUnknown(t *testing.T) {
for _, c := range []struct{ src, want string }{
{"#include \"textflag.h\"\n\nTEXT f(SB), NOSPLIT|TYPO, $0-0\n\tRET\n",
"unexpected TYPO evaluating expression"},
{"#include \"textflag.h\"\n\nTEXT f(SB), $NOSPLIT, $0-0\n\tRET\n",
"TEXT: expected integer constant; found $NOSPLIT"},
{"#include \"textflag.h\"\n\nGLOBL g<>(SB), $8, $8\n",
"GLOBL: expected integer constant; found $8"},
} {
_, errs := parser.ParseWithOptions("f.s", c.src, parser.Options{Expand: true})
if len(errs) == 0 || !strings.Contains(errs[0].Error(), c.want) {
t.Errorf("errors = %v, want %q", errs, c.want)
}
}
}
// TestFlagListByteParity assembles every flag-list shape through the fixed
// front end and holds the function's bytes equal to the installed
// toolchain's, no guard words where the flag suppresses them and none
// missing where it demands one.
func TestFlagListByteParity(t *testing.T) {
if testing.Short() {
t.Skip("live go tool asm oracle: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
}
out, err := exec.Command(goBin, "env", "GOROOT").Output()
if err != nil {
t.Fatalf("go env GOROOT: %v", err)
}
include := filepath.Join(strings.TrimSpace(string(out)), "pkg", "include")
for _, c := range flagListShapes {
t.Run(c.name, func(t *testing.T) {
src := "#include \"textflag.h\"\n\nTEXT f(SB), " + c.flags + ", " + c.frame + "\n" + c.body
dir := t.TempDir()
file := filepath.Join(dir, "f.s")
if err := os.WriteFile(file, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
af, errs := parser.ParseWithOptions(file, src, parser.Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("parse: %v", errs[0])
}
img, err := AssembleFileARM64(af)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
if len(img.Funcs) != 1 {
t.Fatalf("%d functions, want 1", len(img.Funcs))
}
objPath := filepath.Join(t.TempDir(), "oracle.o")
cmd := exec.Command(goBin, "tool", "asm", "-std", "-I", include,
"-p", "flaglisttest", "-o", objPath, file)
cmd.Env = append(os.Environ(), "GOOS=linux", "GOARCH=arm64")
if oout, err := cmd.CombinedOutput(); err != nil {
t.Fatalf("go tool asm: %v\n%s", err, oout)
}
blocks, ok := oracleFuncText(t, mustRead(t, objPath))["f"]
if !ok {
t.Fatal("the oracle output carries no f")
}
if len(blocks) != 1 {
t.Fatalf("%d text blocks for f, want 1", len(blocks))
}
goCode := blocks[0]
fn := img.Funcs[0]
gasmCode := maskCode(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs)
cmpLen := min(len(goCode), len(gasmCode))
if !bytes.Equal(gasmCode[:cmpLen], goCode[:cmpLen]) {
t.Errorf("bytes differ:\ngasm: % x\ngo: % x", gasmCode[:cmpLen], goCode[:cmpLen])
}
if len(goCode) > len(gasmCode) {
for _, b := range goCode[len(gasmCode):] {
if b != 0 {
t.Errorf("non-zero trailing bytes in the oracle output")
break
}
}
}
})
}
}
+8
View File
@@ -63,6 +63,7 @@ const (
const (
kindSTEXT = 1
kindSRODATA = 3
kindSNOPTRDATA = 5
kindSDATA = 7
kindSDWARFFCN = 14
kindSDWARFLINES = 20
@@ -305,9 +306,16 @@ func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, r
if !d.Static {
name = pkgPath + "." + name
}
// RODATA implies no pointers, so it wins over NOPTR: the kind is
// SRODATA either way, exactly as the toolchain chooses it. Plain
// NOPTR data is SNOPTRDATA, which the linker keeps out of the GC's
// type scan; a plain SDATA symbol would demand Go type information
// no assembly file can supply, and the link would fail.
typ := uint8(kindSDATA)
if d.Rodata {
typ = kindSRODATA
} else if d.Noptr {
typ = kindSNOPTRDATA
}
flag := uint8(0)
if d.Dupok {
+3
View File
@@ -45,6 +45,9 @@ func TestReadRuntimeSymbols(t *testing.T) {
// TestResolveExternalSymbols verifies end-to-end resolution of external
// symbol references.
func TestResolveExternalSymbols(t *testing.T) {
if testing.Short() {
t.Skip("resolves through a live go list -export: skipped in -short mode")
}
if _, err := exec.LookPath("go"); err != nil {
t.Skip("go toolchain not available")
}
+143 -1
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// goobjView is a minimal parsed view of a GOOBJ payload, enough to check
@@ -514,6 +514,145 @@ func fieldAfter(line, flag string) string {
return ""
}
// buildLogSteps is what a substitution test needs from a `go build -x -work`
// log: the work directory, the assembler's object, the package archive and
// the link command line.
type buildLogSteps struct {
work string
asmObj string // $WORK expanded
pkgArch string // $WORK expanded
linkLine string // still carries $WORK placeholders
}
// parseBuildLog extracts the build steps from a `go build -x -work` log.
// asmFile names the assembly file whose object the test substitutes. A
// missing step is a failure, not a skip: the toolchain changed shape and the
// substitution would silently test nothing.
func parseBuildLog(t *testing.T, log []byte, asmFile string) buildLogSteps {
t.Helper()
var st buildLogSteps
for line := range strings.SplitSeq(string(log), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
st.work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, asmFile) && !strings.Contains(line, "-gensymabis"):
st.asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"):
rest := strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1])
st.pkgArch = strings.Fields(strings.SplitN(rest, "#", 2)[0])[0]
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
st.linkLine = line
}
}
if st.work == "" || st.asmObj == "" || st.pkgArch == "" || st.linkLine == "" {
t.Fatalf("could not locate the build steps (work=%q asmObj=%q pkgArch=%q link=%q):\n%s",
st.work, st.asmObj, st.pkgArch, st.linkLine, log)
}
st.asmObj = strings.ReplaceAll(st.asmObj, "$WORK", st.work)
st.pkgArch = strings.ReplaceAll(st.pkgArch, "$WORK", st.work)
return st
}
// substituteAndRelink swaps the gasm object into the package archive the
// baseline build produced and re-runs the captured link line against the
// rebuilt archive, writing the binary to outBin (the -x log's link step
// always targets the action graph's internal a.out, which the helper
// redirects; the copy to the -o target is a separate build action the helper
// does not need). The archive handed to the linker is proven to carry the
// gasm object byte for byte, so a build-layout change that skipped the
// substitution fails here instead of passing vacuously.
func substituteAndRelink(t *testing.T, goBin, dir string, st buildLogSteps, outBin string, gasmObj []byte, extraEnv ...string) {
t.Helper()
// The deliberate-run boundary: this path drives a real `go build` and
// cmd/link per invocation, minutes-scale work on the small single-core
// CI runner. Under -short (the push pipeline's mode) it skips; the
// local test gate and the dispatched workflows run it in full.
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
// Extract the archive, overwrite the assembler's member with the gasm
// object and repack (go tool pack has no replace-in-place).
membersDir := filepath.Join(dir, "members")
if err := os.MkdirAll(membersDir, 0o755); err != nil {
t.Fatal(err)
}
extract := exec.Command(goBin, "tool", "pack", "x", st.pkgArch)
extract.Dir = membersDir
if out, err := extract.CombinedOutput(); err != nil {
t.Fatalf("pack x: %v\n%s", err, out)
}
member := filepath.Join(membersDir, filepath.Base(st.asmObj))
if _, err := os.Stat(member); err != nil {
t.Fatalf("the assembler's archive member was not extracted: %v", err)
}
if err := os.Chmod(member, 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(member, gasmObj, 0o644); err != nil {
t.Fatal(err)
}
listCmd := exec.Command(goBin, "tool", "pack", "t", st.pkgArch)
listOut, err := listCmd.CombinedOutput()
if err != nil {
t.Fatalf("pack t: %v\n%s", err, listOut)
}
newArch := filepath.Join(dir, "pkg.a")
args := []string{"tool", "pack", "c", newArch}
seen := map[string]bool{}
for m := range strings.FieldsSeq(string(listOut)) {
if seen[m] {
continue
}
seen[m] = true
if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil {
t.Fatal(err)
}
args = append(args, m)
}
pack := exec.Command(goBin, args...)
pack.Dir = membersDir
if out, err := pack.CombinedOutput(); err != nil {
t.Fatalf("pack c: %v\n%s", err, out)
}
// Prove the substitution: the archive the linker is about to consume
// holds the gasm object, byte for byte.
checkDir := filepath.Join(dir, "check")
if err := os.MkdirAll(checkDir, 0o755); err != nil {
t.Fatal(err)
}
check := exec.Command(goBin, "tool", "pack", "x", newArch)
check.Dir = checkDir
if out, err := check.CombinedOutput(); err != nil {
t.Fatalf("pack x (verification): %v\n%s", err, out)
}
got, err := os.ReadFile(filepath.Join(checkDir, filepath.Base(st.asmObj)))
if err != nil {
t.Fatalf("read the substituted member back: %v", err)
}
if !bytes.Equal(got, gasmObj) {
t.Fatal("the repacked archive does not carry the gasm object")
}
// Re-link. The line carries a GOROOT assignment and $WORK placeholders;
// GOEXPERIMENT must match the toolchain's own, because the linker
// compares the object header against its configuration.
goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output()
linkLine := strings.ReplaceAll(st.linkLine, "$WORK", st.work)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "_pkg_.a"), newArch)
linkLine = strings.ReplaceAll(linkLine, filepath.Join(st.work, "b001", "exe", "a.out"), outBin)
env := append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)))
env = append(env, extraEnv...)
link := exec.Command("sh", "-c", linkLine)
link.Dir = dir
link.Env = env
if out, err := link.CombinedOutput(); err != nil {
t.Fatalf("link with the gasm object: %v\n%s", err, out)
}
}
// TestGOObjectExternalPackageLink is the cross-package end-to-end check: a
// GOOBJ whose code references a real external package symbol (runtime's
// morestack, a plain reference rather than the builtin noctxt form) must
@@ -524,6 +663,9 @@ func fieldAfter(line, flag string) string {
// failed. The binary is not run: morestack returns to the call site's
// stack check, which a hand-written caller has none of.
func TestGOObjectExternalPackageLink(t *testing.T) {
if testing.Short() {
t.Skip("end-to-end go build and link: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
+3
View File
@@ -50,6 +50,8 @@ func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
return relocArm64Branch, 4
case RelArm64LDST64:
return relocArm64LDST64, 8
case RelArm64TLSLE:
return relocArm64TLSLE, 4
default:
return relocArm64Addr, 8
}
@@ -60,6 +62,7 @@ func (img *Image) GOObjectAARCH64(pkgPath, srcPath string) ([]byte, error) {
const (
relocArm64Addr = 3 // R_ADDRARM64, ADRP+ADD pair
relocArm64Branch = 9 // R_CALLARM64, BL instruction
relocArm64TLSLE = 32 // R_ARM64_TLS_LE, MOVZ local-exec TLS load
relocArm64LDST64 = 40 // R_ARM64_PCREL_LDST64, ADRP+LDR/STR pair
)
+5 -1
View File
@@ -30,6 +30,8 @@ func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
return relocRISCVPcrelStype, 8
case RelRISCVJal:
return relocRISCVJal, 4
case RelRISCVTLSLE:
return relocRISCVTLSLE, 8
default:
return relocRISCVPcrelItype, 8
}
@@ -38,11 +40,13 @@ func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) {
// RISC-V relocation types (cmd/internal/objabi). The Go linker applies
// R_RISCV_PCREL_ITYPE/STYPE to an AUIPC + I/S-type instruction pair as a
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction.
// single 8-byte field; R_RISCV_JAL covers a single 4-byte J-type instruction;
// R_RISCV_TLS_LE covers the LUI + I-type pair of a local-exec TLS reference.
const (
relocRISCVJal = 59 // R_RISCV_JAL
relocRISCVPcrelItype = 62 // R_RISCV_PCREL_ITYPE
relocRISCVPcrelStype = 63 // R_RISCV_PCREL_STYPE
relocRISCVTLSLE = 65 // R_RISCV_TLS_LE
)
// toolchainObjectPreambleRISCV returns the "go object ...\n!\n" header
+2 -2
View File
@@ -9,8 +9,8 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The expected bytes are pinned from `go tool asm` output (Go 1.27, amd64,
+303 -39
View File
@@ -90,6 +90,40 @@ var noOperandTable = map[string][]byte{
"LOCK": {0xF0},
"REP": {0xF3},
"REPN": {0xF2},
"ENDBR64": {0xF3, 0x0F, 0x1E, 0xFA},
}
// sysUnaryTable maps the one-operand system instructions to their bytes:
// the prefix, the opcode and the /digit the reg field carries. The operand
// is a register or memory in r/m.
var sysUnaryTable = map[string]struct {
prefix byte
opcode []byte
digit int
}{
"CLWB": {0x66, []byte{0x0F, 0xAE}, 6},
"TPAUSE": {0x66, []byte{0x0F, 0xAE}, 6},
"UMONITOR": {0xF3, []byte{0x0F, 0xAE}, 6},
"UMWAIT": {0xF2, []byte{0x0F, 0xAE}, 6},
"RDPID": {0xF3, []byte{0x0F, 0xC7}, 7},
"CLDEMOTE": {0x00, []byte{0x0F, 0x1C}, 0},
}
// encodeSysUnary emits a one-operand system instruction: the operand in r/m
// under the fixed /digit, no REX.W.
func (e *enc) encodeSysUnary(mnem string, m struct {
prefix byte
opcode []byte
digit int
}, ops []Operand) error {
if len(ops) != 1 {
return fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
i := &instr{prefix: m.prefix, opcode: m.opcode, modrm: -1, sib: -1}
if err := setRMDigit(i, m.digit, ops[0], 8); err != nil {
return err
}
return e.emit(i)
}
// --- MOV --------------------------------------------------------------------
@@ -111,6 +145,102 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
// silently emit REX.W 8B with the wrong operand meaning.
_, srcVec := vecReg(src)
dstReg, dstVec := vecReg(dst)
// Control and debug register moves: 0F 20 (CRn→r64), 0F 22 (r64→CRn),
// 0F 21 (DRn→r64) and 0F 23 (r64→DRn). The CR/DR number rides the reg
// field, the general register r/m; CR8+/DR8+ take REX.R.
if c, ok := src.(Reg); ok && c.ctl != 0 {
g, ok := dst.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register load needs a general register destination")
}
opc := byte(0x20)
if c.ctl == 2 {
opc = 0x21
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
if c, ok := dst.(Reg); ok && c.ctl != 0 {
g, ok := src.(Reg)
if !ok || g.isVec() || g.ctl != 0 {
return fmt.Errorf("MOV: control/debug register store needs a general register source")
}
opc := byte(0x22)
if c.ctl == 2 {
opc = 0x23
}
return e.emit(&instr{
opcode: []byte{0x0F, opc},
modrm: 0xC0 | (c.idx&7)<<3 | (g.idx & 7),
sib: -1, rexR: c.idx >= 8, rexB: g.idx >= 8,
})
}
// MMX register moves: MOVQ M0, mem and MOVQ mem, M0 are the MMX
// load/store pair 0F 6F/0F 7F (no prefix); a register pair takes the
// load opcode. The GPR crossings ride the MOVD opcodes with REX.W
// (0F 6E into the bank, 0F 7E out), and an XMM source crosses into the
// bank through the F2 0F D6 move. The XMM MOVQ forms follow below.
if m, ok := src.(Reg); ok && m.mmx {
switch d := dst.(type) {
case Reg:
if !d.mmx && (d.isVec() || d.fp) {
return fmt.Errorf("MOV: MMX register moves stay inside the M bank")
}
if !d.mmx {
// M → GPR: 0F 7E with REX.W, the bank in reg and the GPR in
// r/m, the store layout the toolchain picks.
i := &instr{opcode: []byte{0x0F, 0x7E}, modrm: -1, sib: -1, rexW: true}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, d, src, 8); err != nil {
return err
}
return e.emit(i)
case Mem:
i := &instr{opcode: []byte{0x0F, 0x7F}, modrm: -1, sib: -1}
if err := setRM(i, m, d, 8); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("MOV: invalid MMX destination")
}
if m, ok := dst.(Reg); ok && m.mmx {
switch src.(type) {
case Mem, sbMem:
i := &instr{opcode: []byte{0x0F, 0x6F}, modrm: -1, sib: -1}
if err := setRM(i, m, src, 8); err != nil {
return err
}
return e.emit(i)
case Reg:
if g := src.(Reg); g.isVec() {
// X → M: F2 0F D6, the bank in reg, the XMM source in r/m.
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xD6}, modrm: -1, sib: -1}
if err := setRM(i, m, src, 8); err != nil {
return err
}
return e.emit(i)
}
// GPR → M: 0F 6E with REX.W, the bank in reg, the GPR in r/m.
i := &instr{opcode: []byte{0x0F, 0x6E}, modrm: -1, sib: -1, rexW: true}
if err := setRM(i, m, src, 8); err != nil {
return err
}
return e.emit(i)
}
return fmt.Errorf("MOV: MMX load takes a register or memory source")
}
if srcVec || dstVec {
if dstVec {
if g, ok := src.(Reg); ok && !g.isVec() {
@@ -219,14 +349,14 @@ func (e *enc) encodeMov(ops []Operand, size int) error {
case Imm:
if dstIsReg {
v := int64(src)
// The Go assembler compresses 64-bit moves whose immediate fits
// a signed int32, choosing per sign:
// The Go assembler compresses 64-bit moves whose immediate
// fits the zero-extending 32-bit span, choosing per sign:
// v >= 0: B8+rd imm32 without REX.W (zero-extended by the
// hardware, REX.B still emitted for R8-R15);
// v < 0: REX.W C7 /0 imm32 (sign-extended, the plain B8+rd
// form would zero-extend and corrupt the value).
// Out-of-range immediates keep the B8+rd imm64 form.
if size == 8 && v >= 0 && v <= (1<<31)-1 {
if size == 8 && v >= 0 && v <= (1<<32)-1 {
i := newInstr(4, []byte{0xB8 + byte(dstReg.idx&7)})
i.rexB = dstReg.idx >= 8
i.imm = le32(v)
@@ -399,8 +529,10 @@ func (e *enc) encodeALUImm(digit int, dst Operand, imm int64, size int) error {
return err
}
// The byte accumulator short form (0x04+digit*8, no ModR/M) when
// the destination is AL, the form the Go assembler prefers here.
if r, ok := dst.(Reg); ok && r.idx == 0 {
// the destination is spelled AL itself; the size-agnostic AX takes
// the generic 0x80 /digit form, the division go tool asm makes
// (ADDB $7, AL is 04 07 while ADDB $3, AX is 80 c0 03).
if r, ok := dst.(Reg); ok && r.idx == 0 && byteReg(r) {
i := &instr{opcode: []byte{byte(0x04 + digit*8)}, modrm: -1, sib: -1}
i.imm = immBytes
return e.emit(i)
@@ -457,8 +589,11 @@ func (e *enc) encodeTest(ops []Operand, size int) error {
if imm, ok := src.(Imm); ok {
// TEST r/m, imm: 0xF6 (8-bit) / 0xF7 /0, but the Go assembler
// always uses the accumulator forms (A8/A9, no ModR/M) when the
// register operand is AL/AX, whatever the immediate's width.
if r, ok := dst.(Reg); ok && r.idx == 0 {
// register operand is AL/AX, whatever the immediate's width. At
// byte width the short form belongs to the AL spelling alone; the
// size-agnostic AX takes the generic F6 /0 (TESTB $7, AX is
// f6 c0 07), the same division the ALU accumulator makes.
if r, ok := dst.(Reg); ok && r.idx == 0 && (size != 1 || byteReg(r)) {
op := byte(0xA9)
if size == 1 {
op = 0xA8
@@ -518,6 +653,14 @@ func (e *enc) encodeLea(ops []Operand, size int) error {
default:
return fmt.Errorf("LEA: source must be a memory operand")
}
// LEA accepts the full unsigned 32-bit displacement span where the
// loads and stores reject it beyond the signed one; the wide values
// ride the same disp32 bytes as their two's-complement bit pattern.
if m, ok := src.(Mem); ok && m.Disp >= 1<<31 && m.Disp <= (1<<32)-1 {
c := m
c.Disp = int64(int32(uint32(m.Disp)))
src = c
}
i := newInstr(size, []byte{0x8D})
if err := setRM(i, dstReg, src, size); err != nil {
return err
@@ -653,7 +796,7 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
}
i.imm = []byte{byte(imm)}
}
if err := setRMReg(i, srcReg.idx, srcReg.idx >= 8, false, dst, size); err != nil {
if err := setRMReg(i, srcReg.idx&7, srcReg.idx >= 8, false, dst, size); err != nil {
return err
}
return e.emit(i)
@@ -663,6 +806,18 @@ func (e *enc) encodeDoubleShift(base string, ops []Operand, size int) error {
func (e *enc) encodeImul(ops []Operand, size int) error {
switch len(ops) {
case 1:
// The one-operand form, IMUL r/m: F6/F7 /5 with AL/AX/EAX/RAX as the
// implied destination (the toolchain's one-register shape).
opc := byte(0xF7)
if size == 1 {
opc = 0xF6
}
i := newInstr(size, []byte{opc})
if err := setRMDigit(i, 5, ops[0], size); err != nil {
return err
}
return e.emit(i)
case 2:
// Two shapes. The leading-immediate spelling IMUL $imm, r multiplies
// r in place (dst = rm = r): the shape GOROOT's clock code writes.
@@ -697,7 +852,7 @@ func (e *enc) encodeImul(ops []Operand, size int) error {
// r/m operand (setRM takes registers and memory alike).
return e.encodeImulImm(imm, ops[1], dstReg, size)
}
return fmt.Errorf("IMUL expects 2 or 3 operands, got %d", len(ops))
return fmt.Errorf("IMUL expects 1, 2 or 3 operands, got %d", len(ops))
}
// encodeImulImm emits the immediate multiply: 0x6B with a sign-extended imm8
@@ -742,6 +897,27 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
w16 := size == 2
switch op := ops[0].(type) {
case Reg:
// Segment registers: FS and GS carry their own one-byte opcodes
// under 0F (A0/A8 push, A1/A9 pop); the other four spellings are
// not pushable in 64-bit mode.
if n, isSeg := op.segNumber(); isSeg {
switch n {
case 4: // FS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA0}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA1}, modrm: -1, sib: -1})
case 5: // GS
if push {
return e.emit(&instr{opcode: []byte{0x0F, 0xA8}, modrm: -1, sib: -1})
}
return e.emit(&instr{opcode: []byte{0x0F, 0xA9}, modrm: -1, sib: -1})
}
return fmt.Errorf("PUSH/POP: only FS and GS are encodable in 64-bit mode")
}
if op.mmx || op.isVec() || op.fp || op.ctl != 0 {
return fmt.Errorf("PUSH/POP: invalid register operand")
}
base := byte(0x50) // PUSH r; POP is 0x58
if !push {
base = 0x58
@@ -772,8 +948,12 @@ func (e *enc) encodePushPop(ops []Operand, size int, push bool) error {
}
// PUSH imm32, sign-extended to 64 bits; go tool asm bounds the
// immediate by the same signed/unsigned 32-bit span as every other
// scalar immediate.
immBytes, err := immediate(int64(op), 8, false)
// scalar immediate. The W spelling takes imm16 alone.
immWidth := 8
if w16 {
immWidth = 2
}
immBytes, err := immediate(int64(op), immWidth, false)
if err != nil {
return err
}
@@ -893,29 +1073,46 @@ func immediate(v int64, size int, full64 bool) ([]byte, error) {
// --- CMOVcc / SETcc ---------------------------------------------------------
// cmovCondition splits a CMOVcc suffix into an optional size letter and the
// condition code. Two vocabularies meet here: the Plan 9 spellings prefix
// the condition with a size letter (CMOVLGT, CMOVQEQ), while the toolchain's
// renderer prints the condition alone (CMOVLE, CMOVG) and leaves the width
// to the operand registers. A suffix that is itself a condition name reads
// as that condition, so the renderer's text re-encodes; CMOVBGT, CMOVWXX and
// the bare CMOV still find no condition and stay rejected.
func cmovCondition(rest string) (size int, cc int, ok bool) {
if cc, ok := jccMap[rest]; ok {
return 0, cc, true
}
if len(rest) >= 2 {
switch rest[0] {
case 'W':
if cc, ok := jccMap[rest[1:]]; ok {
return 2, cc, true
}
case 'L':
if cc, ok := jccMap[rest[1:]]; ok {
return 4, cc, true
}
case 'Q':
if cc, ok := jccMap[rest[1:]]; ok {
return 8, cc, true
}
}
}
return 0, 0, false
}
// encodeCmov encodes a conditional move: CMOV + size (W/L/Q) + condition
// (CMOVLGT, CMOVQEQ, …). The condition reads exactly like the Jcc spellings;
// the instruction is 0F 40+cc with reg = dst, rm = src.
// (CMOVLGT, CMOVQEQ, …), or the renderer's condition alone (CMOVLE) with the
// width taken from the destination register. The instruction is 0F 40+cc
// with reg = dst, rm = src.
func (e *enc) encodeCmov(upper string, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("CMOVcc expects 2 operands, got %d", len(ops))
}
rest := upper[len("CMOV"):]
if len(rest) < 2 {
return fmt.Errorf("unsupported instruction %q", upper)
}
var size int
switch rest[0] {
case 'W':
size = 2
case 'L':
size = 4
case 'Q':
size = 8
default:
return fmt.Errorf("unsupported instruction %q", upper)
}
cc, ok := jccMap[rest[1:]]
size, cc, ok := cmovCondition(rest)
if !ok {
return fmt.Errorf("unsupported instruction %q", upper)
}
@@ -924,6 +1121,14 @@ func (e *enc) encodeCmov(upper string, ops []Operand) error {
if !ok {
return fmt.Errorf("CMOVcc destination must be a register")
}
if size == 0 {
// The renderer's spelling carries no size letter: the width rides
// the destination register's own size class.
size = dstReg.size
if size != 1 && size != 2 && size != 4 && size != 8 {
return fmt.Errorf("CMOVcc destination must be a general register")
}
}
i := newInstr(size, []byte{0x0F, byte(0x40 + cc)})
if err := setRM(i, dstReg, src, size); err != nil {
return err
@@ -1093,6 +1298,38 @@ var sseMoveTable = map[string]sseMove{
"MOVSS": {0xF3, 0x10, 0x11}, // scalar single
}
// sseStoreOnly holds the store-only SSE forms, OP xmm, mem: the XMM register
// rides the reg field and memory r/m (the non-temporal store).
var sseStoreOnly = map[string]struct {
prefix byte
op byte
}{
"MOVNTDQ": {0x66, 0xE7},
}
// encodeSSEStoreOnly encodes OP xmm, mem (reg = the XMM source, r/m = the
// destination memory).
func (e *enc) encodeSSEStoreOnly(mnem string, m struct {
prefix byte
op byte
}, ops []Operand) error {
if len(ops) != 2 {
return fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
srcReg, ok := ops[0].(Reg)
if !ok || !srcReg.isVec() {
return fmt.Errorf("%s source must be a vector register", mnem)
}
if !isX86Mem(ops[1]) {
return fmt.Errorf("%s destination must be a memory operand", mnem)
}
i := &instr{prefix: m.prefix, opcode: []byte{0x0F, m.op}, modrm: -1, sib: -1}
if err := setRM(i, srcReg, ops[1], 8); err != nil {
return err
}
return e.emit(i)
}
// encodeSSEMove encodes a legacy SSE move: a vector-to-vector move uses the
// load form (reg = destination), matching the Go assembler.
func (e *enc) encodeSSEMove(m sseMove, ops []Operand) error {
@@ -1229,13 +1466,15 @@ type sseExtract struct {
op []byte
opMem []byte // used when the destination is memory; nil shares op
rexW bool // PEXTRQ's REX.W
rev bool // PEXTRW's GPR form swaps the fields: the register
// destination rides reg and the XMM source r/m
}
var sseExtractTable = map[string]sseExtract{
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false},
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false},
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true},
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false},
"PEXTRB": {[]byte{0x0F, 0x3A, 0x14}, nil, false, false},
"PEXTRD": {[]byte{0x0F, 0x3A, 0x16}, nil, false, false},
"PEXTRQ": {[]byte{0x0F, 0x3A, 0x16}, nil, true, false},
"PEXTRW": {[]byte{0x0F, 0xC5}, []byte{0x0F, 0x3A, 0x15}, false, true},
}
// sseInsert describes a lane insert: OP $imm, src, xdst with reg = the XMM
@@ -1308,14 +1547,23 @@ func (e *enc) encodeSSEBin(m sseBin, ops []Operand) error {
}
src, dst := ops[0], ops[1]
dstReg, ok := dst.(Reg)
if !ok || !dstReg.isVec() {
if !ok || (!dstReg.isVec() && !dstReg.mmx) {
return fmt.Errorf("SSE binary destination must be a vector register")
}
// The MMX twins of the packed-integer SSE2 ops drop the 0x66 prefix:
// PADDD M2, M1 is 0F FE where the XMM form is 66 0F FE.
prefix := m.prefix
if dstReg.mmx {
if prefix != 0x66 {
return fmt.Errorf("SSE binary: this form takes no MMX register operand")
}
prefix = 0
}
opcode := []byte{0x0F, m.op}
if m.map38 {
opcode = []byte{0x0F, 0x38, m.op}
}
i := &instr{prefix: m.prefix, opcode: opcode, modrm: -1, sib: -1}
i := &instr{prefix: prefix, opcode: opcode, modrm: -1, sib: -1}
if err := setRM(i, dstReg, src, 8); err != nil {
return err
}
@@ -1713,8 +1961,17 @@ func (e *enc) encodeSSEExtract(m sseExtract, ops []Operand) error {
if m.opMem != nil && memOperand(ops[2]) {
opcode = m.opMem
}
reg, rm := srcReg, ops[2]
if m.rev && opcode[1] == 0xC5 {
// the 0F C5 layout: the GPR destination rides reg, the XMM source r/m
if d, ok := ops[2].(Reg); !ok {
return fmt.Errorf("PEXTRW destination must be a register or memory")
} else {
reg, rm = d, ops[1]
}
}
i := &instr{prefix: 0x66, opcode: opcode, modrm: -1, sib: -1, rexW: m.rexW}
if err := setRM(i, srcReg, ops[2], 8); err != nil {
if err := setRM(i, reg, rm, 8); err != nil {
return err
}
i.imm = []byte{immByte}
@@ -1791,12 +2048,19 @@ func (e *enc) encodeSSEShift(name string, ops []Operand) error {
// immediate LAST in Plan 9 order (src, dst, $imm), unlike the shuffle family:
// F2 0F C2 with reg = dst, rm = src.
func (e *enc) encodeCmpsd(ops []Operand) error {
return e.encodeSSECmp("CMPSD", 0xF2, ops)
}
// encodeSSECmp encodes the SSE compare family (CMPSD/CMPSS/CMPPS/CMPPD):
// 0F C2 /r ib with the predicate immediate last in Plan 9 order
// (src, dst, $imm) and the packed forms' prefixes.
func (e *enc) encodeSSECmp(mnem string, prefix byte, ops []Operand) error {
if len(ops) != 3 {
return fmt.Errorf("CMPSD expects 3 operands (src, dst, $imm), got %d", len(ops))
return fmt.Errorf("%s expects 3 operands (src, dst, $imm), got %d", mnem, len(ops))
}
imm, ok := ops[2].(Imm)
if !ok {
return fmt.Errorf("CMPSD predicate must be an immediate")
return fmt.Errorf("%s predicate must be an immediate", mnem)
}
immByte, err := imm8(int64(imm))
if err != nil {
@@ -1804,9 +2068,9 @@ func (e *enc) encodeCmpsd(ops []Operand) error {
}
dstReg, ok2 := ops[1].(Reg)
if !ok2 || !dstReg.isVec() {
return fmt.Errorf("CMPSD destination must be a vector register")
return fmt.Errorf("%s destination must be a vector register", mnem)
}
i := &instr{prefix: 0xF2, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
i := &instr{prefix: prefix, opcode: []byte{0x0F, 0xC2}, modrm: -1, sib: -1}
if err := setRM(i, dstReg, ops[0], 8); err != nil {
return err
}
+3 -3
View File
@@ -16,7 +16,7 @@ import (
"golang.org/x/arch/x86/x86asm"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestAssembleGoFlacAVX2Kernel assembles the whole production AVX2 kernel;
@@ -25,7 +25,7 @@ import (
func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx2_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
t.Skip("go-libraries repository not present next to gasm-sdk")
}
src, err := os.ReadFile(path)
if err != nil {
@@ -86,7 +86,7 @@ func TestAssembleGoFlacAVX2Kernel(t *testing.T) {
func TestAssembleGoFlacAVX512Kernel(t *testing.T) {
path := "../../go-libraries/go-flac/avx512_amd64.s"
if _, err := os.Stat(path); err != nil {
t.Skip("go-libraries repository not present next to gasm-devkit")
t.Skip("go-libraries repository not present next to gasm-sdk")
}
src, err := os.ReadFile(path)
if err != nil {
+11 -2
View File
@@ -13,7 +13,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The differential kernels for the DATA-path and front-end gaps are kept in
@@ -23,9 +23,18 @@ import (
// bytes must agree with the relocation sites masked on both sides.
// toolAsmObject assembles path with the installed toolchain's assembler for
// goarch ("" = the host) and returns the object bytes.
// goarch ("" = the host) and returns the object bytes. Every live-oracle
// comparison funnels through here, so this is also where the deliberate-run
// boundary sits: under -short (the push pipeline's mode) the comparisons
// skip, because each spawns a go tool asm subprocess and the small single-
// core runner pays seconds per spawn. The encodings stay pinned by the
// golden-byte tests in every mode; the live oracle runs in the local test
// gate and the dispatched workflows.
func toolAsmObject(t *testing.T, path, goarch string) []byte {
t.Helper()
if testing.Short() {
t.Skip("live go tool asm oracle: skipped in -short mode")
}
goBin, err := exec.LookPath("go")
if err != nil {
t.Skip("no Go toolchain available")
+1 -1
View File
@@ -12,7 +12,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks:
+132 -9
View File
@@ -9,7 +9,7 @@ import (
"sort"
"strconv"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Image is an assembled file: the function bodies laid out in source order,
@@ -98,6 +98,7 @@ const (
RelRISCVPCRELIType // R_RISCV_PCREL_ITYPE (AUIPC + I-type pair)
RelRISCVPCRELSType // R_RISCV_PCREL_STYPE (AUIPC + S-type pair)
RelRISCVJal // R_RISCV_JAL (J-type call)
RelRISCVTLSLE // R_RISCV_TLS_LE (LUI + I-type local-exec pair)
RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i)
RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st)
RelArm64Addr // R_ADDRARM64 (ADRP + ADD pair)
@@ -105,6 +106,7 @@ const (
RelArm64LDST64 // R_ARM64_PCREL_LDST64 (ADRP + 64-bit LDR/STR pair)
RelLoong64Branch // R_CALLLOONG64 (BL instruction)
RelAddr // R_ADDR: the absolute address of a symbol held in a DATA field
RelArm64TLSLE // R_ARM64_TLS_LE (MOVZ local-exec TLS load)
)
type Reloc struct {
@@ -134,7 +136,8 @@ type DataSymbol struct {
Offset int // byte offset within Data
Size int
Static bool // the <> marker: file-local, not exported
Rodata bool // the RODATA flag: read-only data
Rodata bool // the RODATA flag: read-only data (implies no pointers)
Noptr bool // the NOPTR flag: data with no pointers, kept out of GC scanning
Dupok bool // the DUPOK flag: duplicate-OK
// Relocs carries the symbol-valued DATA initialisers ("DATA s+0(SB)/8,
// $other(SB)"): fields of this symbol's data that hold another symbol's
@@ -269,6 +272,7 @@ func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
Size: len(d.buf),
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
img.Data = append(img.Data, d.buf...)
@@ -333,6 +337,35 @@ func AssembleFile(f *ast.File, opts ...AssembleOption) (*Image, error) {
return img, nil
}
// riscvTLSSymbols names the symbols the file declares with the TLSBSS flag
// (or its legacy numeric constant 256 from textflag.h): the assembler gives
// their SB references the local-exec TLS sequence.
func riscvTLSSymbols(f *ast.File) map[string]bool {
var tls map[string]bool
for _, d := range f.Decls {
gd, ok := d.(*ast.Globl)
if !ok || gd.Name == nil || gd.Name.Pseudo != "SB" {
continue
}
for _, fl := range gd.Flags {
isTLS := fl == "TLSBSS"
if !isTLS {
if n, err := strconv.Atoi(fl); err == nil && n&256 != 0 {
isTLS = true
}
}
if isTLS {
if tls == nil {
tls = map[string]bool{}
}
tls[gd.Name.Name] = true
break
}
}
}
return tls
}
// AssembleFileRISCV assembles every TEXT function of a parsed RISC-V file
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
// code. SB references in the code are encoded as AUIPC pairs with zero
@@ -342,6 +375,10 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if err != nil {
return nil, err
}
// The symbols the file declares TLSBSS resolve through the local-exec
// sequence (LUI + ADDIW + ADD of TP), exactly as the toolchain routes
// every SB reference whose symbol carries the STLSBSS type.
tlsSyms := riscvTLSSymbols(f)
// The pooled $i64 constants the wide MOV immediate loads refer to join
// the declared data as read-only symbols, deduplicated across the file
// (the toolchain synthesises the same symbols into its rodata).
@@ -353,7 +390,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
if !ok {
continue
}
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t)
code, labels, relocs, lines, spadj, lits, err := assembleRISCV(t, tlsSyms)
if err != nil {
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
}
@@ -413,6 +450,7 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
@@ -485,6 +523,7 @@ func AssembleFileLOONG64(f *ast.File) (*Image, error) {
Size: d.size,
Static: d.static,
Rodata: d.rodata,
Noptr: d.noptr,
Dupok: d.dupok,
})
}
@@ -546,19 +585,91 @@ type dataSym struct {
size int
static bool
rodata bool
noptr bool
dupok bool
// relocs are the symbol-valued DATA fields, in declaration order; Off
// is relative to the symbol's data start.
relocs []Reloc
}
// checkDuplicateDecls rejects a symbol the file declares twice, the
// toolchain's OnList rule (cmd/internal/obj, InitTextSym and GloblPos,
// measured with go tool asm): the second declaration of a symbol is
// diagnosed as a redeclaration whether the pair is two TEXTs, two GLOBLs
// or one of each, and a second TEXT carries the other declaration's line
// the way the toolchain's note does. The DUPOK flag plays no part at
// assembly time: it legalises duplicate definitions across files
// (AttrDuplicateOK, which the linker resolves), never two declarations
// inside one file, so a DUPOK pair here is rejected exactly like a plain
// one. Symbol identity follows the object model: the package prefix and
// the <> static marker are part of the name (foo(SB), foo<>(SB) and
// other·foo(SB) are three symbols); the ABI selector is not, because
// outside the runtime package, where alone it is legal, foo<ABIInternal>(SB)
// resolves to foo(SB) and is a redeclaration.
func checkDuplicateDecls(f *ast.File) error {
type decl struct {
text bool
line int
}
seen := make(map[string]decl)
symKey := func(s *ast.Symbol) string {
key := s.Pkg + "\x00" + s.Name
if s.Static {
key += "\x00<>"
}
return key
}
for _, d := range f.Decls {
switch t := d.(type) {
case *ast.Text:
if first, dup := seen[symKey(t.Name)]; dup {
if first.text {
return fmt.Errorf("symbol %q redeclared (other declaration on line %d)", t.Name.Name, first.line)
}
return fmt.Errorf("symbol %q redeclared", t.Name.Name)
}
seen[symKey(t.Name)] = decl{text: true, line: t.Pos().Line}
case *ast.Globl:
if t.Name == nil || t.Name.Pseudo != "SB" {
continue
}
if _, dup := seen[symKey(t.Name)]; dup {
return fmt.Errorf("symbol %q redeclared", t.Name.Name)
}
seen[symKey(t.Name)] = decl{line: t.Pos().Line}
}
}
return nil
}
// The data section's size ceilings. The image materialises every GLOBL's
// bytes at assembly time, where the toolchain defers the cost to its linker,
// so gasm needs its own bound and a diagnostic in place of an allocation
// failure. The toolchain's own limit (obj's "symbol too large", 2,000,000,000
// bytes) would let a twenty-five byte input demand four gigabytes of resident
// memory (measured: `GLOBL d(SB), $2000000000` peaks at 3.92 GiB RSS), which
// starves the memory fence the test recipes run under when the fuzz workers
// share it. 64 MiB per symbol and 128 MiB per file sit two orders above any
// real assembly symbol (the runtime's largest GLOBL is KiB-scale) and keep a
// worker's worst-case peak near its fence share.
const (
maxSymbolSize = 64 << 20
maxDataSize = 128 << 20
)
// collectData gathers the file's static symbols (GLOBL) and their initial
// contents (DATA) into byte buffers. Two passes: the Plan 9 convention puts
// every DATA line before its symbol's GLOBL, so the symbols are registered
// before the initialisers are applied.
// before the initialisers are applied. Every per-architecture entry point
// collects data before assembling text, so the duplicate-declaration check
// rides here and covers them all.
func collectData(f *ast.File) ([]dataSym, error) {
if err := checkDuplicateDecls(f); err != nil {
return nil, err
}
index := map[string]int{}
var syms []dataSym
total := 0
for _, d := range f.Decls {
gd, ok := d.(*ast.Globl)
if !ok {
@@ -568,12 +679,19 @@ func collectData(f *ast.File) ([]dataSym, error) {
continue
}
name := gd.Name.Name
if _, dup := index[name]; dup {
return nil, fmt.Errorf("duplicate GLOBL %q", name)
}
size := 0
if gd.Size != nil && gd.Size.Imm.HasVal {
size = int(gd.Size.Imm.Val)
if gd.Size.Imm.Neg {
size = -size
}
if size < 0 || size > maxSymbolSize {
return nil, fmt.Errorf("GLOBL %q: symbol too large (%d bytes > %d bytes)", name, size, maxSymbolSize)
}
if total+size > maxDataSize {
return nil, fmt.Errorf("GLOBL %q: data section too large (%d bytes > %d bytes)", name, total+size, maxDataSize)
}
total += size
}
index[name] = len(syms)
ds := dataSym{
@@ -587,12 +705,14 @@ func collectData(f *ast.File) ([]dataSym, error) {
switch f {
case "RODATA":
ds.rodata = true
case "NOPTR":
ds.noptr = true
case "DUPOK":
ds.dupok = true
default:
// Legacy numeric flag constants (runtime/textflag.h):
// DUPOK is 2, RODATA is 8; combinations arrive as one
// number (e.g. 10 = RODATA|DUPOK).
// DUPOK is 2, RODATA is 8, NOPTR is 16; combinations arrive
// as one number (e.g. 10 = RODATA|DUPOK).
if n, err := strconv.Atoi(f); err == nil {
if n&2 != 0 {
ds.dupok = true
@@ -600,6 +720,9 @@ func collectData(f *ast.File) ([]dataSym, error) {
if n&8 != 0 {
ds.rodata = true
}
if n&16 != 0 {
ds.noptr = true
}
}
}
}
+110 -38
View File
@@ -12,9 +12,107 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestDuplicateDecls pins the duplicate-declaration rule against the
// toolchain's, measured with go tool asm (Go 1.27.1): a symbol declared
// twice in one file is rejected at the second declaration whether the pair
// is two TEXTs, two GLOBLs or one of each, and DUPOK never legalises it
// within a file (the toolchain's duperror.s testdata asserts the plain
// pair). DUPOK legalises duplicates across files, so each file of a pair
// assembles cleanly on its own, and symbol identity keeps the package
// prefix and the <> static marker: foo(SB), foo<>(SB) and other·foo(SB)
// are three symbols, not one. Every case runs through all four
// per-architecture entry points.
func TestDuplicateDecls(t *testing.T) {
targets := []struct {
goarch string
file string
asm func(*ast.File) (*Image, error)
}{
{"amd64", "dup_amd64.s", func(f *ast.File) (*Image, error) { return AssembleFile(f) }},
{"arm64", "dup_arm64.s", AssembleFileARM64},
{"riscv64", "dup_riscv64.s", AssembleFileRISCV},
{"loong64", "dup_loong64.s", AssembleFileLOONG64},
}
cases := []struct {
name string
src string
want string // substring of the error, "" for no error
}{
{
"plain TEXT duplicate rejected",
"TEXT foo(SB), NOSPLIT, $0\n\tRET\nTEXT foo(SB), NOSPLIT, $0\n\tRET\n",
`symbol "foo" redeclared (other declaration on line 1)`,
},
{
"DUPOK pair in one file rejected",
"TEXT foo(SB), DUPOK, $0\n\tRET\nTEXT foo(SB), DUPOK, $0\n\tRET\n",
`symbol "foo" redeclared (other declaration on line 1)`,
},
{
"TEXT then GLOBL rejected",
"TEXT foo(SB), NOSPLIT, $0\n\tRET\nGLOBL foo(SB), NOPTR, $8\n",
`symbol "foo" redeclared`,
},
{
"GLOBL then TEXT rejected",
"GLOBL foo(SB), NOPTR, $8\nTEXT foo(SB), NOSPLIT, $0\n\tRET\n",
`symbol "foo" redeclared`,
},
{
"plain GLOBL duplicate rejected",
"GLOBL bar(SB), NOPTR, $8\nGLOBL bar(SB), NOPTR, $8\n",
`symbol "bar" redeclared`,
},
{
"static and package markers separate symbols",
"TEXT foo(SB), NOSPLIT, $0\n\tRET\nTEXT foo<>(SB), NOSPLIT, $0\n\tRET\nTEXT other\u00b7foo(SB), NOSPLIT, $0\n\tRET\n",
"",
},
{
"single DUPOK TEXT assembles",
"TEXT foo(SB), DUPOK, $0\n\tRET\n",
"",
},
}
for _, tg := range targets {
for _, c := range cases {
f, errs := parser.Parse(tg.file, c.src)
if len(errs) > 0 {
t.Fatalf("%s/%s: parse: %v", tg.goarch, c.name, errs)
}
_, err := tg.asm(f)
if c.want == "" {
if err != nil {
t.Errorf("%s/%s: error %v, want none", tg.goarch, c.name, err)
}
continue
}
if err == nil || !strings.Contains(err.Error(), c.want) {
t.Errorf("%s/%s: error %v, want substring %q", tg.goarch, c.name, err, c.want)
}
}
}
// A DUPOK pair across files is the case DUPOK exists for: the linker
// resolves it, the assembler never sees both sides, so each file of the
// pair assembles cleanly on its own.
for _, tg := range targets {
for _, name := range []string{"first", "second"} {
f, errs := parser.Parse(tg.file, "TEXT foo(SB), DUPOK, $0\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("%s/dupok cross-file %s: parse: %v", tg.goarch, name, errs)
}
if _, err := tg.asm(f); err != nil {
t.Errorf("%s/dupok cross-file %s: error %v, want none", tg.goarch, name, err)
}
}
}
}
// TestAssembleFileStaticData checks the whole-image layout; code, padding
// and the data section; and that the RIP-relative displacements of static
// symbol loads resolve to the right bytes.
@@ -60,23 +158,15 @@ DATA small<>+0(SB)/4, $0x1234
}
}
// TestAssembleFileErrors checks the static-symbol error paths.
// TestAssembleFileErrors checks the static-symbol error paths. A reference
// to a static symbol no GLOBL defines defers to the linker exactly as the
// toolchain does (an external relocation), so it is not an error here.
func TestAssembleFileErrors(t *testing.T) {
cases := []struct {
name string
src string
want string // substring of the error
}{
{
"undefined symbol",
`
#include "textflag.h"
TEXT ·f(SB), NOSPLIT, $0
VMOVDQU nope<>(SB), X0
RET
`,
"undefined symbol",
},
{
"DATA without GLOBL",
`
@@ -369,23 +459,8 @@ func main() {
if err != nil {
t.Fatalf("baseline build: %v\n%s", err, buildLog)
}
var work, linkLine, asmObj string
for line := range strings.SplitSeq(string(buildLog), "\n") {
switch {
case strings.HasPrefix(line, "WORK="):
work = strings.TrimPrefix(line, "WORK=")
case strings.Contains(line, "/asm ") && strings.Contains(line, "main_amd64.s") && !strings.Contains(line, "-gensymabis"):
asmObj = fieldAfter(line, "-o")
case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"):
linkLine = line
}
}
if work == "" || asmObj == "" || linkLine == "" {
t.Skipf("could not parse build log (work=%q asmObj=%q link=%q)", work, asmObj, linkLine)
}
defer os.RemoveAll(work)
asmObj = strings.ReplaceAll(asmObj, "$WORK", work)
linkLine = strings.ReplaceAll(linkLine, "$WORK", work)
st := parseBuildLog(t, buildLog, "main_amd64.s")
defer os.RemoveAll(st.work)
// Assemble the same source with gasm and substitute the object.
src, err := os.ReadFile(filepath.Join(dir, "main_amd64.s"))
@@ -400,21 +475,18 @@ func main() {
if err != nil {
t.Fatalf("AssembleFile: %v", err)
}
gasmObj, err := img.GOObject("dlink", "main_amd64.s")
// The package path is "main": the linker resolves the Go code's
// references against main.<name>, so the object must define the symbols
// under that prefix whatever the module is called.
gasmObj, err := img.GOObject("main", "main_amd64.s")
if err != nil {
t.Fatalf("GOObject: %v", err)
}
if err := os.WriteFile(asmObj, gasmObj, 0o644); err != nil {
t.Fatalf("write gasm object: %v", err)
}
linkCmd := exec.Command("bash", "-c", "cd "+dir+" && "+linkLine)
if out, err := linkCmd.CombinedOutput(); err != nil {
t.Fatalf("re-link with gasm object: %v\n%s", err, out)
}
substituteAndRelink(t, goBin, dir, st, filepath.Join(dir, "prog2"), gasmObj)
// The linked program must run and find the right function behind the
// data word.
out, err := exec.Command(filepath.Join(dir, "prog")).CombinedOutput()
out, err := exec.Command(filepath.Join(dir, "prog2")).CombinedOutput()
if err != nil {
t.Fatalf("linked program failed: %v\n%s", err, out)
}
+445 -30
View File
@@ -9,7 +9,7 @@ import (
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
@@ -360,6 +360,59 @@ func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) {
return mnem, false
}
// l64PairRegister reports whether name names a register the pair sugar may
// carry: the general, FP, FCC and FCSR spellings loong64RegNum resolves, or
// a bare LSX/LASX vector register (V0-V31, X0-X31), which lives outside
// that table and parses only through the vector reader.
func l64PairRegister(name string) bool {
if loong64RegNum(name) >= 0 {
return true
}
v, ok := l64ParseVecOperand(&ast.Operand{Kind: ast.OpAddr, Raw: name})
return ok && !v.hasSuf
}
// l64ExpandPairs rewrites the toolchain's register-pair spellings into the
// operand list its own parser produces: a top-level colon between two
// registers splits the operand in two with the halves swapped (the old x86
// "register pair" syntax the shared grammar keeps on every GOARCH), so
// INSTR R4:R5, R6 encodes exactly as INSTR R5, R4, R6, bytes included.
// The sugar is pure syntax: every acceptance question the reordered list
// raises is answered by the ordinary operand matching. changed reports
// whether any operand carried it; the operand objects are copied, never
// edited in the shared syntax tree.
func l64ExpandPairs(ops []*ast.Operand) ([]*ast.Operand, bool) {
changed := false
var out []*ast.Operand
for i, op := range ops {
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
if op.Addr.Base != "" || !strings.HasPrefix(sfx, ":") || !l64PairRegister(sfx[1:]) {
if changed {
out = append(out, op)
}
continue
}
if !changed {
out = make([]*ast.Operand, 0, len(ops)+1)
out = append(out, ops[:i]...)
changed = true
}
name := sfx[1:]
// The second register lands first. Both halves keep the operand
// shape a plain register spelling parses to (a bare symbol
// reference), which is what every operand consumer reads.
second := *op
second.Addr.Shift = ""
second.Addr.Sym = &ast.Symbol{Raw: name, Name: name}
second.Raw = name
first := *op
first.Addr.Shift = ""
first.Raw = operandRegName(op)
out = append(out, &second, &first)
}
return out, changed
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
@@ -367,10 +420,22 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
ops := instr.Operands
var neg bool
mnem, neg = l64SubToAdd(mnem, ops)
// The register-pair sugar widens the operand list exactly as the encode
// pass sees it, so both passes count the same instruction.
if pairOps, changed := l64ExpandPairs(ops); changed {
ops = pairOps
}
if mnem == "RET" {
return len(loong64Return(fi))
}
// BYTE lays down one raw byte per operand, a front-end pseudo-op the
// toolchain spells only on x86 but accepts here the same way the arm64
// and riscv64 encoders do (a superset spelling, shippable via the goobj
// path).
if mnem == "BYTE" {
return len(ops)
}
switch mnem {
case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes
@@ -380,6 +445,16 @@ func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
return 8 // bne/beq over the BREAK, then BREAK
case "PRELDX":
return 20 // the four-instruction constant materialisation + preldx
case "LL", "LLW", "LLV", "SC", "SCW", "SCV", "MOVWP", "MOVVP":
// The 2RI14 families: one word inside the signed 16-bit offset
// span, otherwise the toolchain's materialisation sequences (see
// l64firr14Words, which decides the width). A memory operand that
// will not parse leaves the single-word size: the encode pass
// reports the error.
if _, _, off, _, err := l64MemOperands(ops, fi); err == nil {
return len(l64firr14Words(0, int(off), 0, 0))
}
return 4
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
@@ -468,8 +543,45 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
copy(ops2[1:], ops[1:])
ops = ops2
}
// The toolchain's register-pair sugar: INSTR R4:R5, R6 encodes exactly
// as INSTR R5, R4, R6. The expansion is a fresh operand list; the
// instruction pointer stays the original one, because the PC-relative
// layout tables are keyed on it.
if pairOps, changed := l64ExpandPairs(ops); changed {
ops = pairOps
}
// Pseudo-instructions and the branches first.
// Before any of them: the toolchain's loong64 operand grammar has no
// shifted-register composition (R0<<2, R0>>R1, R0->3, R0@>3 are parse
// errors under GOARCH=loong64 go tool asm), and the shift suffix the
// shared parser records is arm64's. Encoding on would silently drop
// the composition and emit the bare register, so every operand whose
// verbatim suffix is not the element-selector index (V1.B[3] records
// the name as V1.B and the suffix "[3]", which is a real loong64
// form) or the register-pair sugar expanded above is rejected
// outright, with the toolchain's own wording where the shape is one
// it diagnoses.
for _, op := range ops {
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
if sfx == "" || strings.HasPrefix(sfx, "[") {
continue
}
if strings.HasPrefix(sfx, ":") {
if op.Addr.Base != "" {
// (Rj:Rk) inside an address: the pair never splits there.
return nil, fmt.Errorf("%s: indirect through register pair", mnem)
}
// A register-pair spelling with a register right half was
// expanded above, so whatever survives carries a right half
// outside the register table: the toolchain's parse-stage
// objection (R4:label).
return nil, fmt.Errorf("%s: illegal or missing addressing mode for symbol %s",
mnem, strings.TrimPrefix(sfx, ":"))
}
return nil, fmt.Errorf("%s: shifted register operand %s%s is not a loong64 form",
mnem, operandRegName(op), sfx)
}
switch mnem {
case "RET":
return loong64Return(fi), nil
@@ -484,6 +596,18 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "BYTE":
// BYTE $b lays down one raw byte per operand, the same front-end
// pseudo-op the arm64 and riscv64 encoders accept.
var out []byte
for _, op := range ops {
b := l64Imm64(op)
if b < 0 || b > 0xFF {
return nil, fmt.Errorf("BYTE: immediate %d does not fit a byte", b)
}
out = append(out, byte(b))
}
return out, nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm
@@ -585,11 +709,11 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
l64rrr(preldx, 30, rj, hint),
), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
return encodeLOONG64Branch(instr, ops, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
return encodeLOONG64Branch(instr, ops, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
return encodeLOONG64Mov(ops, mnem, fi, relocs)
}
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
@@ -659,7 +783,7 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
// the integer/FP table (their mnemonics overlap the table's 2R format
// but resolve vector-bank registers).
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
if code, handled, err := encodeLOONG64Vector(mnem, ops, fi); handled {
if err != nil {
return nil, err
}
@@ -701,6 +825,35 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fllsc:
// LLACQ{W,V} (Rj), Rd loads and SCREL{W,V} Rd, (Rj) stores, both
// 2R encodings op | rj<<5 | rd against a zero-offset memory operand
// (the toolchain's C_ZOREG, which rejects any displacement).
rd, rj, off, _, err := l64MemOperands(ops, fi)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Fscq:
// SCQ first, middle, (base): op | middle<<10 | base<<5 | first,
// against a zero-offset memory operand as with the LL/SC pair.
if len(ops) != 3 || !isMemOperand(ops[2]) || isMemOperand(ops[0]) || isMemOperand(ops[1]) {
return nil, fmt.Errorf("%s expects reg, reg, (reg)", mnem)
}
first, middle := l64Reg(ops[0]), l64Reg(ops[1])
rj, off := l64MemWithFrame(ops[2], fi)
if first < 0 || middle < 0 || rj < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
if off != 0 {
return nil, fmt.Errorf("%s: only a zero-offset memory operand is allowed", mnem)
}
return l64wordLE(l64rrr(enc.op, middle, rj, first)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
@@ -719,7 +872,10 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
case l64Firr16:
// ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be
// a multiple of 65536 and is shifted right by 16.
// a multiple of 65536 and is shifted right by 16. Both registers are
// general: the toolchain's class match refuses an F, V or X bank
// register here ("illegal combination ADDV16 ... FREG ..."), and the
// 5-bit numbering would otherwise encode it silently.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
@@ -727,6 +883,11 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
if v&0xFFFF != 0 {
return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem)
}
for _, op := range ops[1:] {
if c := loong64RegClass(operandRegName(op)); c != l64ClsGR {
return nil, fmt.Errorf("%s: illegal combination: the registers must be general", mnem)
}
}
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
@@ -749,7 +910,26 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
// ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field.
op -= 1 << 24
}
return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil
if off&3 != 0 {
return nil, fmt.Errorf("%s: offset must be a multiple of 4", mnem)
}
// The 64-bit span reproduces the toolchain's case 73/74 size-24
// expansion word for word, its missing load negation included: a
// MOVWP/MOVVP load emits the store opcode, because case 74 spells
// opirr(p.As) where the narrower spans spell opirr(-p.As), and the
// LL family has no unnegated opirr entry at all, so the toolchain
// refuses those loads with "bad irr opcode".
if off < -2147483650 || off >= 2147483646 {
if load {
switch mnem {
case "MOVWP", "MOVVP":
op += 1 << 24
default:
return nil, fmt.Errorf("bad irr opcode %s", mnem)
}
}
}
return l64firr14Words(op, int(off), rj, rd), nil
case l64Fir20:
// LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm.
@@ -868,11 +1048,14 @@ func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loo
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
//
// instr is the original instruction, the key the PC-relative layout table
// is keyed on; ops are the operands to encode, the pair expansion included.
func encodeLOONG64Branch(instr *ast.Instr, ops []*ast.Operand, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(ops) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
}
op := instr.Operands[0]
op := ops[0]
// PC-relative displacement: N(PC) resolves to the instruction N slots
// away in source order (the toolchain's parse-time count), and the field
// carries the final pc distance in instruction units.
@@ -1107,8 +1290,8 @@ func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.
//
// ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes)
// 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8)
// AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4)
// −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8)
// AND/OR/XOR: 0..0xfff → andi/ori/xori directly (4)
// −2048..−1 → addi.w r30, r0, v; op rd, rj, r30 (8)
// 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12)
// 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20)
func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) {
@@ -1146,12 +1329,18 @@ func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]by
if v == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= 0 && v <= 0x7ff {
if v >= 0 && v <= 0xfff {
// andi/ori/xori take the full unsigned 12-bit range, the
// toolchain's C_UU12CON class.
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= -2048 && v < 0 {
// The toolchain's case 10 loads the negative constant with
// addi.w (opirr(AADD)), whose 32-bit result zero-extends into
// the register; addi.d here would sign-extend instead and the
// 64-bit AND/OR/XOR would keep bits the toolchain clears.
return l64WordsLE(
l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v
l64irr(0x00a<<22, int(v), 0, 30), // addi.w r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
@@ -1209,26 +1398,61 @@ func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) {
// l64MemOperands extracts (rd, rj, off, load) from a load/store instruction:
// INSTR mem, rd is a load, INSTR rd, mem a store.
func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) {
// l64MemOperands splits the two-operand load/store forms: the register and
// the memory side, the memory's base and full-width byte offset (64-bit
// offsets reach the 2RI14 families' six-word span), and the direction.
func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int64, load bool, err error) {
if len(ops) != 2 {
return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops))
}
var mem *ast.Operand
if isMemOperand(ops[0]) {
rd = l64Reg(ops[1])
rj, off = l64MemWithFrame(ops[0], fi)
mem = ops[0]
load = true
} else if isMemOperand(ops[1]) {
rd = l64Reg(ops[0])
rj, off = l64MemWithFrame(ops[1], fi)
mem = ops[1]
} else {
return 0, 0, 0, false, fmt.Errorf("expected a memory operand")
}
if mem.Addr.Sym != nil && mem.Addr.Sym.Pseudo != "" {
pj, poff := loong64ResolvePseudo(mem.Addr.Sym, fi)
rj, off = pj, int64(poff)
} else {
rj, off = loong64RegNum(mem.Addr.Base), mem.Addr.Offset
}
if rd < 0 || rj < 0 {
return 0, 0, 0, false, fmt.Errorf("invalid operand")
}
return rd, rj, off, load, nil
}
// l64ImmMem reads the `$off(rj)` immediate form off an operand's raw text:
// the shared immediate parse reduces it to the bare number and keeps only
// the text as a witness of the base register. ok reports the form was
// found, with the base's register number (or -1 when the name is not a
// general register).
func l64ImmMem(op *ast.Operand) (off int32, base int, ok bool) {
if op.Kind != ast.OpImmediate || !op.Imm.HasVal {
return 0, 0, false
}
raw := strings.ReplaceAll(op.Raw, " ", "")
if !strings.HasPrefix(raw, "$") || !strings.HasSuffix(raw, ")") {
return 0, 0, false
}
open := strings.LastIndexByte(raw, '(')
if open < 2 {
return 0, 0, false
}
base = loong64RegNum(raw[open+1 : len(raw)-1])
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), base, base >= 0
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
@@ -1243,8 +1467,7 @@ func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off in
// MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d)
// MOVx sym(SB), rd load from a static symbol (pcalau12i+ld)
// MOVx rd, sym(SB) store to a static symbol (pcalau12i+st)
func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) {
ops := instr.Operands
func encodeLOONG64Mov(ops []*ast.Operand, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
@@ -1262,6 +1485,29 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
// MOVx $off(rj), rd computes an address: the toolchain's `mov
// $soreg, r` case, a plain addi.d whatever the move's width (both
// MOVW and MOVV $4(R4), R5 encode the same addi.d in its testdata).
// A wider offset materialises in R30 first (lu12i.w + ori + add.d,
// its case 10). The immediate's Raw carries the base register,
// which the shared immediate parse reduces to the bare number.
if off, base, ok := l64ImmMem(src); ok {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm(rj): invalid destination register", mnem)
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return nil, fmt.Errorf("%s $imm(rj): illegal combination with an F register destination", mnem)
}
if off >= -2048 && off <= 2047 {
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(off), base, rd)), nil
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(off)>>12, 30),
l64irr(l64DualTable["OR"].imm, int(off)&0xFFF, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, base, rd),
), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
@@ -1298,6 +1544,11 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
// Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk).
if src.Addr.Index != "" && !isMemOperand(dst) {
if src.Addr.HasOff {
// The toolchain rejects off(rj)(rk) with an illegal
// combination; the offset would be silently dropped here.
return nil, fmt.Errorf("%s: the register-indexed form takes no offset", mnem)
}
rd := l64Reg(dst)
rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index)
if rd < 0 || rj < 0 || rk < 0 {
@@ -1310,6 +1561,10 @@ func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs
return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil
}
if dst.Addr.Index != "" && !isMemOperand(src) {
if dst.Addr.HasOff {
// The store twin of the load rejection above.
return nil, fmt.Errorf("%s: the register-indexed form takes no offset", mnem)
}
rs := l64Reg(src)
rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index)
if rs < 0 || rj < 0 || rk < 0 {
@@ -1356,6 +1611,14 @@ func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
// The $off(rj) address immediate: addi.d in the 12-bit window,
// lu12i.w + ori + add.d beyond it (the toolchain's case 10).
if off, _, ok := l64ImmMem(src); ok {
if off >= -2048 && off <= 2047 {
return 4
}
return 12
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
@@ -1835,8 +2098,14 @@ func operandRegName(op *ast.Operand) string {
return ""
}
// l64Reg returns the register number of an operand, or -1.
// l64Reg returns the register number of a register operand, or -1. A
// memory reference is not a register, however register-shaped its base:
// the toolchain's class match refuses one wherever a C_REG is required,
// and reading the base's number here would encode it silently.
func l64Reg(op *ast.Operand) int {
if isMemOperand(op) {
return -1
}
return loong64RegNum(operandRegName(op))
}
@@ -1868,6 +2137,19 @@ func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
return l64Mem(op)
}
// l64VmovqMem resolves a VMOVQ/XVMOVQ memory operand. The toolchain's
// vector table falls back to the zero register as the FP-relative base
// (`VMOVQ V2, y+16(FP)` stores through R0 while MOVW reads the same operand
// through R3), so the vector moves keep the resolved offset but the zero
// base, exactly as `go tool asm` emits them.
func l64VmovqMem(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
rj, off = l64MemWithFrame(op, fi)
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "FP" {
rj = 0
}
return rj, off
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
@@ -1892,7 +2174,7 @@ func l64Label(op *ast.Operand) string {
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
width byte // suffix width letter (B/H/W/V/Q), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
@@ -1932,7 +2214,7 @@ func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
if w != 'B' && w != 'H' && w != 'W' && w != 'V' && w != 'Q' {
return v, false
}
v.width, v.hasSuf = w, true
@@ -2042,16 +2324,15 @@ func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
// mnemonic belongs to the vector slice; the operand shapes and opcode
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
func encodeLOONG64Vector(mnem string, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, bool, error) {
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", ops, fi)
return code, true, err
}
lasx, ok := l64VecBank[mnem]
if !ok {
return nil, false, nil
}
ops := instr.Operands
bank := "V"
if lasx {
bank = "X"
@@ -2174,6 +2455,10 @@ func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
// VMOVQ vj.T[i], vd.T vreplvei (broadcast one element, LSX)
// XVMOVQ xj, xd.T xvreplve0 (broadcast element zero, LASX)
// XVMOVQ xj, xd.T[i] xvinsve0 (insert element zero, LASX)
// XVMOVQ xj.T[i], xd xvpickve (extract one element, LASX)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
@@ -2200,6 +2485,118 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
return loong64RegNum(name), nil
}
// Element broadcast: VMOVQ vj.T[i], vd.T (vreplvei.{b,h,w,d}), the
// source element width matching the destination arrangement. An LSX-only
// form: the toolchain's table gives vreplvei no LASX counterpart.
if srcVec && dstVec && src.hasEl && dst.hasSuf && !dst.hasEl {
if lasx || src.lasx || dst.lasx {
return nil, fmt.Errorf("VMOVQ: vreplvei has no %s-bank form", bank)
}
if src.unsig {
return nil, fmt.Errorf("VMOVQ: vreplvei takes no unsigned element suffix")
}
if src.width != dst.width {
return nil, fmt.Errorf("VMOVQ: element width does not match arrangement %q", ops[1].Raw)
}
if _, ok := l64VecSuffixWidth(false, dst); !ok {
return nil, fmt.Errorf("VMOVQ: invalid arrangement %q", ops[1].Raw)
}
var op uint32
limit := 0
switch src.width {
case 'B':
op, limit = enc.rveiB, 15
case 'H':
op, limit = enc.rveiH, 7
case 'W':
op, limit = enc.rveiW, 3
default:
op, limit = enc.rveiD, 1
}
if src.elem > limit {
return nil, fmt.Errorf("VMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Broadcast of element zero: XVMOVQ xj, xd.T (xvreplve0.{b,h,w,d,q}),
// a bare X source into an arranged X destination. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasSuf && !dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvreplve0 is the %s-bank form alone", bank)
}
var op uint32
switch dst.width {
case 'B':
if dst.lanes != 32 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0B
case 'H':
if dst.lanes != 16 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0H
case 'W':
if dst.lanes != 8 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0W
case 'V':
if dst.lanes != 4 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0D
case 'Q':
if dst.lanes != 2 {
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
op = enc.rve0Q
default:
return nil, fmt.Errorf("XVMOVQ: invalid arrangement %q", ops[1].Raw)
}
return l64wordLE(op | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Insert of element zero: XVMOVQ xj, xd.T[i] (xvinsve0.{w,d}), a bare X
// source into one word or double-word lane. LASX only.
if srcVec && dstVec && !src.hasSuf && dst.hasEl {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 is the %s-bank form alone", bank)
}
op, limit := enc.xinsW, 7
if dst.width != 'W' {
op, limit = enc.xinsD, 3
if dst.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvinsve0 takes word or double-word lanes, got %q", ops[1].Raw)
}
}
if dst.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", dst.elem, limit)
}
return l64wordLE(op | uint32(dst.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Element extract into a vector register: XVMOVQ xj.T[i], xd
// (xvpickve.{w,d}), one word or double-word lane out to a bare X
// register. LASX only.
if srcVec && src.hasEl && dstVec && !dst.hasSuf {
if !lasx || src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("XVMOVQ: xvpickve is the %s-bank form alone", bank)
}
op, limit := enc.xpickW, 7
if src.width != 'W' {
op, limit = enc.xpickD, 3
if src.width != 'V' {
return nil, fmt.Errorf("XVMOVQ: xvpickve takes word or double-word lanes, got %q", ops[0].Raw)
}
}
if src.elem > limit {
return nil, fmt.Errorf("XVMOVQ: element index %d out of range [0, %d]", src.elem, limit)
}
return l64wordLE(op | uint32(src.elem)<<10 | uint32(src.num)<<5 | uint32(dst.num)), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
@@ -2224,7 +2621,7 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
rj, off := l64VmovqMem(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
@@ -2247,9 +2644,9 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
rj, off := l64VmovqMem(ops[0], fi)
if rj < 0 {
return nil, fmt.Errorf("VMOVQ: invalid load operand")
}
op := enc.ld
if dst.hasSuf {
@@ -2257,16 +2654,34 @@ func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]b
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
// vldrepl keeps the byte offset raw for bytes and scales it by
// the element width for the wider forms, the immediate field
// shrinking a bit per scale exactly as the toolchain encodes it
// (the field mask keeps the two's complement inside its width).
scale, mask, lo, hi := 1, int32(0xFFF), -2048, 2047
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
scale, mask, lo, hi = 2, 0x7FF, -1024, 1023
case 2:
op = enc.replW
scale, mask, lo, hi = 4, 0x3FF, -512, 511
default:
op = enc.replD
scale, mask, lo, hi = 8, 0x1FF, -256, 255
}
if off%int32(scale) != 0 {
return nil, fmt.Errorf("VMOVQ: offset %d must be a multiple of %d", off, scale)
}
off /= int32(scale)
if off < int32(lo) || off > int32(hi) {
return nil, fmt.Errorf("VMOVQ: offset out of range [%d, %d]", lo*scale, hi*scale)
}
off &= mask
} else if off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
+63 -5
View File
@@ -257,6 +257,37 @@ func l64WordsLE(ws ...uint32) []byte {
return out
}
// l64firr14Words returns the words the 2RI14 families (LL/LLW/LLV, SC/SCW/
// SCV, MOVWP/MOVVP) encode to for a 4-aligned byte offset. The toolchain
// classifies the memory operand by span, with the class constants' -2 slop
// included verbatim: a signed 16-bit offset (C_SOREG_16) rides in si14 alone;
// a 32-bit one (C_LOREG_32) splits between addu16i.d (bits 31:16, materialised
// in R30, the assembler temp) and si14 (bits 15:2); anything wider
// (C_LOREG_64) builds the whole constant in R30 through lu12i.w + ori +
// lu32i.d + lu52i.d and folds the base into it. The caller adjusts the
// opcode for the 64-bit span's load-direction quirks before arriving here.
func l64firr14Words(op uint32, off, rj, rd int) []byte {
switch {
case off >= -32766 && off < 32766:
return l64wordLE(l64irr14(op, off>>2, rj, rd))
case off >= -2147483650 && off < 2147483646:
return l64WordsLE(
l64irr16(l64InstrTable["ADDV16"].op, off>>16, 0, 30),
l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30),
l64irr14(op, off>>2, 30, rd),
)
default:
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, off>>12, 30),
l64irr(l64DualTable["OR"].imm, off&0xFFF, 30, 30),
l64ir(l64InstrTable["LU32ID"].op, off>>32, 30),
l64irr(l64InstrTable["LU52ID"].op, off>>52, 30, 30),
l64rrr(l64DualTable["ADDV"].rrr, 30, rj, rj),
l64irr14(op, 0, rj, rd),
)
}
}
// ---- instruction formats ----
type l64Format uint8
@@ -279,6 +310,8 @@ const (
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
l64Fllsc // acquire/release LL/SC (2R against a zero-offset memory operand)
l64Fscq // sc.q: op | middle<<10 | base<<5 | first against a zero-offset memory operand
)
// l64Enc is one instruction's encoding: its bit layout (format) and the
@@ -341,8 +374,9 @@ var l64Vec2R = map[string]bool{}
// such as vshuf.b).
var l64Vec4R = map[string]bool{}
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants, each pre-shifted to
// its exact bit range, read off `go tool objdump` of GOARCH=loong64
// `go tool asm` kernels and the toolchain's specialLsxMovInst table.
type l64VmovqEnc struct {
ld, st, ldx, stx uint32 // plain and indexed load/store
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
@@ -350,6 +384,11 @@ type l64VmovqEnc struct {
ins uint32 // vinsgr2vr element insert
dup uint32 // vreplgr2vr duplicate (width in [11:10])
move uint32 // vori.b/xvori.b $0 register move
rveiB, rveiH, rveiW, rveiD uint32 // vreplvei: broadcast one element (LSX)
rve0B, rve0H, rve0W uint32 // xvreplve0 broadcast of element zero (LASX)
rve0D, rve0Q uint32 // xvreplve0.{d,q}, ditto
xinsW, xinsD uint32 // xvinsve0: insert element zero (LASX)
xpickW, xpickD uint32 // xvpickve: extract element (LASX)
}
var l64VmovqTable = map[bool]l64VmovqEnc{
@@ -358,12 +397,17 @@ var l64VmovqTable = map[bool]l64VmovqEnc{
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
rveiB: 0x01CBDE << 14, rveiH: 0x0397BE << 13, rveiW: 0x072F7E << 12, rveiD: 0x0E5EFE << 11,
},
true: { // XVMOVQ, the LASX (X) bank
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
rve0B: 0x1DC1C0 << 10, rve0H: 0x1DC1E0 << 10, rve0W: 0x1DC1F0 << 10,
rve0D: 0x1DC1F8 << 10, rve0Q: 0x1DC1FC << 10,
xinsW: 0x03B7FE << 13, xinsD: 0x076FFE << 12,
xpickW: 0x03B81E << 13, xpickD: 0x07703E << 12,
},
}
@@ -373,7 +417,7 @@ func init() {
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15,
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
@@ -467,6 +511,20 @@ func init() {
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
// Acquire/release LL/SC (2R against a zero-offset memory operand):
// LLACQV (Rj), Rd loads, SCRELV Rd, (Rj) stores, both encoding
// op | rj<<5 | rd. Opcodes from cmd/internal/obj/loong64/instOp.go
// (ll.acq.{w,d}, sc.rel.{w,d}).
l64InstrTable["LLACQW"] = l64Enc{format: l64Fllsc, op: 0x0E15E0 << 10}
l64InstrTable["SCRELW"] = l64Enc{format: l64Fllsc, op: 0x0E15E1 << 10}
l64InstrTable["LLACQV"] = l64Enc{format: l64Fllsc, op: 0x0E15E2 << 10}
l64InstrTable["SCRELV"] = l64Enc{format: l64Fllsc, op: 0x0E15E3 << 10}
// SCQ (sc.q first, middle, (base)) keeps its own operand order: the
// encoding is op | middle<<10 | base<<5 | first, the memory operand's
// base in the rj field, not the toolchain's generic 3R layout.
l64InstrTable["SCQ"] = l64Enc{format: l64Fscq, op: 0x070AE << 15}
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
// selected by the operand kind; the shift mnemonics pair the 3R form
// with a 5/6-bit shift immediate.
@@ -868,7 +926,7 @@ func init() {
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
"XVSEQB": {0xED00 << 15, true, -16, 15, 0, 0x1F},
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
// 7-bit field, but the toolchain range-checks it down to si5 as well
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
@@ -877,7 +935,7 @@ func init() {
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
"XVSEQV": {0xED03 << 15, true, -16, 15, 0, 0x7F},
// vslti compares against a signed (or, in the U spellings, unsigned)
// si5/ui5 constant.
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
+316 -2
View File
@@ -8,8 +8,8 @@ import (
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstTextLOONG64 parses assembly source and returns the first TEXT body.
@@ -831,3 +831,317 @@ TEXT ·atoms(SB), NOSPLIT, $0
0x4C000020,
)
}
// TestLOONG64_llacqScrel pins the acquire/release LL/SC pair. The oracle
// words come from GOARCH=loong64 go tool objdump and the toolchain's own
// loong64enc1.s golden bytes.
func TestLOONG64_llacqScrel(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·llsc(SB), NOSPLIT, $0
LLACQW (R5), R4
LLACQV (R5), R4
SCRELW R4, (R6)
SCRELV R4, (R6)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x385780A4, // ll.acq.w r4, r5
0x385788A4, // ll.acq.d r4, r5
0x385784C4, // sc.rel.w r4, r6
0x38578CC4, // sc.rel.d r4, r6
0x4C000020,
)
// The toolchain accepts the zero-offset memory form alone.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
LLACQW 4(R5), R4
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
SCRELV R4, 8(R6)
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_vmovqSuffixed pins the element-broadcast and element-move
// VMOVQ/XVMOVQ forms, with the oracle words lifted verbatim from the
// toolchain's loong64enc1.s.
func TestLOONG64_vmovqSuffixed(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·vmovq(SB), NOSPLIT, $0
VMOVQ V1.B[3], V9.B16
VMOVQ V2.H[2], V8.H8
VMOVQ V3.W[1], V7.W4
VMOVQ V4.V[0], V6.V2
XVMOVQ X0, X31.B32
XVMOVQ X1, X30.H16
XVMOVQ X2, X29.W8
XVMOVQ X3, X28.V4
XVMOVQ X3, X27.Q2
XVMOVQ X0, X31.W[7]
XVMOVQ X1, X29.W[0]
XVMOVQ X3, X28.V[3]
XVMOVQ X4, X27.V[0]
XVMOVQ X31.W[7], X0
XVMOVQ X29.W[0], X1
XVMOVQ X28.V[3], X8
XVMOVQ X27.V[0], X9
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x72F78C29, // vreplvei.b v9, v1, 3
0x72F7C848, // vreplvei.h v8, v2, 2
0x72F7E467, // vreplvei.w v7, v3, 1
0x72F7F086, // vreplvei.d v6, v4, 0
0x7707001F, // xvreplve0.b x31, x0
0x7707803E, // xvreplve0.h x30, x1
0x7707C05D, // xvreplve0.w x29, x2
0x7707E07C, // xvreplve0.d x28, x3
0x7707F07B, // xvreplve0.q x27, x3
0x76FFDC1F, // xvinsve0.w x31, x0, 7
0x76FFC03D, // xvinsve0.w x29, x1, 0
0x76FFEC7C, // xvinsve0.d x28, x3, 3
0x76FFE09B, // xvinsve0.d x27, x4, 0
0x7703DFE0, // xvpickve.w x0, x31, 7
0x7703C3A1, // xvpickve.w x1, x29, 0
0x7703EF88, // xvpickve.d x8, x28, 3
0x7703E369, // xvpickve.d x9, x27, 0
0x4C000020,
)
// The rejected shapes: a width mismatch between the element and the
// arrangement, an element index past the lane count, a wrong-bank
// vreplvei and an arrangement the LASX bank does not spell.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.H[3], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ V1.B[16], V9.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X1.B[3], X9.B32
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.B16
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
XVMOVQ X0, X31.W[8]
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_parityFixes pins the operand forms whose encodings were found
// diverging from the toolchain by the loong64enc1.s differential: the
// $off(reg) address immediate (addi.d), the SCQ operand order, the scaled
// vldrepl offsets (with their field masks), the XVSEQB/XVSEQV immediate
// opcodes and the zero-register base the toolchain gives FP-relative
// VMOVQ/XVMOVQ memory operands. Golden words from loong64enc1.s.
func TestLOONG64_parityFixes(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·parity(SB), NOSPLIT, $0-32
MOVW $4(R4), R5
MOVV $4(R4), R5
MOVW $65536(R4), R5
MOVW $-4096(R4), R5
SCQ R4, R5, (R6)
VMOVQ 2(R4), V1.H8
VMOVQ -6(R4), V1.H8
VMOVQ -12(R4), V2.W4
VMOVQ -16(R4), V3.V2
XVMOVQ -10(R4), X1.H16
XVSEQB $0, X2, X4
XVSEQH $3, X2, X4
XVSEQW $12, X2, X4
XVSEQV $15, X2, X4
XVSEQV $-15, X2, X4
VMOVQ V2, y+16(FP)
VMOVQ y+16(FP), V2
VMOVQ V2, x+2030(FP)
XVMOVQ X6, y+16(FP)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x02C01085, // addi.d $4, r4, r5
0x02C01085, // addi.d $4, r4, r5 (MOVW keeps the 64-bit addi.d)
0x1400021E, // lu12i.w $16, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x15FFFFFE, // lu12i.w $-1, r30
0x038003DE, // ori $0, r30, r30
0x0010F885, // add.d r5, r4, r30
0x385714C4, // sc.q r4, r5, (r6): middle<<10 | base<<5 | first
0x30400481, // vldrepl.h v1, 2(r4)
0x305FF481, // vldrepl.h v1, -6(r4)
0x302FF482, // vldrepl.w v2, -12(r4)
0x3017F883, // vldrepl.d v3, -16(r4)
0x325FEC81, // xvldrepl.h x1, -10(r4)
0x76800044, // xvseqi.b x4, x2, 0
0x76808C44, // xvseqi.h x4, x2, 3
0x76813044, // xvseqi.w x4, x2, 12
0x7681BC44, // xvseqi.d x4, x2, 15
0x7681C444, // xvseqi.d x4, x2, -15
0x2C406002, // vst v2, 24(r0): FP-relative keeps the zero base
0x2C006002, // vld v2, 24(r0)
0x2C5FD802, // vst v2, 2038(r0)
0x2CC06006, // xvst x6, 24(r0)
0x4C000020,
)
// Misaligned vldrepl offsets are rejected, as the toolchain does.
for i, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
VMOVQ 3(R4), V1.H8
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
MOVW $4(R4), F1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("case %d: expected an error, got none", i)
}
}
}
// TestLOONG64_bytePseudo pins the BYTE literal-data pseudo-op, which the
// loong64 toolchain does not spell but the arm64 and riscv64 encoders of
// this package already accept for byte-exact data layout (a superset
// spelling, shippable via the goobj path).
func TestLOONG64_bytePseudo(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·bytes(SB), NOSPLIT, $0
BYTE $2
BYTE $1; BYTE $0
BYTE $255
RET
`)
code := assembleLOONG64Helper(t, fn)
// Four literal bytes, then RET (jirl r0, r1, 0); the trailing bytes pad
// the final word the way any sub-word tail does.
want := []byte{2, 1, 0, 0xFF, 0x20, 0x00, 0x00, 0x4C}
if !bytes.Equal(code[:len(want)], want) {
t.Errorf("bytes = % x, want % x", code, want)
}
for _, src := range []string{
`TEXT ·e(SB), NOSPLIT, $0
BYTE $256
RET
`,
`TEXT ·e(SB), NOSPLIT, $0
BYTE $-1
RET
`,
} {
fn := firstTextLOONG64(t, src)
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%q: expected an error, got none", src)
}
}
}
// TestLOONG64PairSugar pins the toolchain's register-pair spellings: a
// top-level colon between two registers splits the operand in two with the
// halves swapped, so INSTR R4:R5, R6 encodes exactly as INSTR R5, R4, R6.
// Each row is pinned two ways: against the word GOARCH=loong64 go tool asm
// emits for the pair spelling, and against gasm's own encoding of the
// direct spelling, which must be the same bytes.
func TestLOONG64PairSugar(t *testing.T) {
for _, tt := range []struct {
pair, direct string
want uint32 // the word go tool asm emits for the pair
}{
{"MULV R4:R5, R6", "MULV R5, R4, R6", 0x001D9486},
{"MULV R4:R5", "MULV R5, R4", 0x001D9484},
{"MULV R6, R4:R5", "MULV R6, R5, R4", 0x001D98A4},
{"DIVV R4:R5, R6", "DIVV R5, R4, R6", 0x00221486},
{"MULHV R4:R5, R6", "MULHV R5, R4, R6", 0x001E1486},
{"ADDV R4:R5, R6", "ADDV R5, R4, R6", 0x00109486},
{"AND R4:R5, R6", "AND R5, R4, R6", 0x00149486},
{"MOVV R4:R5", "MOVV R5, R4", 0x001500A4},
{"MULD F4:F5, F6", "MULD F5, F4, F6", 0x01051486},
{"VADDW V1:V2, V3", "VADDW V2, V1, V3", 0x700B0823},
} {
t.Run(tt.pair, func(t *testing.T) {
pair := assembleLOONG64Helper(t, firstTextLOONG64(t, `#include "textflag.h"
TEXT ·p(SB), NOSPLIT, $0
`+tt.pair+`
RET
`))
direct := assembleLOONG64Helper(t, firstTextLOONG64(t, `#include "textflag.h"
TEXT ·d(SB), NOSPLIT, $0
`+tt.direct+`
RET
`))
if len(pair) < 4 {
t.Fatalf("pair spelling produced %d bytes", len(pair))
}
got := binary.LittleEndian.Uint32(pair)
if got != tt.want {
t.Errorf("pair word = %08x, want %08x (go tool asm)", got, tt.want)
}
if len(direct) < 4 || !bytes.Equal(pair[:4], direct[:4]) {
t.Errorf("pair bytes % x differ from the direct spelling's % x", pair[:4], direct[:4])
}
})
}
}
// TestLOONG64_addv16 pins the ADDV16 immediate family, the toolchain's
// assembler name for the addu16i.d hardware instruction (loong64enc1.s
// spells it ADDV16; the spelling ADDU16I.D is the decoder's name for the
// encoding and no assembler input at all). The immediate must be a
// multiple of 65536 and rides the instruction field shifted right by 16;
// the oracle words are the toolchain's own loong64enc1.s rows, each also
// confirmed against GOARCH=loong64 go tool asm.
func TestLOONG64_addv16(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·a16(SB), NOSPLIT, $0
ADDV16 $(-32768<<16), R4, R5
ADDV16 $(0<<16), R4, R5
ADDV16 $(8<<16), R4, R5
ADDV16 $(32767<<16), R4, R5
ADDV16 $(16<<16), R4
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x12000085, // addu16i.d r5, r4, -32768
0x10000085, // addu16i.d r5, r4, 0
0x10002085, // addu16i.d r5, r4, 8
0x11FFFC85, // addu16i.d r5, r4, 32767
0x10004084, // addu16i.d r4, r4, 16 (destination-only form)
0x4C000020,
)
// The non-multiples the loong64error.s catalogue pins (its rows 3 and
// 4): $1 and $65535 both refuse.
for _, src := range []string{
"ADDV16 $1, R4, R5",
"ADDV16 $65535, R4, R5",
"ADDV16 $(16<<16), F4",
} {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%q: expected an error, got none", src)
}
}
}
+453
View File
@@ -0,0 +1,453 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"path/filepath"
"regexp"
"slices"
"strconv"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// loong64AcceptedErrorShapes lists the toolchain's loong64error.s spellings
// gasm still accepts, each an acceptance superset with a known shape. The
// list only shrinks: every tightening of the encoder moves spellings out of
// it, and a spelling reappearing here means a regression.
var loong64AcceptedErrorShapes = []string{}
// TestLoong64ToolchainErrorParity walks the toolchain's loong64error.s (Go
// 1.27, loong64) and requires gasm to reject every case the toolchain rejects
// with an equivalent diagnostic, the documented acceptance supersets above
// excepted. A live Go toolchain is needed for the source file; the test
// skips without one or in -short.
func TestLoong64ToolchainErrorParity(t *testing.T) {
goroot := loong64Goroot(t)
path := filepath.Join(goroot, "src", "cmd", "asm", "internal", "asm", "testdata", "loong64error.s")
data, err := os.ReadFile(path)
if err != nil {
t.Skip(err)
}
allowed := map[string]bool{}
for _, s := range loong64AcceptedErrorShapes {
allowed[s] = true
}
for raw := range strings.SplitSeq(string(data), "\n") {
line := strings.TrimSpace(raw)
if line == "" || strings.HasPrefix(line, "//") || strings.HasPrefix(line, "TEXT") || !strings.Contains(line, "ERROR") {
continue
}
body := line
if i := strings.Index(body, "//"); i >= 0 {
body = strings.TrimSpace(body[:i])
}
want := ""
if m := regexp.MustCompile(`ERROR "([^"]*)"`).FindStringSubmatch(line); m != nil {
want = m[1]
}
body = strings.ReplaceAll(body, "\t", " ")
body = strings.Join(strings.Fields(body), " ")
// Label-relative and non-operand lines need their function context.
if strings.Contains(body, "(PC)") || strings.Contains(body, "(SB)") || strings.Contains(body, "PCALIGN") ||
strings.HasPrefix(body, "RET") || strings.HasPrefix(body, "NOP") || strings.HasPrefix(body, "BRK") ||
strings.Contains(body, "again") || strings.Contains(body, "next") || strings.Contains(body, "loop") {
continue
}
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t" + body + "\n\tRET\n"
f, perr := parser.Parse("errorparity.s", src)
if len(perr) > 0 {
continue // the parser already rejects the spelling
}
_, aerr := AssembleFileLOONG64(f)
if aerr == nil {
if !allowed[body] {
t.Errorf("gasm accepts what the toolchain rejects: %s", body)
}
continue
}
if want != "" && !loong64DiagEquivalent(want, aerr.Error()) {
t.Errorf("gasm rejects %s with an inequivalent diagnostic:\n toolchain: %s\n gasm: %s", body, want, aerr)
}
}
}
// loong64DiagEquivalent answers whether gasm's rejection carries the same
// information as the toolchain's expected message. The two word the same
// facts differently: the toolchain writes "operand out of range 0 to 15"
// where gasm writes "immediate out of range [0, 15]", and both terminate
// sentences the toolchain punctuates. Canonicalisation drops the subject
// word and the connectives and compares the remaining token sequence.
func loong64DiagEquivalent(want, got string) bool {
return loong64DiagSubseq(loong64DiagTokens(want), loong64DiagTokens(got))
}
// loong64DiagTokens lower-cases a diagnostic, strips punctuation and the
// connective tokens, and returns its words. "to" survives only between
// digits, where the toolchain's range wording uses it; gasm's bracketed
// form never produces it.
func loong64DiagTokens(msg string) []string {
msg = strings.ToLower(msg)
for _, r := range []string{"[", "]", ",", ".", ":", "\n"} {
msg = strings.ReplaceAll(msg, r, " ")
}
fields := strings.Fields(msg)
out := make([]string, 0, len(fields))
for i, w := range fields {
switch w {
case "the", "a", "operand", "immediate":
// subjects and articles carry no constraint
case "to":
if i > 0 && i+1 < len(fields) && isDigits(fields[i-1]) && isDigits(fields[i+1]) {
continue // the range connective
}
out = append(out, w)
default:
out = append(out, w)
}
}
return out
}
// loong64DiagSubseq answers whether want is a subsequence of got.
func loong64DiagSubseq(want, got []string) bool {
i := 0
for _, w := range got {
if i < len(want) && w == want[i] {
i++
}
}
return i == len(want)
}
func isDigits(s string) bool {
for _, r := range s {
if r < '0' || r > '9' {
return false
}
}
return len(s) > 0
}
// loong64ForeignErrorCorpus lists the corpus directories whose .s files the
// loong64 build attempts: the toolchain's own assembler testdata (the
// foreign-architecture files the corpus audit indicts on loong64) and the
// go/build malformed-source probes.
var loong64ForeignErrorCorpus = []string{
filepath.Join("cmd", "asm", "internal", "asm", "testdata"),
filepath.Join("go", "build", "testdata", "bads"),
}
// loong64ForeignAcceptedLines lists toolchain rejections gasm still
// accepts, keyed by corpus-relative path and line, each with its reason.
// The list only shrinks: every tightening moves lines out of it, and a line
// reappearing here means a regression.
var loong64ForeignAcceptedLines = map[string]string{
// duperror.s rejections are whole-file semantics: the toolchain indicts
// the second declaration against the first in the same file, and a
// single TEXT or GLOBL line is legal on its own. gasm rejects the
// whole file with the same redeclaration diagnostic.
"cmd/asm/internal/asm/testdata/duperror.s:7": "symbol foo redeclared is a whole-file judgement",
"cmd/asm/internal/asm/testdata/duperror.s:11": "symbol bar redeclared is a whole-file judgement",
// The toolchain deletes the R22 spelling from the loong64 register map
// (its comment: avoid unintentionally clobbering g) and keeps g alone;
// gasm resolves R22 to register 22 as part of its deliberate ABI-alias
// superset (A0, SP, T0, ... share the table), so the three mips64.s
// spellings that name R22 assemble here.
"cmd/asm/internal/asm/testdata/mips64.s:135": "R22 spelling: gasm's deliberate ABI-alias superset",
"cmd/asm/internal/asm/testdata/mips64.s:138": "R22 spelling: gasm's deliberate ABI-alias superset",
"cmd/asm/internal/asm/testdata/mips64.s:218": "R22 spelling: gasm's deliberate ABI-alias superset",
}
// TestLoong64ForeignErrorParity takes go tool asm's own per-line rejections
// on GOARCH=loong64 as the ground truth over the foreign corpus files and
// requires gasm to reject every one of those lines too, the documented
// acceptance supersets above excepted. The toolchain is run once per file;
// the test skips without a live toolchain or in -short.
func TestLoong64ForeignErrorParity(t *testing.T) {
goroot := loong64Goroot(t)
files := loong64CorpusFiles(t, goroot)
if len(files) == 0 {
t.Skip("no corpus files")
}
for _, path := range files {
if filepath.Base(path) == "loong64error.s" {
continue // the native catalogue above walks it deeply
}
rel := loong64CorpusRel(goroot, path)
rejected := loong64ToolchainRejects(t, goroot, path)
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read %s: %v", path, err)
}
lines := strings.Split(string(data), "\n")
for _, ln := range rejected {
if ln < 1 || ln > len(lines) {
t.Fatalf("%s: diagnostic names line %d beyond the file", rel, ln)
}
body := strings.TrimSpace(lines[ln-1])
if body == "" || strings.HasPrefix(body, "//") {
continue
}
if i := strings.Index(body, "//"); i >= 0 {
body = strings.TrimSpace(body[:i])
}
body = strings.Join(strings.Fields(strings.ReplaceAll(body, "\t", " ")), " ")
key := rel + ":" + strconv.Itoa(ln)
if _, ok := loong64ForeignAcceptedLines[key]; ok {
continue
}
if loong64ProbeAssembles(body) {
t.Errorf("gasm accepts what the toolchain rejects at %s: %s", key, body)
}
}
}
}
// loong64RejectedNames pins the names the toolchain's loong64 table carries
// but refuses to encode: each is an "illegal combination" under every
// operand shape GOARCH=loong64 go tool asm accepts. gasm rejects them too,
// and the pair of rejections is the parity this catalogue asserts.
var loong64RejectedNames = []string{"RFE", "DUFFCOPY", "DUFFZERO", "PCALIGNMAX"}
// loong64PairErrorShapes walks the register-pair sugar's negative shapes,
// every one measured against GOARCH=loong64 go tool asm. want carries the
// toolchain's diagnostic where gasm's wording matches it (checked through
// the same equivalence as the catalogue above); an empty want pins the
// rejection alone, and the row's comment records the toolchain's own words
// for the divergence.
var loong64PairErrorShapes = []struct {
body string
want string
}{
// A pair never splits inside an address; the toolchain's parse stage
// refuses it outright.
{"MULV (R4:R5), R6", "indirect through register pair"},
// The reordered list fails the instruction's operand shape. The
// toolchain words these "illegal combination ..." at the class match;
// gasm words them at the operand count.
{"MOVV R4:R5, R6", ""},
{"BEQ R4:R5, R6", ""},
{"JMP R4:R5", ""},
{"MULV R4:R5, R7:R8, R6", ""},
{"MULV R4:R5, (R6)", ""},
{"BEQ R4:R5, 2(PC)", ""},
// A right half outside the register table; the toolchain's parse stage
// again ("illegal or missing addressing mode for symbol label").
{"MULV R4:label", "illegal or missing addressing mode for symbol label"},
// An immediate half on either side. The toolchain reorders the pair
// first and rejects at the class match ("illegal combination MULV
// U3CON ...", "MULV: expected register; found $4"); gasm rejects the
// shape without the reorder.
{"MOVV $4:R5", ""},
{"MULV R4:$5", ""},
}
// TestLoong64PairErrorParity requires gasm to refuse every register-pair
// spelling the toolchain refuses, with an equivalent diagnostic where the
// row carries one. The toolchain's accept side of the sugar is pinned
// byte for byte by TestLOONG64PairSugar; the empty-want rows are documented
// wording divergences, not acceptance divergences.
func TestLoong64PairErrorParity(t *testing.T) {
for _, tt := range loong64PairErrorShapes {
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t" + tt.body + "\n\tRET\n"
f, perr := parser.Parse("pairparity.s", src)
if len(perr) > 0 {
continue // the parser already rejects the spelling
}
_, aerr := AssembleFileLOONG64(f)
if aerr == nil {
t.Errorf("gasm accepts what the toolchain rejects: %s", tt.body)
continue
}
if tt.want != "" && !loong64DiagEquivalent(tt.want, aerr.Error()) {
t.Errorf("gasm rejects %s with an inequivalent diagnostic:\n toolchain: %s\n gasm: %s", tt.body, tt.want, aerr)
}
}
}
// TestLoong64BacklogNameParity walks the audit's known-but-unencodable
// backlog: the names go tool asm recognises on loong64 yet refuses to
// encode must be refused by gasm as well, and the one name the audit
// misses because its probe battery lacks the shape (SCQ, whose only form
// is the three-operand store-conditional) must encode to the toolchain's
// own bytes. The refusal claim is walked under the full operand-shape
// battery below, against the live toolchain and against gasm alike.
func TestLoong64BacklogNameParity(t *testing.T) {
if testing.Short() {
t.Skip("live toolchain backlog: skipped in -short mode")
}
for _, name := range loong64RejectedNames {
if loong64ProbeAssembles(name) {
t.Errorf("gasm encodes %s, which the toolchain refuses under every shape", name)
}
}
// The shapes the plain audit probes, walked here so the backlog's
// claim holds under any operand shape, not only the bare mnemonic:
// the toolchain must refuse every combination, and gasm must refuse
// it too.
goroot := loong64Goroot(t)
for _, name := range loong64RejectedNames {
for _, shape := range loong64BacklogShapes {
body := strings.TrimSpace(name + " " + shape)
if loong64ProbeAssembles(body) {
t.Errorf("gasm encodes %s, which the toolchain refuses under every shape", body)
}
// The toolchain side of the same row: the file holds the one
// instruction, and the live assembler must report it. A
// spelling the toolchain accepts would open the name for
// encoding and close nothing here but the parity.
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\np2:\n\t" + body + "\n\tRET\n"
dir := t.TempDir()
path := filepath.Join(dir, "backlogshape.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatalf("write: %v", err)
}
if lines := loong64ToolchainRejects(t, goroot, path); len(lines) == 0 {
t.Errorf("the toolchain assembles %q, against the backlog's claim", body)
}
}
}
// sc.q: loong64enc1.s pins the encoding as c4145738
// (SCQ R4, R5, (R6): sc.q rd, rk, (rj)).
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tSCQ R4, R5, (R6)\n\tRET\n"
f, perr := parser.Parse("backlog.s", src)
if len(perr) > 0 {
t.Fatalf("parse SCQ: %v", perr[0])
}
img, aerr := AssembleFileLOONG64(f)
if aerr != nil {
t.Fatalf("assemble SCQ: %v", aerr)
}
want := []byte{0xc4, 0x14, 0x57, 0x38, 0x20, 0x00, 0x00, 0x4c}
if !slices.Equal(img.Code, want) {
t.Errorf("SCQ encodes to % X, want % X (RET included)", img.Code, want)
}
}
// loong64BacklogShapes is the operand-shape battery the backlog walk runs
// every rejected name through: the register, immediate, memory, vector and
// label positions the loong64 grammar has, mirroring the plain audit's
// probe battery (cmd/gasm's probeShapes) so the two answer the same
// question.
var loong64BacklogShapes = []string{
"", "R4", "R4, R5", "R4, R5, R6", "$1, R4", "R4, (R5)", "(R4), R5",
"F0, F1", "F0, F1, F2", "V1, V2", "X1, X2", "V1, FCC0", "$65536, R4",
"R4, R5, p2", "p2(SB)", "R5, (R4), R6", "$1, $0", "R1, R5, 0",
"0(R7), $5, $0", "V1, V2, V3, V4", "R4, R5, (R6)", "$16", "(R4)",
}
// loong64ProbeAssembles answers whether the single corpus line, wrapped in
// its own function, parses and assembles on loong64.
func loong64ProbeAssembles(body string) bool {
src := "#include \"textflag.h\"\n\n"
if strings.HasPrefix(body, "TEXT ") {
src += body + "\n\tRET\n"
} else if strings.HasPrefix(body, "DATA ") || strings.HasPrefix(body, "GLOBL ") {
src += body + "\n\nTEXT ·probe(SB), NOSPLIT, $0-0\n\tRET\n"
} else {
src += "TEXT ·probe(SB), NOSPLIT, $0-0\n\t" + body + "\n\tRET\n"
}
f, perr := parser.Parse("foreignparity.s", src)
if len(perr) > 0 {
return false // the parser already rejects the line
}
_, aerr := AssembleFileLOONG64(f)
return aerr == nil
}
// loong64Goroot resolves the live toolchain root, the environment's own
// value first, `go env GOROOT` as the fallback.
func loong64Goroot(t *testing.T) string {
t.Helper()
if goroot := os.Getenv("GOROOT"); goroot != "" {
return goroot
}
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no GOROOT: %v", err)
}
return strings.TrimSpace(string(out))
}
// loong64CorpusFiles lists every .s file under the corpus roots, the
// testdata tree walked recursively.
func loong64CorpusFiles(t *testing.T, goroot string) []string {
t.Helper()
var files []string
for _, rel := range loong64ForeignErrorCorpus {
root := filepath.Join(goroot, "src", rel)
err := filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
if err != nil {
return err
}
if !d.IsDir() && strings.HasSuffix(path, ".s") {
files = append(files, path)
}
return nil
})
if err != nil {
t.Skipf("walk %s: %v", root, err)
}
}
slices.Sort(files)
return files
}
// loong64CorpusRel renders a corpus file's path relative to GOROOT/src, the
// key the accepted-lines catalogue uses.
func loong64CorpusRel(goroot, path string) string {
rel, err := filepath.Rel(filepath.Join(goroot, "src"), path)
if err != nil {
return path
}
return rel
}
var (
// loong64ParseDiag matches the parse-stage "file:line: message".
loong64ParseDiag = regexp.MustCompile(`(?m)^(?:\S*asm: )?([^:\s]+\.s):(\d+):`)
// loong64EncodeDiag matches the encode-stage
// "asm: message: 003c (file.s:61)".
loong64EncodeDiag = regexp.MustCompile(`\(([^:\s()]+\.s):(\d+)\)`)
)
// loong64ToolchainRejects runs go tool asm on the file for GOARCH=loong64
// and returns the line numbers it rejects, both diagnostic shapes
// collected. An infrastructure failure fails the test: without the ground
// truth there is no parity to assert.
func loong64ToolchainRejects(t *testing.T, goroot, path string) []int {
t.Helper()
obj := filepath.Join(t.TempDir(), "probe.o")
cmd := exec.Command("go", "tool", "asm", "-I", filepath.Join(goroot, "pkg", "include"), "-o", obj, path)
cmd.Env = append(os.Environ(), "GOARCH=loong64")
out, err := cmd.CombinedOutput()
var lines []int
seen := map[int]bool{}
for _, m := range loong64ParseDiag.FindAllStringSubmatch(string(out), -1) {
if strings.HasSuffix(m[1], ".s") {
ln, perr := strconv.Atoi(m[2])
if perr == nil && !seen[ln] {
seen[ln] = true
lines = append(lines, ln)
}
}
}
for _, m := range loong64EncodeDiag.FindAllStringSubmatch(string(out), -1) {
ln, perr := strconv.Atoi(m[2])
if perr == nil && !seen[ln] {
seen[ln] = true
lines = append(lines, ln)
}
}
slices.Sort(lines)
if err != nil && len(lines) == 0 {
t.Fatalf("go tool asm %s: %v\n%s", filepath.Base(path), err, out)
}
return lines
}
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// Loong64 frame mapping, matching the Go toolchain's loong64 backend.
+93 -1
View File
@@ -7,7 +7,7 @@ import (
"bytes"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestLOONG64_sys exercises the no-operand system instructions and the
@@ -131,6 +131,98 @@ TEXT ·ptr(SB), NOSPLIT, $0
)
}
// TestLOONG64_firr14Spans pins the 2RI14 offset spans of the LL/SC/MOVWP
// families against the toolchain's operand classes: an unaligned offset is
// rejected ("offset must be a multiple of 4"), a signed 16-bit offset rides
// in si14 alone, and anything wider materialises its high half in R30 with
// addu16i.d before the base folds in, with si14 keeping bits 15:2. The
// pinned words are GOARCH=loong64 go tool asm's own bytes for the same
// sources, boundaries included (-32766 and 32766 are the class edges).
func TestLOONG64_firr14Spans(t *testing.T) {
fn := firstTextLOONG64(t, `#include "textflag.h"
TEXT ·spans(SB), NOSPLIT, $0
SC R4, 32764(R5)
SC R4, -32764(R5)
SC R4, 32768(R5)
SC R4, -32768(R5)
SC R4, -32772(R5)
LL 65540(R5), R4
LLV 65540(R5), R4
SCV R4, 65540(R5)
MOVWP R4, 65540(R5)
RET
`)
code := assembleLOONG64Helper(t, fn)
wantWords(t, code,
0x217FFCA4, // sc.w r4, 8191(r5) — inside si14
0x218004A4, // sc.w r4, -8191(r5)
0x1000001E, // addu16i.d r30, r0, 0
0x001097DE, // add.d r30, r30, r5
0x218003C4, // sc.w r4, 8192(r30)
0x13FFFC1E, // addu16i.d r30, r0, -1
0x001097DE, // add.d r30, r30, r5
0x218003C4, // sc.w r4, 8192(r30)
0x13FFFC1E, // addu16i.d r30, r0, -1
0x001097DE, // add.d r30, r30, r5
0x217FFFC4, // sc.w r4, 8191(r30)
0x1000041E, // addu16i.d r30, r0, 1
0x001097DE, // add.d r30, r30, r5
0x200007C4, // ll.w r4, 1(r30)
0x1000041E, // addu16i.d r30, r0, 1
0x001097DE, // add.d r30, r30, r5
0x220007C4, // ldptr.d r4, 1(r30)
0x1000041E, // addu16i.d r30, r0, 1
0x001097DE, // add.d r30, r30, r5
0x230007C4, // sc.d r4, 1(r30)
0x1000041E, // addu16i.d r30, r0, 1
0x001097DE, // add.d r30, r30, r5
0x250007C4, // stptr.w r4, 1(r30)
0x4C000020,
)
// The expansion participates in layout: the function size carries the
// six three-word forms plus the single-word pair and the RET.
f, errs := parser.Parse("spans_loong64.s", `#include "textflag.h"
TEXT ·spans(SB), NOSPLIT, $0
SC R4, 32768(R5)
LL 65540(R5), R4
RET
`)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileLOONG64(f)
if err != nil {
t.Fatalf("AssembleFileLOONG64: %v", err)
}
if img.Funcs[0].Size != 8*3+4 {
t.Errorf("function size = %d, want %d", img.Funcs[0].Size, 8*3+4)
}
}
// TestLOONG64_firr14Alignment checks the LL/SC/MOVWP alignment rule the
// toolchain enforces as "offset must be a multiple of 4": the LoongArch
// ll/sc family requires a naturally scaled offset, and si14 cannot express
// the remainder. GOARCH=loong64 go tool asm rejects every case here.
func TestLOONG64_firr14Alignment(t *testing.T) {
cases := []string{
"SC R4, 1(R5)",
"SCW R4, -1(R5)",
"SCV R4, 2(R5)",
"LL 1(R5), R4",
"LLW 3(R5), R4",
"LLV 6(R5), R4",
"MOVWP R4, 1(R5)",
"MOVVP R4, 2(R5)",
}
for _, src := range cases {
fn := firstTextLOONG64(t, "#include \"textflag.h\"\nTEXT ·e(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, err := assembleLOONG64(fn); err == nil {
t.Errorf("%s: expected an error, got none", src)
}
}
}
// TestLOONG64_atomics exercises the AM* read-modify-write forms and
// RDTIME, plus the MOVV FP→GP move.
func TestLOONG64_atomics(t *testing.T) {
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestLOONG64RelocOffsetsIncludePrologue pins the function-relative
+83
View File
@@ -3,6 +3,8 @@
package asm
import "fmt"
// Operand is an instruction operand: a Reg, a Mem reference or an Imm value.
type Operand interface {
isOperand()
@@ -107,3 +109,84 @@ func isX86Mem(o Operand) bool {
return false
}
}
// byteReg reports whether r is a general-purpose register whose name spells
// a byte width (AL, CL, DL, BL, AH-DH, SPL-DIL, R8B-R15B): the name itself
// fixes an 8-bit access, unlike the size-agnostic spellings (AX, EAX, RAX,
// R11) whose width the mnemonic supplies. The vector, opmask, x87, MMX,
// control and segment registers never name a byte access.
func byteReg(r Reg) bool {
return r.size == 1 && !r.isVec() && !r.mask && !r.fp && !r.mmx && r.ctl == 0 && r.seg == 0
}
// byteFormBase lists the scalar families that own an 8-bit encoding beside
// the 16/32/64-bit ones: the arithmetic and logic group, TEST, the moves,
// the unary group, the shifts, the exchanges and atomics, the one-operand
// IMUL and CRC32. Families without a byte form (LEA, BT, the bit scans,
// PUSH/POP, ADCX/ADOX, BSWAP) stay out, so their width comes from the
// mnemonic alone and a byte register in them is the handler's own error to
// make.
var byteFormBase = map[string]bool{
"MOV": true,
"ADD": true, "OR": true, "ADC": true, "SBB": true,
"AND": true, "SUB": true, "XOR": true, "CMP": true,
"TEST": true,
"INC": true, "DEC": true, "NEG": true, "NOT": true,
"MUL": true, "DIV": true, "IDIV": true,
"SHL": true, "SAL": true, "SHR": true, "SAR": true,
"ROL": true, "ROR": true, "RCL": true, "RCR": true,
"XCHG": true, "CMPXCHG": true, "XADD": true,
"CRC32": true,
"IMUL": true,
}
// operandWidth settles the operand width of a suffixed scalar instruction
// against the widths its register operands spell. A byte-spelled register
// operand (AL, DL, R8B, ...) forces the 8-bit form whatever the mnemonic's
// L suffix or lack of one says: that is the text the toolchain's own
// disassembly prints for the byte encodings (the rendered suffix rides the
// operand-size attribute, so an L or no suffix at all can name a byte
// form), and every register operand joins the form at its low byte, the way
// go tool asm reads the classic names there (TESTL R11, DL encodes TESTB
// R11B, DL). A W or Q suffix never rides a byte form, so that mix is a
// conflict rejected the way the toolchain rejects it (MOVQ AL, AX). With
// no byte operand the mnemonic decides alone and 0 returns, keeping the
// caller's width: the classic names are size-agnostic at every width.
func operandWidth(mnem string, suffix int, ops []Operand) (int, error) {
for _, o := range ops {
r, ok := o.(Reg)
if !ok || !byteReg(r) {
continue
}
if suffix == 2 || suffix == 8 {
kind := "quad"
if suffix == 2 {
kind = "word"
}
return 0, fmt.Errorf("%s: byte register cannot take the %s form", mnem, kind)
}
return 1, nil
}
return 0, nil
}
// widthOperands selects the operands that carry the instruction's data
// width for base: every operand of the scalar families but a shift's count,
// which names the CL register without narrowing the shifted value (RCLW CL,
// 0(R11) stays 16-bit), and nothing of IMUL's two- and three-operand forms,
// which have no byte encoding at all.
func widthOperands(base string, ops []Operand) []Operand {
switch base {
case "SHL", "SAL", "SHR", "SAR", "ROL", "ROR", "RCL", "RCR":
if len(ops) > 1 {
return ops[1:]
}
return nil
case "IMUL":
if len(ops) == 1 {
return ops
}
return nil
}
return ops
}
+30 -1
View File
@@ -17,13 +17,28 @@ import "strings"
// size. The high flag marks the legacy high-byte registers AH/CH/DH/BH, which
// occupy indices 4-7 yet take no REX prefix, unlike SPL/BPL/SIL/DIL that share
// those indices but require one. The mask flag marks the AVX-512 opmask
// registers K0-K7, the fp flag the x87 stack registers F0-F7.
// registers K0-K7, the fp flag the x87 stack registers F0-F7, the mmx flag the
// MMX registers M0-M7, the seg field a bare segment register (FS, GS) and the
// ctl field the control and debug registers, whose number rides an
// instruction's reg field rather than r/m.
type Reg struct {
idx int
size int // informational width implied by the name; the mnemonic decides
high bool // AH/CH/DH/BH
mask bool // K0-K7 opmask register
fp bool // F0-F7 x87 stack register
mmx bool // M0-M7 MMX register
seg int // segment register number plus one (ES=1..GS=6); 0 = not one
ctl byte // 0 none, 1 CRn control register, 2 DRn debug register
}
// segNumber returns the segment register number (ES=0..GS=5) when r names a
// bare segment register.
func (r Reg) segNumber() (int, bool) {
if r.seg == 0 {
return 0, false
}
return r.seg - 1, true
}
// Index returns the register number (0-15 for GPRs, 0-31 for vectors).
@@ -149,6 +164,20 @@ func buildRegByName() map[string]Reg {
for i := 0; i <= 7; i++ {
m["F"+itoa(i)] = Reg{idx: i, size: 8, fp: true}
}
// MMX: M0..M7.
for i := 0; i <= 7; i++ {
m["M"+itoa(i)] = Reg{idx: i, size: 8, mmx: true}
}
// Bare segment registers: ES, CS, SS, DS, FS, GS (the memory-base and
// index spellings of FS and GS are handled before register lookup).
for i, n := range []string{"ES", "CS", "SS", "DS", "FS", "GS"} {
m[n] = Reg{idx: i, size: 2, seg: i + 1}
}
// Control and debug registers: CR0..CR15, DR0..DR15.
for i := 0; i <= 15; i++ {
m["CR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 1}
m["DR"+itoa(i)] = Reg{idx: i, size: 8, ctl: 2}
}
return m
}
+51
View File
@@ -0,0 +1,51 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/hex"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestRenderedTextEncodes pins the encoder against the disassembler's
// renderings: every line here is a text the toolchain's own decoder prints
// for bytes its assembler produced (the disasm package's parity fixtures),
// so assembling the printed text must reproduce those bytes. Each row was
// read back through the toolchain's decoder; the hex is the fixture's.
func TestRenderedTextEncodes(t *testing.T) {
for _, tt := range []struct {
text string
want string
}{
// The segment-absolute rendering lowers to the FS-prefixed disp32
// absolute, exactly what the 0(FS) spelling encodes.
{"MOVQ FS:0, DX", "64488b142500000000"},
// The bank-crossing quadword move takes the prefix by direction: the
// MMX source carries F3, the XMM source F2.
{"MOVQ2DQ M2, X11", "f3440fd6da"},
} {
src := "TEXT \u00b7k(SB), NOSPLIT, $0\n\t" + tt.text + "\n\tRET\n"
f, errs := parser.Parse("k.s", src)
if len(errs) > 0 {
t.Errorf("%s: parse: %v", tt.text, errs[0])
continue
}
img, err := AssembleFile(f)
if err != nil {
t.Errorf("%s: assemble: %v", tt.text, err)
continue
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
// The no-fallthrough rule appends the RET; the pinned instruction is
// the prefix.
got := hex.EncodeToString(code)
if !strings.HasPrefix(got, tt.want) {
t.Errorf("%s: bytes %s, want prefix %s", tt.text, got, tt.want)
}
}
}
+3263 -409
View File
File diff suppressed because it is too large. Load diff
+105
View File
@@ -0,0 +1,105 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"path/filepath"
"testing"
)
// TestRISCVBitManipDifferential proves the Zba, Zbb, Zbc and Zbs families
// against the toolchain: the toolchain's own testdata spellings (every
// operand form each section carries) assembled by gasm and by go tool asm
// must agree byte for byte. The lines are the oracle's own, so a wrong
// funct6, a swapped operand pair or a missed two-operand collapse names
// itself through the first differing word.
func TestRISCVBitManipDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·bitmanip(SB), NOSPLIT, $0
ADDUW X10, X11, X12
ADDUW X10, X11
SH1ADD X11, X12, X13
SH1ADD X11, X12
SH1ADDUW X12, X13, X14
SH1ADDUW X12, X13
SH2ADD X13, X14, X15
SH2ADD X13, X14
SH2ADDUW X14, X15, X16
SH2ADDUW X14, X15
SH3ADD X15, X16, X17
SH3ADD X15, X16
SH3ADDUW X16, X17, X18
SH3ADDUW X16, X17
SLLIUW $31, X17, X18
SLLIUW $63, X17
SLLIUW $63, X17, X18
SLLIUW $1, X18, X19
ANDN X19, X20, X21
ANDN X19, X20
CLZ X20, X21
CLZW X21, X22
CPOP X22, X23
CPOPW X23, X24
CTZ X24, X25
CTZW X25, X26
MAX X26, X28, X29
MAX X26, X28
MAXU X28, X29, X30
MAXU X28, X29
MIN X29, X30, X5
MIN X29, X30
MINU X30, X5, X6
MINU X30, X5
ORN X6, X7, X8
ORN X6, X7
SEXTB X16, X17
SEXTH X17, X18
XNOR X18, X19, X20
XNOR X18, X19
ZEXTH X19, X20
ROL X8, X9, X10
ROL X8, X9
ROLW X9, X10, X11
ROLW X9, X10
ROR X10, X11, X12
ROR X10, X11
ROR $63, X11
RORI $63, X11, X12
RORI $1, X12, X13
RORIW $31, X13, X14
RORIW $1, X14, X15
RORW X15, X16, X17
RORW X15, X16
RORW $31, X13
ORCB X5, X6
REV8 X7, X8
CLMUL X5, X6, X7
CLMUL X5, X6
CLMULH X5, X6, X7
CLMULH X5, X6
CLMULR X5, X6, X7
CLMULR X5, X6
BCLR X23, X24, X25
BCLR $63, X24
BCLRI $1, X25, X26
BEXT X26, X28, X29
BEXT $63, X28
BEXTI $1, X29, X30
BINV X30, X5, X6
BINV $63, X6
BINVI $1, X7, X8
BSET X8, X9, X10
BSET $63, X9
BSETI $1, X10, X11
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "bitmanip_riscv64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertRISCVDifferential(t, path, src, "bitmanip")
}
+142
View File
@@ -0,0 +1,142 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"path/filepath"
"testing"
)
// TestRISCVCompressedDifferential proves the explicit compressed-instruction
// mnemonics against the toolchain: the "C" extension block of the toolchain's
// own testdata (every stack, register, control-transfer, constant-generation,
// shift and register-register spelling it carries) assembled by gasm and by
// go tool asm must agree halfword for halfword. The lines are the oracle's
// own, so a wrong bit pattern, scale or register field names itself through
// the first differing halfword.
func TestRISCVCompressedDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·compressed(SB), NOSPLIT, $0
CLWSP 20(SP), X10
CLDSP 24(SP), X10
CFLDSP 32(SP), F10
CSWSP X10, 20(SP)
CSDSP X10, 24(SP)
CFSDSP F10, 32(SP)
CLW 20(X10), X11
CLD 24(X10), X11
CFLD 32(X10), F11
CSW X11, 20(X10)
CSD X11, 24(X10)
CFSD F11, 32(X10)
CJ 1(PC)
CJR X5
CJALR X5
CBEQZ X10, 1(PC)
CBNEZ X10, 1(PC)
CLI $-32, X5
CLI $31, X5
CLUI $-32, X5
CLUI $31, X5
CADD $-32, X5
CADD $31, X5
CADDI $-32, X5
CADDI $31, X5
CADDW $-32, X5
CADDW $31, X5
CADDIW $-32, X5
CADDIW $31, X5
CADDI16SP $-512, SP
CADDI16SP $496, SP
CADDI4SPN $4, SP, X10
CADDI4SPN $1020, SP, X10
CSLLI $63, X5
CSRLI $63, X10
CSRAI $63, X10
CAND $-32, X10
CAND $31, X10
CANDI $-32, X10
CANDI $31, X10
CMV X6, X5
CADD X9, X8
CAND X9, X8
COR X9, X8
CXOR X9, X8
CSUB X9, X8
CADDW X9, X8
CSUBW X9, X8
CNOP
CEBREAK
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "compressed_riscv64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertRISCVDifferential(t, path, src, "compressed")
}
// TestRISCVCompressedRange pins the compressed immediate and offset ranges at
// the toolchain's own boundaries: a stack load off the scale or range, a
// non-prime register in the CL/CS and CA shapes, a zero immediate where the
// toolchain forbids one and a CLUI into SP are all rejected on sight.
func TestRISCVCompressedRange(t *testing.T) {
asmOne := func(t *testing.T, stmt string) error {
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·c(SB), NOSPLIT, $0\n\t"+stmt+"\n\tRET\n")
_, _, _, _, _, _, err := assembleRISCV(fn, nil)
return err
}
for _, s := range []string{
"CLWSP $0(SP), X10", // never spelled; the parser rejects the shape
} {
if err := asmOne(t, s); err == nil {
t.Errorf("%s must be rejected", s)
}
}
for _, s := range []string{
"CLWSP 21(SP), X10", // not a multiple of 4
"CLWSP 256(SP), X10", // out of range
"CLDSP 25(SP), X10", // not a multiple of 8
"CFLDSP 33(SP), F10", // not a multiple of 8
"CLWSP 20(X10), X10", // base must be SP
"CLW 22(X10), X11", // not a multiple of 4
"CLW 128(X10), X11", // out of range
"CLW 20(X5), X11", // base must be prime
"CLW 20(X10), X5", // rd must be prime
"CLI $32, X5", // out of range
"CLI $-33, X5", // out of range
"CLUI $0, X5", // zero
"CLUI $3, X2", // SP as destination
"CSLLI $0, X5", // zero shift
"CSLLI $64, X5", // out of range
"CSRLI $63, X5", // rd must be prime
"CANDI $63, X10", // out of range
"CMV X0, X5", // X0 in rd
"CMV X5, X0", // X0 in rs2
"CADD X5, X0", // X0 in rs2
"CSUB X5, X5", // X0-free but rd prime required
"CADDI4SPN $4, X5, X10", /* base must be SP */
} {
if err := asmOne(t, s); err == nil {
t.Errorf("%s must be rejected, as go tool asm rejects it", s)
}
}
for _, s := range []string{
"CLWSP 20(SP), X10", "CLDSP 24(SP), X10",
"CLW 20(X10), X11", "CSD X11, 24(X10)",
"CLI $-32, X5", "CLUI $-32, X5",
"CADD $-32, X5", "CADDIW $31, X5",
"CSLLI $63, X5", "CSRLI $63, X10", "CANDI $-32, X10",
"CMV X6, X5", "CADD X9, X8", "CSUB X9, X8", "CADDW X9, X8",
"CADDI16SP $496, SP", "CADDI4SPN $1020, SP, X10",
} {
if err := asmOne(t, s); err != nil {
t.Errorf("%s must assemble: %v", s, err)
}
}
}
+110
View File
@@ -0,0 +1,110 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestRISCVCSRDifferential proves the whole CSR name table against the
// toolchain at once: one TEXT whose body reads every register the table
// carries (CSRRW X0, NAME, X5), assembled by gasm and by go tool asm, must
// agree byte for byte. A single wrong address names its register through
// the first differing word: the CSR address occupies the instruction's top
// twelve bits, so 329 names spread over distinct words identify themselves.
func TestRISCVCSRDifferential(t *testing.T) {
names := make([]string, 0, len(riscvCSRNames))
for name := range riscvCSRNames {
names = append(names, name)
}
slices.Sort(names)
var body strings.Builder
for _, name := range names {
body.WriteString(fmt.Sprintf("\tCSRRW X0, %s, X5\n", name))
}
src := "#include \"textflag.h\"\n\nTEXT ·csrs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n"
dir := t.TempDir()
path := filepath.Join(dir, "csrs_riscv64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertRISCVDifferential(t, path, src, "csrs")
}
// TestRISCVCSRPseudosDifferential pins the CSR pseudo spellings against the
// oracle: the write-only forms (source first, CSR second), their immediate
// variants and the CSRR read pseudo, one word each.
func TestRISCVCSRPseudosDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·csrps(SB), NOSPLIT, $0
CSRS X5, TIME
CSRC X5, CYCLE
CSRW X5, INSTRET
CSRSI $1, TIME
CSRCI $2, CYCLE
CSRWI $3, INSTRET
CSRR VL, X10
CSRRW X0, TIME, X11
CSRRS X0, CYCLE, X12
CSRRC X0, INSTRET, X13
CSRRWI $4, TIME, X14
CSRRSI $5, CYCLE, X15
CSRRCI $6, INSTRET, X16
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "csrps_riscv64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertRISCVDifferential(t, path, src, "csrps")
}
// assertRISCVDifferential assembles the same source with gasm and with the
// toolchain for riscv64 and requires the named function's code bytes to
// agree. The live oracle is a deliberate-run comparison, so -short skips it
// (the push pipeline's mode); the golden bytes of the individual encoders
// are pinned separately in every mode.
func assertRISCVDifferential(t *testing.T, path, src, fn string) {
t.Helper()
oracle := oracleFuncCode(t, toolAsmObject(t, path, "riscv64"))
// The oracle keys its functions by the qualified object name
// (pkg.name); match on the local part.
want := map[string][]byte{}
for name, code := range oracle {
if _, after, ok := strings.Cut(name, "."); ok {
want[after] = code
} else {
want[name] = code
}
}
if want[fn] == nil {
t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want))
}
f, perrs := parser.Parse(path, src)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs[0])
}
img, err := AssembleFileRISCV(f)
if err != nil {
t.Fatalf("AssembleFileRISCV: %v", err)
}
got := trimTrailingZeroWords(img.Code)
wantB := trimTrailingZeroWords(want[fn])
if !slices.Equal(got, wantB) {
t.Errorf("%s: gasm and go tool asm disagree:\n gasm % x\n go % x", fn, got, wantB)
}
}
+746 -165
View File
File diff suppressed because it is too large. Load diff
+168 -16
View File
@@ -7,11 +7,13 @@ import (
"bytes"
"encoding/binary"
"encoding/hex"
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// firstTextRISCV parses assembly source and returns the first TEXT function body.
@@ -33,7 +35,7 @@ func firstTextRISCV(t *testing.T, src string) *ast.Text {
// assembleRISCVHelper assembles one TEXT function and returns its code bytes.
func assembleRISCVHelper(t *testing.T, fn *ast.Text) []byte {
t.Helper()
code, _, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -107,6 +109,34 @@ TEXT ·imm(SB), NOSPLIT, $0
}
}
// TestRISCV_utypeImmediate pins the U-type immediate semantics byte for byte
// against go tool asm (riscv64.s lines 76 to 85): the source immediate is the
// raw 20-bit field, not a byte address to divide, and both sign extremes
// encode. The range beyond the signed 20 bits is the toolchain's own error.
func TestRISCV_utypeImmediate(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·utype(SB), NOSPLIT, $0
AUIPC $524287, X10
LUI $524287, X15
LUI $167, X15
AUIPC $-524288, X15
RET
`)
code := assembleRISCVHelper(t, fn)
want := "17f5ff7f" + "b7f7ff7f" + "b7770a00" + "97070080"
if got := hex.EncodeToString(code[:16]); got != want {
t.Errorf("U-type immediates: %s, want %s", got, want)
}
fn = firstTextRISCV(t, `#include "textflag.h"
TEXT ·wide(SB), NOSPLIT, $0
AUIPC $524288, X10
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn, nil); err == nil {
t.Error("AUIPC $524288: expected the 20-bit range error, got none")
}
}
func TestRISCV_branches(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·br(SB), NOSPLIT, $0
@@ -785,6 +815,50 @@ TEXT ·sys(SB), NOSPLIT, $0
}
}
// TestRISCV_privilegedWords pins the privileged ISA slice. The toolchain's
// object table carries the encodings (its assembler accepts no mnemonic for
// them), so the golden vectors come from the privileged and debug
// specifications: SFENCEVMA X10, X11 is 0x12b50073, SRET 0x10200073,
// MRET 0x30200073, WFI 0x10500073 and DRET 0x7b200073.
func TestRISCV_privilegedWords(t *testing.T) {
fn := firstTextRISCV(t, `#include "textflag.h"
TEXT ·priv(SB), NOSPLIT, $0
SFENCEVMA X10, X11
SRET
MRET
WFI
DRET
RET
`)
code := assembleRISCVHelper(t, fn)
riscvWants(t, code,
0x12B50073, // sfence.vma x10, x11
0x10200073, // sret
0x30200073, // mret
0x10500073, // wfi
0x7B200073, // dret
)
}
// TestRISCV_privilegedAliases checks the supervisor spellings SCALL and
// SBREAK against the toolchain: both alias ECALL and EBREAK and must come
// out byte-identical.
func TestRISCV_privilegedAliases(t *testing.T) {
src := `#include "textflag.h"
TEXT ·alias(SB), NOSPLIT, $0
SCALL
SBREAK
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "alias_riscv64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertRISCVDifferential(t, path, src, "alias")
}
func TestRISCV_MOV_sym_FP(t *testing.T) {
// MOV $sym(FP), rd lowers to the frame-adjusted ADDI against SP: the
// toolchain's argframe spelling. A zero frame leaves the offset at the
@@ -794,7 +868,7 @@ TEXT ·argfp(SB), NOSPLIT, $0
MOV $arg(FP), X10
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -817,7 +891,7 @@ TEXT ·book(SB), NOSPLIT, $0-8
MOV X10, ret+0(FP)
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -841,7 +915,7 @@ TEXT ·slots(SB), NOSPLIT, $0-0
JMP -3(PC)
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -869,7 +943,7 @@ TEXT ·wide(SB), NOSPLIT, $0-0
MOV $0x000fffffffffffda, X5
RET
`)
code, _, _, _, _, _, err := assembleRISCV(fn)
code, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -929,7 +1003,7 @@ TEXT ·calltest(SB), NOSPLIT, $0
CALL ext(SB)
RET
`)
code, _, relocs, _, _, _, err := assembleRISCV(fn)
code, _, relocs, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("assemble: %v", err)
}
@@ -958,7 +1032,7 @@ TEXT ·calllocal(SB), NOSPLIT, $0
sub:
RET
`)
_, _, _, _, _, _, err := assembleRISCV(fn)
_, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err == nil {
t.Error("expected error for CALL to local label, got nil")
}
@@ -992,7 +1066,7 @@ func encodeOneInstrRISCV(t *testing.T, src string, pc int, offsets map[string]in
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\n"+src)
instr := fn.Body[0].(*ast.Instr)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil)
return encodeRISCVInstr(instr, pc, offsets, riscvFrameInfo{}, nil, nil, nil, nil)
}
// TestRISCVBranchJumpRange checks that displacements beyond the B-type span
@@ -1041,7 +1115,7 @@ func TestRISCVBranchFarBody(t *testing.T) {
}
sb.WriteString("done:\n\tRET\n")
fn := firstTextRISCV(t, sb.String())
out, _, _, _, _, _, err := assembleRISCV(fn)
out, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
@@ -1067,7 +1141,7 @@ TEXT ·csrhi(SB), NOSPLIT, $0
CSRRW $4096, X10, X11
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
if _, _, _, _, _, _, err := assembleRISCV(fn, nil); err == nil {
t.Error("expected an out-of-range error for CSR $4096, got none")
}
fn = firstTextRISCV(t, `#include "textflag.h"
@@ -1075,7 +1149,7 @@ TEXT ·csrmax(SB), NOSPLIT, $0
CSRRW $4095, X10, X11
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
if _, _, _, _, _, _, err := assembleRISCV(fn, nil); err != nil {
t.Errorf("CSR $4095 must assemble: %v", err)
}
}
@@ -1092,7 +1166,7 @@ func TestRISCV_Imm64Rejected(t *testing.T) {
}
for _, src := range cases {
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·wide(SB), NOSPLIT, $0\n\t"+src+"\n\tRET\n")
if _, _, _, _, _, _, err := assembleRISCV(fn); err == nil {
if _, _, _, _, _, _, err := assembleRISCV(fn, nil); err == nil {
t.Errorf("%s: expected an out-of-range error, got none", src)
}
}
@@ -1105,7 +1179,7 @@ TEXT ·edge(SB), NOSPLIT, $0
SUB $0x80000000, X12, X13
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
if _, _, _, _, _, _, err := assembleRISCV(fn, nil); err != nil {
t.Errorf("int32-span immediates must assemble: %v", err)
}
// Beyond the span the MOV forms materialise the constant like the
@@ -1115,7 +1189,7 @@ TEXT ·pool(SB), NOSPLIT, $0
MOV $0x123456789, X10
RET
`)
if _, _, _, _, _, _, err := assembleRISCV(fn); err != nil {
if _, _, _, _, _, _, err := assembleRISCV(fn, nil); err != nil {
t.Errorf("MOV with a 64-bit immediate must assemble: %v", err)
}
}
@@ -1340,3 +1414,81 @@ TEXT ·v(SB), NOSPLIT, $0
0xEA056207, // vlsseg8e32.v v4, (x10), x0
)
}
// TestRISCV_rawDataRange pins the WORD and BYTE immediate ranges at the
// toolchain's own boundaries: WORD takes [0, 0xffffffff] and BYTE [0, 0xff],
// and a value the source spells wider must reach the check whole. The read
// once truncated through int32, which rejected WORD $0xffffffff as -1 while
// accepting WORD $0x100000000 as 0.
func TestRISCV_rawDataRange(t *testing.T) {
asmOne := func(t *testing.T, stmt string) ([]byte, error) {
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·w(SB), NOSPLIT, $0\n\t"+stmt+"\n\tRET\n")
code, _, _, _, _, _, err := assembleRISCV(fn, nil)
return code, err
}
t.Run("word bounds", func(t *testing.T) {
code, err := asmOne(t, "WORD $0")
if err != nil {
t.Fatalf("WORD $0: %v", err)
}
if string(code[:4]) != "\x00\x00\x00\x00" {
t.Errorf("WORD $0 = % x", code[:4])
}
code, err = asmOne(t, "WORD $0xffffffff")
if err != nil {
t.Fatalf("WORD $0xffffffff must assemble, as go tool asm accepts it: %v", err)
}
if string(code[:4]) != "\xff\xff\xff\xff" {
t.Errorf("WORD $0xffffffff = % x", code[:4])
}
for _, w := range []string{"$-1", "$0x100000000", "$-4294967296"} {
if _, err := asmOne(t, "WORD "+w); err == nil {
t.Errorf("WORD %s must be rejected, as go tool asm rejects it", w)
}
}
})
t.Run("byte bounds", func(t *testing.T) {
code, err := asmOne(t, "BYTE $255")
if err != nil {
t.Fatalf("BYTE $255: %v", err)
}
if code[0] != 0xff {
t.Errorf("BYTE $255 = % x", code[:1])
}
for _, b := range []string{"$256", "$0x100000001", "$-1"} {
if _, err := asmOne(t, "BYTE "+b); err == nil {
t.Errorf("BYTE %s must be rejected: out of the byte range", b)
}
}
})
}
// TestRISCV_shiftImmediateRange pins the shift immediate at the toolchain's
// validation boundary: 0-63 on the doubleword forms, 0-31 on the word forms,
// values beyond rejected on sight rather than masked into the field.
func TestRISCV_shiftImmediateRange(t *testing.T) {
asmOne := func(t *testing.T, stmt string) error {
t.Helper()
fn := firstTextRISCV(t, "#include \"textflag.h\"\nTEXT ·s(SB), NOSPLIT, $0\n\t"+stmt+"\n\tRET\n")
_, _, _, _, _, _, err := assembleRISCV(fn, nil)
return err
}
for _, s := range []string{
"SLLI $63, X5, X6", "SLLI $0, X5, X6", "SRLI $63, X5", "SRAI $1, X5, X6",
"SLLIW $31, X5, X6", "SRLIW $31, X5", "SRAIW $1, X5, X6",
} {
if err := asmOne(t, s); err != nil {
t.Errorf("%s must assemble: %v", s, err)
}
}
for _, s := range []string{
"SLLI $64, X5, X6", "SLLI $-1, X5", "SLLI $0x100000000, X5, X6",
"SRLI $64, X5", "SRAI $-1, X5, X6",
"SLLIW $32, X5, X6", "SRLIW $-1, X5", "SRAIW $32, X5, X6",
} {
if err := asmOne(t, s); err == nil {
t.Errorf("%s must be rejected, as go tool asm rejects it", s)
}
}
}
+225
View File
@@ -0,0 +1,225 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"os"
"os/exec"
"path/filepath"
"regexp"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// riscv64ErrorCatalogues are the toolchain's own negative-case files for
// riscv64, relative to cmd/asm's testdata: riscv64error.s carries the
// preprocess-stage rejections and riscv64validation.s the validate-stage
// ones (register banks, compressed constraints, vector shapes).
var riscv64ErrorCatalogues = []string{"riscv64error.s", "riscv64validation.s"}
// riscv64AcceptedErrorShapes lists the toolchain's catalogue spellings gasm
// still accepts, each an acceptance superset with a documented reason. The
// list only shrinks: every tightening of the encoder moves spellings out of
// it, and a spelling reappearing here means a regression.
var riscv64AcceptedErrorShapes = []string{}
// TestRISCVToolchainErrorParity walks the toolchain's riscv64error.s and
// riscv64validation.s (Go 1.27) and requires gasm to reject every case the
// toolchain rejects, with an equivalent diagnostic, the documented acceptance
// supersets above excepted. A live Go toolchain is needed for the source
// files; the test skips without one or in -short.
func TestRISCVToolchainErrorParity(t *testing.T) {
goroot := riscv64Goroot(t)
dir := filepath.Join(goroot, "src", "cmd", "asm", "internal", "asm", "testdata")
allowed := map[string]bool{}
for _, s := range riscv64AcceptedErrorShapes {
allowed[s] = true
}
for _, name := range riscv64ErrorCatalogues {
data, err := os.ReadFile(filepath.Join(dir, name))
if err != nil {
t.Skip(err)
}
for raw := range strings.SplitSeq(string(data), "\n") {
line := strings.TrimSpace(raw)
if line == "" || strings.HasPrefix(line, "//") || strings.HasPrefix(line, "TEXT") || !strings.Contains(line, "ERROR") {
continue
}
body := line
if i := strings.Index(body, "//"); i >= 0 {
body = strings.TrimSpace(body[:i])
}
want := ""
if m := regexp.MustCompile(`ERROR "([^"]*)"`).FindStringSubmatch(line); m != nil {
want = m[1]
}
body = strings.ReplaceAll(body, "\t", " ")
body = strings.Join(strings.Fields(body), " ")
// A symbol reference needs the file's own declarations, which a
// one-line probe cannot carry.
if strings.Contains(body, "(SB)") {
continue
}
src := "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\t" + body + "\n\tRET\n"
f, perr := parser.Parse("errorparity.s", src)
if len(perr) > 0 {
continue // the parser already rejects the spelling
}
_, aerr := AssembleFileRISCV(f)
if aerr == nil {
if !allowed[body] {
t.Errorf("gasm accepts what the toolchain rejects: %s", body)
}
continue
}
if want != "" && !riscv64DiagEquivalent(want, aerr.Error()) {
t.Errorf("gasm rejects %s with an inequivalent diagnostic:\n toolchain: %s\n gasm: %s", body, want, aerr)
}
}
}
}
// riscv64DiagEquivalent answers whether gasm's rejection carries the same
// information as the toolchain's expected message. The two word the same
// facts differently: the toolchain writes "immediate out of range 0 to 31"
// where gasm may spell the field's own name, and gasm prefixes the mnemonic.
// Canonicalisation drops the subjects and connectives and compares the
// remaining token sequence.
func riscv64DiagEquivalent(want, got string) bool {
return riscv64DiagSubseq(riscv64DiagTokens(want), riscv64DiagTokens(got))
}
// riscv64DiagTokens lower-cases a diagnostic, strips punctuation and the
// connective tokens, and returns its words. "to" survives only between
// digits, where the range wording uses it.
func riscv64DiagTokens(msg string) []string {
msg = strings.ToLower(msg)
for _, r := range []string{"[", "]", ",", ".", ":", "\n", "(", ")"} {
msg = strings.ReplaceAll(msg, r, " ")
}
fields := strings.Fields(msg)
out := make([]string, 0, len(fields))
for i, w := range fields {
switch w {
case "the", "a", "operand", "immediate", "an", "for":
// subjects and articles carry no constraint
case "to":
if i > 0 && i+1 < len(fields) && riscv64IsNumeric(fields[i-1]) && riscv64IsNumeric(fields[i+1]) {
continue // the range connective
}
out = append(out, w)
default:
out = append(out, w)
}
}
return out
}
// riscv64DiagSubseq answers whether want is a subsequence of got.
func riscv64DiagSubseq(want, got []string) bool {
i := 0
for _, w := range got {
if i < len(want) && w == want[i] {
i++
}
}
return i == len(want)
}
// riscv64IsNumeric reports whether s parses as a signed decimal number, the
// token shape the range connective sits between.
func riscv64IsNumeric(s string) bool {
if s == "" {
return false
}
if s[0] == '-' || s[0] == '+' {
s = s[1:]
}
if s == "" {
return false
}
for _, r := range s {
if r < '0' || r > '9' {
return false
}
}
return true
}
// riscv64Goroot resolves the live toolchain root, the environment's own value
// first, `go env GOROOT` as the fallback.
func riscv64Goroot(t *testing.T) string {
t.Helper()
if goroot := os.Getenv("GOROOT"); goroot != "" {
return goroot
}
out, err := exec.Command("go", "env", "GOROOT").Output()
if err != nil {
t.Skipf("no GOROOT: %v", err)
}
return strings.TrimSpace(string(out))
}
// riscv64RejectedNames pins the names the toolchain's riscv64 table carries
// but refuses to encode: each is a "no encoding for instruction" under every
// operand shape GOARCH=riscv64 go tool asm accepts. gasm rejects them too,
// and the pair of rejections is the parity this catalogue asserts.
var riscv64RejectedNames = []string{"DUFFCOPY", "DUFFZERO", "PCALIGNMAX"}
// TestRISCVBacklogNameParity walks the audit's known-but-unencodable
// backlog: the names go tool asm recognises on riscv64 yet refuses to encode
// must be refused by gasm as well.
func TestRISCVBacklogNameParity(t *testing.T) {
if testing.Short() {
t.Skip("live toolchain backlog: skipped in -short mode")
}
for _, name := range riscv64RejectedNames {
if riscv64ProbeAssembles(name + " X5") {
t.Errorf("gasm encodes %s, which the toolchain refuses under every shape", name)
}
if riscv64ProbeAssembles(name) {
t.Errorf("gasm encodes %s, which the toolchain refuses under every shape", name)
}
}
}
// riscv64ProbeAssembles answers whether a single statement wrapped in its
// own function parses and assembles on riscv64.
func riscv64ProbeAssembles(body string) bool {
src := "#include \"textflag.h\"\n\nTEXT ·probe(SB), NOSPLIT, $0-0\n\t" + body + "\n\tRET\n"
f, perr := parser.Parse("backlogparity.s", src)
if len(perr) > 0 {
return false // the parser already rejects the line
}
_, aerr := AssembleFileRISCV(f)
return aerr == nil
}
// riscv64CatalogueSelfCheck guards the catalogue walk itself: every ERROR
// line in the two files must yield a probe body, so a formatting change in
// the toolchain's files cannot silently empty the parity set.
func TestRISCVErrorCatalogueSelfCheck(t *testing.T) {
if testing.Short() {
t.Skip("live toolchain catalogue: skipped in -short mode")
}
goroot := riscv64Goroot(t)
dir := filepath.Join(goroot, "src", "cmd", "asm", "internal", "asm", "testdata")
total := 0
for _, name := range riscv64ErrorCatalogues {
data, err := os.ReadFile(filepath.Join(dir, name))
if err != nil {
t.Skip(err)
}
n := strings.Count(string(data), "ERROR")
total += n
if n == 0 {
t.Errorf("%s carries no ERROR lines; the catalogue walk is empty", name)
}
}
if total < 1000 {
t.Errorf("catalogue carries %d ERROR lines, want at least 1000", total)
}
}
+17 -8
View File
@@ -7,7 +7,7 @@ import (
"fmt"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// RISC-V frame mapping, matching the Go toolchain's riscv64 backend.
@@ -34,6 +34,11 @@ import (
type riscvFrameInfo struct {
autosize int // the real SP adjustment (locals + saved LR)
// leaf mirrors the toolchain's cursym.Leaf: the body makes no call, so
// the link register still holds the caller's address. GETCALLERPC reads
// it there; a framed body reads the prologue's save at 0(SP) instead.
leaf bool
// Stack-split guard state: the toolchain emits the check for every
// non-NOSPLIT function whose autosize is nonzero (a zero autosize is
// "effectively NOSPLIT"); unlike amd64 and arm64 there is no leaf
@@ -44,12 +49,12 @@ type riscvFrameInfo struct {
// riscvComputeFrame derives the frame layout for a TEXT function.
func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
frame := frameSize(t)
if frame != 0 || !riscvIsLeaf(t) {
leaf := riscvIsLeaf(t)
if frame := frameSize(t); frame != 0 || !leaf {
// FixedFrameSize = 8: space for the saved link register. A
// zero-frame non-leaf function still opens an 8-byte frame for LR.
autosize := frame + 8
fi := riscvFrameInfo{autosize: autosize}
fi := riscvFrameInfo{autosize: autosize, leaf: leaf}
if !hasNoSplitFlag(t) {
fi.needSplit = true
switch {
@@ -63,7 +68,9 @@ func riscvComputeFrame(t *ast.Text) riscvFrameInfo {
}
return fi
}
return riscvFrameInfo{}
// A zero-frame leaf: no prologue, no guard, and the link register still
// holding the caller's address.
return riscvFrameInfo{leaf: leaf}
}
// hasNoSplitFlag reports whether the TEXT directive carries NOSPLIT.
@@ -96,8 +103,10 @@ func riscvIsLeaf(t *ast.Text) bool {
case "JALR":
// JALR rd, offset(rs1) links when the destination register (the
// first operand) is X1; JALR rs1, rd links when the second
// register is X1; JALR offset(rs1) always links to X1.
if len(in.Operands) == 1 {
// register is X1; JALR offset(rs1) always links to X1. A bare
// operand-less spelling is malformed and encoding rejects it;
// conservatively count it as a link so the frame stays honest.
if len(in.Operands) <= 1 {
return false
}
if isMemOperand(in.Operands[1]) {
@@ -344,7 +353,7 @@ func riscvGuard(fi riscvFrameInfo) ([]byte, Reloc, error) {
off := int32(fi.autosize - stackSmall)
mov := encodeRISCVLoadImm(7, off)
out = append(out, mov...)
addiLen := riscvItypeImmediateSize("ADDI", -off)
addiLen := riscvItypeImmediateSize("ADDI", 7, 2, -off)
out = append(out, wordLE(riscvBType(riscvEnc{0x63, 0x06, 0x00}, 2, 7, int32(addiLen+8)))...)
addi, err := encodeRISCVItypeImmediate("ADDI", riscvEnc{0x13, 0x0, 0x00}, 7, 2, -off)
if err != nil {
+1 -1
View File
@@ -6,7 +6,7 @@ package asm
import (
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestRISCVFrameSpadjAndLines checks that a framed function records its
+1 -1
View File
@@ -13,7 +13,7 @@ import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestGOObjectRISCVCallReloc checks that CALL sym(SB) emits a single JAL
+128
View File
@@ -0,0 +1,128 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"strings"
"testing"
)
// TestRISCVBigMemOffset_Differential proves the memory-offset expansion
// against the oracle: a load or store whose offset leaves the signed 12-bit
// span materialises the high part in the assembler's temporary register
// (C.LUI, or LUI beyond its six-bit immediate) followed by C.ADD of the base,
// and accesses through it, exactly as the toolchain's instructionsForLoad and
// instructionsForStore synthesise. The battery spans both sides of the
// 12-bit edge, both int32 boundaries, the compressed and uncompressed LUI
// halves, every width family and the MOV spellings, under a prologue too so
// the size accounting keeps the layout honest.
func TestRISCVBigMemOffset_Differential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·bigmem(SB), NOSPLIT, $16
LD 2047(X6), X5 // single word: the edge that still fits
LD 2048(X6), X5 // hi 1, lo -2048
LD 4096(X6), X5 // hi 1, lo 0
SD X5, 8192(X6) // store side of the same split
FLD 4096(X6), F5 // FP widths share the expansion
FSD F5, 8192(X6)
MOVB 4097(X6), X8 // the MOV widths route through the same load
MOVHU 4098(X6), X9
MOV X7, 16384(X6) // and the same store
LD 1048576(X6), X10 // hi 256: LUI, not C.LUI
LD -4097(X6), X11 // hi -1, lo -1
SD X11, -8192(X6)
LD 2147483647(X6), X12 // int32 upper boundary
LD -2147483648(X6), X13 // int32 lower boundary
RET
`
path := writeRISCVSrc(t, "bigmem_riscv64.s", src)
assertRISCVDifferential(t, path, src, "bigmem")
}
// TestRISCVBigMemOffsetFrame_Differential proves the expansion under the
// stack-split guard: a function that opens a frame and calls out gets the
// guard, the prologue and the epilogue ahead of the body, so every body
// offset now depends on the expansion's size accounting having kept the
// layout in step with the emitted bytes.
func TestRISCVBigMemOffsetFrame_Differential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·bigframe(SB), $16
CALL extcal(SB)
LD 4096(X6), X5
SD X5, 4096(X6)
FLD 8192(X6), F5
FSD F5, 8192(X6)
RET
`
path := writeRISCVSrc(t, "bigframe_riscv64.s", src)
assertRISCVDifferential(t, path, src, "bigframe")
}
// TestRISCVBigMemOffsetRejections pins the honest rejections the expansion
// cannot carry: a constant beyond the signed 32-bit span is the toolchain's
// "constant too large" (its Split32BitImmediate has no wider split), and a
// JALR displacement beyond the signed 12-bit span is its I-type range check,
// because the jump would land somewhere else entirely.
func TestRISCVBigMemOffsetRejections(t *testing.T) {
cases := []struct {
name string
src string
want string
}{
{
name: "store beyond int32",
src: "\tSD X5, 4294967295(X6)\n",
want: "SD: constant 4294967295 too large",
},
{
name: "load below int32",
src: "\tLD -2147483649(X6), X5\n",
want: "LD: constant -2147483649 too large",
},
{
name: "MOV load beyond int32",
src: "\tMOV 4294967296(X6), X5\n",
want: "MOV: constant 4294967296 too large",
},
{
name: "MOV store beyond int32",
src: "\tMOV X5, 4294967296(X6)\n",
want: "MOV: constant 4294967296 too large",
},
{
name: "FP store beyond int32",
src: "\tFSD F5, 4294967296(X6)\n",
want: "FSD: constant 4294967296 too large",
},
{
name: "FP load beyond int32",
src: "\tFLD 4294967296(X6), F5\n",
want: "FLD: constant 4294967296 too large",
},
{
name: "JALR displacement above 12 bits",
src: "\tJALR 4096(X7)\n",
want: "JALR: signed immediate 4096 must be in range [-2048, 2047] (12 bits)",
},
{
name: "JALR displacement below 12 bits",
src: "\tJALR X5, -2049(X7)\n",
want: "JALR: signed immediate -2049 must be in range [-2048, 2047] (12 bits)",
},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
fn := firstTextRISCV(t, "#include \"textflag.h\"\n\nTEXT ·r(SB), NOSPLIT, $0\n"+tc.src+"\tRET\n")
_, _, _, _, _, _, err := assembleRISCV(fn, nil)
if err == nil {
t.Fatalf("source assembled, want rejection %q", tc.want)
}
if !strings.Contains(err.Error(), tc.want) {
t.Errorf("error %q does not carry %q", err.Error(), tc.want)
}
})
}
}
Loaded 100 of 295 files, more files were not shown because too many files have changed in this diff. Show more