Files
gasm-sdk/disasm/parity_test.go
T
petrbalvin a7f9d5eb67 feat(disasm): render the loong64 families x/arch names Unknown
x/arch's Plan 9 renderer decodes thirty-odd scalar loong64 operations
perfectly but prints them as "Unknown OP args", and names the
sign-extension pair EXT.W.B/EXT.W.H "?".  The supplementary naming
pass re-renders them with the toolchain's own spellings: the families
whose operand order x/arch already prints the Plan 9 way trade only
the mnemonic, and the pointer loads and stores, the acquire loads,
the release stores, PRELD and ALSL are rebuilt from the decoded
arguments with the toolchain's operand order and its raw displacement
reading.  ADDU16I.D, a macro helper the assembler never takes as
input, keeps the decoder's Unknown render.

The loong64 parity fixture grows from 73 to 385 rows, pinning every
unique four-byte corpus word the decoder accepts, and the LL/SC
displacement divergence in the encoder is documented for asm.

Assisted-by: GLM 5.3
2026-10-07 13:50:54 +02:00

390 lines
15 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package disasm
import (
"encoding/hex"
"os"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The parity fixtures carry a deterministic slice of the Go toolchain's own
// view of the encoder corpus, one instruction per line as
//
// HEXBYTES TOOLCHAINTEXT
//
// where HEXBYTES are the instruction's bytes in storage order and
// TOOLCHAINTEXT is what `go tool objdump` prints for them (its decoder is
// the ground truth). The slices were sampled from GOROOT's assembler test
// data with a fixed stride: amd64enc.s (the x87, SSE/MMX, VEX, XSAVE and
// system families), amd64enc_extra.s, amd64.s, arm64enc.s, riscv64.s and
// loong64enc1.s.
//
// parity_amd64_unlisted.txt covers the amd64 lines the objdump chunk oracle
// cannot attribute: the listing fragments at the opcodes its vendored decoder
// refuses and desynchronises around. Their bytes are the corpus's own
// expected-encoding comments, which the toolchain's assembler test suite
// machine-checks, and their text is the decoder's rendering, pinned here so a
// decoder bump cannot move it silently; the round-trip test is the oracle
// that keeps the text honest against the encoder.
var parityFixtures = []struct {
arch arch.Arch
file string
}{
{arch.AMD64, "testdata/parity_amd64.txt"},
{arch.AMD64, "testdata/parity_amd64_extra.txt"},
{arch.AMD64, "testdata/parity_amd64_sys.txt"},
{arch.AMD64, "testdata/parity_amd64_unlisted.txt"},
{arch.ARM64, "testdata/parity_arm64.txt"},
{arch.RISCV, "testdata/parity_riscv64.txt"},
{arch.LOONG64, "testdata/parity_loong64.txt"},
}
// TestOracleKnownBytes pins the oracle's own fidelity on a few encodings
// whose bytes the sources name outright, before the bulk fixtures are
// trusted: each row was cross-checked against `go tool objdump` over an
// object the Go toolchain's own assembler produced.
func TestOracleKnownBytes(t *testing.T) {
for _, tt := range []struct {
a arch.Arch
code []byte
text string
}{
// go tool asm: MOVQ AX, CX -> 4889c1.
{arch.AMD64, []byte{0x48, 0x89, 0xc1}, "MOVQ AX, CX"},
// arm64enc.s: "ADCW ZR, R8, R10 // 1a1f010a".
{arch.ARM64, []byte{0x0a, 0x01, 0x1f, 0x1a}, "ADCW ZR, R8, R10"},
// riscv64.s: "ADDI $2047, X5 // 9382f27f".
{arch.RISCV, []byte{0x93, 0x82, 0xf2, 0x7f}, "ADDI $2047, X5, X5"},
// The loong64 RET the runtime emits: 0x4c000020.
{arch.LOONG64, []byte{0x20, 0x00, 0x00, 0x4c}, "RET"},
} {
ins, err := Decode(tt.a, tt.code, 0)
if err != nil {
t.Errorf("%s % x: %v", tt.a, tt.code, err)
continue
}
if ins.Text != tt.text || ins.Len != len(tt.code) {
t.Errorf("%s % x: %q (%d bytes), want %q (%d)",
tt.a, tt.code, ins.Text, ins.Len, tt.text, len(tt.code))
}
}
}
// TestToolchainDecodeParity requires the decoder to reproduce the
// toolchain's text for every fixture instruction, and to agree on the
// instruction's length: both views of the same bytes.
func TestToolchainDecodeParity(t *testing.T) {
for _, fx := range parityFixtures {
lines := readFixture(t, fx.file)
for i, ln := range lines {
// The amd64 branch rendering carries an absolute target, which
// is a property of the decode base and not of the bytes; those
// lines are pinned by TestBranchTargetConvention instead.
if fx.arch == arch.AMD64 && relativeBranch(ln.text) {
continue
}
ins, err := Decode(fx.arch, ln.code, 0)
if err != nil {
t.Errorf("%s line %d: %v", fx.file, i+1, err)
continue
}
if ins.Len != len(ln.code) {
t.Errorf("%s line %d: % x decoded to %d bytes, toolchain says %d",
fx.file, i+1, ln.code, ins.Len, len(ln.code))
continue
}
if ins.Text != ln.text {
t.Errorf("%s line %d: % x decodes to %q, toolchain says %q",
fx.file, i+1, ln.code, ins.Text, ln.text)
}
}
}
}
// TestBranchTargetConvention pins the rendering convention for relative
// branches: the target prints as an absolute address in the address space
// of the addr argument, which is exactly what `go tool objdump` does. Both
// views are faithful to the bytes; they must be decoded at the same base to
// compare.
func TestBranchTargetConvention(t *testing.T) {
ins, err := Decode(arch.AMD64, []byte{0x72, 0x02}, 0x1ded)
if err != nil {
t.Fatal(err)
}
// The toolchain prints "JB 0x1df1" for these bytes at the same base.
if ins.Text != "JB 0x1df1" {
t.Errorf("JB at base 0x1ded: %q, want JB 0x1df1", ins.Text)
}
// The toolchain's listing documents relocations after the text
// ("[3:7]R_PCREL:foo+4"); that is bookkeeping of the listing, not part
// of the disassembler's text, and the fixtures carry it stripped.
}
// relativeBranch reports whether the rendered text carries a PC-relative
// or position-dependent target. Such an operand is an address, not a
// datum: it is faithful to the bytes only at the address the instruction
// sits at, so re-assembling it inside a fresh wrapper cannot be expected
// to reproduce them. The amd64 J- and LOOP/CALL families, plus the
// arm64, riscv64 and loong64 "(PC)" forms, all fall here.
func relativeBranch(text string) bool {
if strings.Contains(text, "(PC)") || strings.Contains(text, "(RPC)") {
return true
}
// Mnemonic is the first whitespace-free token.
m := text
if i := strings.IndexAny(text, " \t"); i >= 0 {
m = text[:i]
}
switch {
case strings.HasPrefix(m, "J"), // amd64 Jcc, JMP; arm64/riscv JMP
m == "CALL", m == "XBEGIN",
m == "LOOP", m == "LOOPE", m == "LOOPNE", m == "LOOPZ", m == "LOOPNZ",
m == "CBNZ", m == "CBZ", m == "TBZ", m == "TBNZ", m == "BL":
return true
}
return false
}
// toolchainRenderNames lists mnemonics the toolchain's own renderer prints
// but the gasm encoder cannot assemble. Two kinds sit here: the spellings
// golang.org/x/arch keeps in its non-Plan 9 form (MOVZX and MOVSX where
// the assembler spells MOVBLZX and MOVBLSX, the CMOV* family, the packed
// shuffles it names MOVDQA and MOVDQU, the CVT conversions), so the listing
// text is not assembler input at all, and the handful of Plan 9 names the
// encoder has no table entry for yet (CMPPD, STOSB, PUSHL). A parse
// failure outside this set is a test failure, so the set shrinks as the
// encoder's vocabulary grows.
var toolchainRenderNames = map[string]bool{
"MOVZX": true, "MOVSX": true, "MOVSXD": true,
"SHLDL": true, "SHLDW": true, "SHLDQ": true,
"SHRDL": true, "SHRDW": true, "SHRDQ": true,
"MOVBE": true, "LSL": true, "LAR": true,
"CMOVA": true, "CMOVAE": true, "CMOVB": true, "CMOVBE": true,
"CMOVE": true, "CMOVG": true, "CMOVGE": true, "CMOVL": true,
"CMOVNE": true, "CMOVNO": true, "CMOVNP": true, "CMOVNS": true,
"CMOVO": true, "CMOVP": true, "CMOVS": true,
"PUNPCKLWD": true, "PUNPCKLDQ": true, "PUNPCKHWD": true, "PUNPCKHDQ": true,
"PSLLD": true, "PSRLD": true, "PSRAD": true,
"PMULUDQ": true, "PMADDWD": true, "PACKSSDW": true,
"MOVDQA": true, "MOVDQU": true, "MASKMOVDQU": true,
"LSS": true, "LGS": true, "LFS": true,
"WRGSBASE": true, "RDFSBASE": true, "STR": true, "SLDT": true,
"RDRAND": true, "POPF": true, "LRET": true,
"WRFSBASE": true, "FCOM": true,
"FDIV": true, "FADD": true, "FRINTS": true, "FRINTM": true,
"CMPPD": true, "STOSB": true, "PUSHL": true,
"MOVSD_XMM": true, "CMPSD_XMM": true,
"CMPPS": true, "CMPSS": true, "CMPSB": true, "MOVSQ": true, "OUTSW": true,
"SLLIUW": true, "SH1ADD": true, "SH2ADDUW": true,
"BSETI": true, "BEXTI": true, "CLMULR": true,
"CPOPW": true, "ORCB": true, "FABS": true, "FABSS": true, "FMUL": true,
"MRS": true, "LU12IW": true,
"CVTTSS2SIL": true, "CVTTSD2SIL": true,
"CVTTPS2DQ": true, "CVTTPD2DQ": true,
"CVTSS2SIL": true, "CVTSI2SSQ": true, "CVTSI2SSL": true,
"CVTSI2SDQ": true, "CVTSI2SDL": true, "CVTSD2SIL": true,
"CVTPS2DQ": true, "CVTPD2DQ": true, "CVTDQ2PS": true, "CVTDQ2PD": true,
}
// encoderDivergences lists fixture lines whose disassembler text re-encodes
// to a DIFFERENT reading, so neither the byte-exact invariant nor the
// fixed-point one can hold. Every entry is a finding for the asm package's
// encoder, reported and not fixed here; the disassembler text itself is the
// toolchain's. amd64: CMOVLE encodes the CMOVE condition code; MOVQ to a
// memory operand drops the FS segment prefix; MOVQ2DQ takes the F2 prefix
// and lands in MOVDQ2Q; MOVD re-encodes through the 64-bit MOVQ alias; the
// byte-register lines that render with the L suffix encode the byte form
// since the operand-width reconciliation, MOVL $0x7, DL keeping only the
// legal-encoding difference (the toolchain's own table says "c6c207 or
// b207", and go tool asm emits b207, so the fixed-point invariant holds
// where the pinned bytes took the alternative). arm64: the CRC32 forms
// take the wrong Rm; the register-indexed load and store forms lose the
// index operand; BFXIL encodes as BFI with shifted immediates. loong64:
// the LL and SC displacements encode scaled where the toolchain takes them
// raw ("LL 1024(R5), R4" assembles to a field of 256, and the re-decoded
// text reads 256(R5)). riscv64: FSGNJXS encodes as FMIN.S; FCLASSS and
// FCLASSD encode as MOVF; the AUIPC immediate loses its high bits. Keyed
// by the fixture text.
var encoderDivergences = map[string]bool{
"CMOVLE 0(BX), DX": true,
"MOVQ FS:0, DX": true,
"MOVQ2DQ M2, X11": true,
"MOVL $0x7, DL": true,
"BFXIL $26, R8, $16, R20": true,
"CRC32B R17, R8, R16": true,
"CRC32CB R19, R27, R22": true,
"MOVBU (R27)(R23), R14": true,
"MOVHU (R5)(R25.SXTW), R15": true,
"MOVB (R5)(R15), R16": true,
"MOVD R27, (R5)(R15.UXTW<<3)": true,
"MOVH R11, (R27)(R14.SXTW<<1)": true,
"SC R4, 1024(R5)": true,
"LL 1024(R5), R4": true,
"LLV 1024(R5), R4": true,
"SCV R4, 1024(R5)": true,
"FSGNJXS F1, F0, F2": true,
"FCLASSS F0, X5": true,
"FCLASSD F0, X5": true,
"AUIPC $524287, X10": true,
// From the unlisted amd64 fixture: MOVD re-encodes through the 64-bit
// MOVQ alias; POPW drops the operand-size prefix. Findings for asm,
// reported and not fixed here.
"POPW FS": true,
"POPW GS": true,
"MOVD DX, M2": true,
"MOVD R11, M2": true,
"MOVD DX, M3": true,
"MOVD R11, M3": true,
"MOVD M2, DX": true,
"MOVD M3, DX": true,
"MOVD M2, R11": true,
"MOVD M3, R11": true,
"MOVD X2, DX": true,
"MOVD X11, DX": true,
"MOVD X2, R11": true,
"MOVD X11, R11": true,
"MOVD DX, X2": true,
"MOVD R11, X2": true,
"MOVD DX, X11": true,
"MOVD R11, X11": true,
}
// TestDisassemblyRoundTrip is the cheap invariant over the same slice:
// dis(assemble(x)) must re-encode to x's bytes. The listing text goes back
// through the parser and the encoder inside a fresh function, and the
// result must match the fixture bytes exactly.
//
// Three documented kinds of line cannot carry the byte-exact invariant and
// degrade to the weaker fixed-point check, decode(assemble(text)) == text:
// position-dependent branches (relativeBranch), mnemonics the toolchain
// renderer leaves outside Plan 9 vocabulary (toolchainRenderNames), and
// the encoder divergences reported for asm (encoderDivergences) only for
// the byte comparison. A line of none of these kinds must round-trip byte
// for byte.
func TestDisassemblyRoundTrip(t *testing.T) {
for _, fx := range parityFixtures {
lines := readFixture(t, fx.file)
skipped := 0
for i, ln := range lines {
if relativeBranch(ln.text) {
skipped++
continue
}
m := mnemonic(ln.text)
switch {
case toolchainRenderNames[m]:
// The toolchain renders a spelling the encoder cannot take.
skipped++
continue
case fx.arch == arch.RISCV && strings.HasPrefix(m, "V"):
// The RISC-V vector slice of the corpus is the open encoder
// roadmap item; its mnemonics do not assemble yet. The
// decode side is fully covered by the parity test, which
// never parses.
skipped++
continue
case m == "Unknown":
// The x/arch loong64 renderer names opcodes it has no Go
// spelling for "Unknown ..."; that text is the toolchain's
// own and no assembler input.
skipped++
continue
}
src := "TEXT \u00b7k(SB), NOSPLIT, $0\n\t" + ln.text + "\n\tRET\n"
f, errs := parser.Parse("k.s", src)
if len(errs) > 0 {
t.Errorf("%s line %d (%s): parse: %v", fx.file, i+1, ln.text, errs[0])
continue
}
img, err := assembleFor(fx.arch, f)
if err != nil {
t.Errorf("%s line %d (%s): assemble: %v", fx.file, i+1, ln.text, err)
continue
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if len(code) >= len(ln.code) && string(code[:len(ln.code)]) == string(ln.code) {
continue
}
// Weaker invariant for the documented classes: the text must be
// a fixed point of decode followed by encode, that is, the
// re-encoded bytes must carry the same reading. Byte-equality
// divergences outside the encoder set are real failures.
back, err := Decode(fx.arch, code, 0)
if err != nil || back.Text != ln.text {
if encoderDivergences[ln.text] {
continue
}
got := "undecodable"
if err == nil {
got = back.Text
}
t.Errorf("%s line %d (%s): round trip re-decodes as %q, bytes % x",
fx.file, i+1, ln.text, got, code)
}
}
if skipped > 0 {
t.Logf("%s: %d of %d lines excluded (branches and toolchain-only spellings)", fx.file, skipped, len(lines))
}
}
}
// mnemonic returns the first whitespace-free token of the text.
func mnemonic(text string) string {
if i := strings.IndexAny(text, " \t"); i >= 0 {
return text[:i]
}
return text
}
type fixtureLine struct {
code []byte
text string
}
func readFixture(t *testing.T, name string) []fixtureLine {
t.Helper()
data, err := os.ReadFile(name)
if err != nil {
t.Fatalf("%s: %v", name, err)
}
var out []fixtureLine
for i, line := range strings.Split(strings.TrimSuffix(string(data), "\n"), "\n") {
if line == "" {
continue
}
tab := strings.IndexByte(line, '\t')
if tab < 0 {
t.Fatalf("%s line %d: no tab separator", name, i+1)
}
code, err := hex.DecodeString(line[:tab])
if err != nil {
t.Fatalf("%s line %d: bad hex: %v", name, i+1, err)
}
out = append(out, fixtureLine{code: code, text: line[tab+1:]})
}
return out
}
// assemble assembles the parsed file with the encoder for a.
func assembleFor(a arch.Arch, f *ast.File) (*asm.Image, error) {
switch a {
case arch.ARM64:
return asm.AssembleFileARM64(f)
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.LOONG64:
return asm.AssembleFileLOONG64(f)
default:
return asm.AssembleFile(f)
}
}