// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package disasm import ( "encoding/hex" "os" "strings" "testing" "sourcedock.dev/petrbalvin/gasm-sdk/arch" "sourcedock.dev/petrbalvin/gasm-sdk/asm" "sourcedock.dev/petrbalvin/gasm-sdk/ast" "sourcedock.dev/petrbalvin/gasm-sdk/parser" ) // The parity fixtures carry a deterministic slice of the Go toolchain's own // view of the encoder corpus, one instruction per line as // // HEXBYTES\tTOOLCHAINTEXT // // where HEXBYTES are the instruction's bytes in storage order and // TOOLCHAINTEXT is what `go tool objdump` prints for them (its decoder is // the ground truth). The slices were sampled from GOROOT's assembler test // data with a fixed stride: amd64enc.s (the x87, SSE/MMX, VEX, XSAVE and // system families), amd64enc_extra.s, amd64.s, arm64enc.s, riscv64.s and // loong64enc1.s. var parityFixtures = []struct { arch arch.Arch file string }{ {arch.AMD64, "testdata/parity_amd64.txt"}, {arch.AMD64, "testdata/parity_amd64_extra.txt"}, {arch.AMD64, "testdata/parity_amd64_sys.txt"}, {arch.ARM64, "testdata/parity_arm64.txt"}, {arch.RISCV, "testdata/parity_riscv64.txt"}, {arch.LOONG64, "testdata/parity_loong64.txt"}, } // TestOracleKnownBytes pins the oracle's own fidelity on a few encodings // whose bytes the sources name outright, before the bulk fixtures are // trusted: each row was cross-checked against `go tool objdump` over an // object the Go toolchain's own assembler produced. func TestOracleKnownBytes(t *testing.T) { for _, tt := range []struct { a arch.Arch code []byte text string }{ // go tool asm: MOVQ AX, CX -> 4889c1. {arch.AMD64, []byte{0x48, 0x89, 0xc1}, "MOVQ AX, CX"}, // arm64enc.s: "ADCW ZR, R8, R10 // 1a1f010a". {arch.ARM64, []byte{0x0a, 0x01, 0x1f, 0x1a}, "ADCW ZR, R8, R10"}, // riscv64.s: "ADDI $2047, X5 // 9382f27f". {arch.RISCV, []byte{0x93, 0x82, 0xf2, 0x7f}, "ADDI $2047, X5, X5"}, // The loong64 RET the runtime emits: 0x4c000020. {arch.LOONG64, []byte{0x20, 0x00, 0x00, 0x4c}, "RET"}, } { ins, err := Decode(tt.a, tt.code, 0) if err != nil { t.Errorf("%s % x: %v", tt.a, tt.code, err) continue } if ins.Text != tt.text || ins.Len != len(tt.code) { t.Errorf("%s % x: %q (%d bytes), want %q (%d)", tt.a, tt.code, ins.Text, ins.Len, tt.text, len(tt.code)) } } } // TestToolchainDecodeParity requires the decoder to reproduce the // toolchain's text for every fixture instruction, and to agree on the // instruction's length: both views of the same bytes. func TestToolchainDecodeParity(t *testing.T) { for _, fx := range parityFixtures { lines := readFixture(t, fx.file) for i, ln := range lines { // The amd64 branch rendering carries an absolute target, which // is a property of the decode base and not of the bytes; those // lines are pinned by TestBranchTargetConvention instead. if fx.arch == arch.AMD64 && relativeBranch(ln.text) { continue } ins, err := Decode(fx.arch, ln.code, 0) if err != nil { t.Errorf("%s line %d: %v", fx.file, i+1, err) continue } if ins.Len != len(ln.code) { t.Errorf("%s line %d: % x decoded to %d bytes, toolchain says %d", fx.file, i+1, ln.code, ins.Len, len(ln.code)) continue } if ins.Text != ln.text { t.Errorf("%s line %d: % x decodes to %q, toolchain says %q", fx.file, i+1, ln.code, ins.Text, ln.text) } } } } // TestBranchTargetConvention pins the rendering convention for relative // branches: the target prints as an absolute address in the address space // of the addr argument, which is exactly what `go tool objdump` does. Both // views are faithful to the bytes; they must be decoded at the same base to // compare. func TestBranchTargetConvention(t *testing.T) { ins, err := Decode(arch.AMD64, []byte{0x72, 0x02}, 0x1ded) if err != nil { t.Fatal(err) } // The toolchain prints "JB 0x1df1" for these bytes at the same base. if ins.Text != "JB 0x1df1" { t.Errorf("JB at base 0x1ded: %q, want JB 0x1df1", ins.Text) } // The toolchain's listing documents relocations after the text // ("[3:7]R_PCREL:foo+4"); that is bookkeeping of the listing, not part // of the disassembler's text, and the fixtures carry it stripped. } // relativeBranch reports whether the rendered text carries a PC-relative // or position-dependent target. Such an operand is an address, not a // datum: it is faithful to the bytes only at the address the instruction // sits at, so re-assembling it inside a fresh wrapper cannot be expected // to reproduce them. The amd64 J- and LOOP/CALL families, plus the // arm64, riscv64 and loong64 "(PC)" forms, all fall here. func relativeBranch(text string) bool { if strings.Contains(text, "(PC)") || strings.Contains(text, "(RPC)") { return true } // Mnemonic is the first whitespace-free token. m := text if i := strings.IndexAny(text, " \t"); i >= 0 { m = text[:i] } switch { case strings.HasPrefix(m, "J"), // amd64 Jcc, JMP; arm64/riscv JMP m == "CALL", m == "XBEGIN", m == "LOOP", m == "LOOPE", m == "LOOPNE", m == "LOOPZ", m == "LOOPNZ", m == "CBNZ", m == "CBZ", m == "TBZ", m == "TBNZ", m == "BL": return true } return false } // toolchainRenderNames lists mnemonics the toolchain's own renderer prints // but the gasm encoder cannot assemble. Two kinds sit here: the spellings // golang.org/x/arch keeps in its non-Plan 9 form (MOVZX and MOVSX where // the assembler spells MOVBLZX and MOVBLSX, the CMOV* family, the packed // shuffles it names MOVDQA and MOVDQU, the CVT conversions), so the listing // text is not assembler input at all, and the handful of Plan 9 names the // encoder has no table entry for yet (CMPPD, STOSB, PUSHL). A parse // failure outside this set is a test failure, so the set shrinks as the // encoder's vocabulary grows. var toolchainRenderNames = map[string]bool{ "MOVZX": true, "MOVSX": true, "MOVSXD": true, "SHLDL": true, "SHLDW": true, "SHLDQ": true, "SHRDL": true, "SHRDW": true, "SHRDQ": true, "MOVBE": true, "LSL": true, "LAR": true, "CMOVA": true, "CMOVAE": true, "CMOVB": true, "CMOVBE": true, "CMOVE": true, "CMOVG": true, "CMOVGE": true, "CMOVL": true, "CMOVNE": true, "CMOVNO": true, "CMOVNP": true, "CMOVNS": true, "CMOVO": true, "CMOVP": true, "CMOVS": true, "PUNPCKLWD": true, "PUNPCKLDQ": true, "PUNPCKHWD": true, "PUNPCKHDQ": true, "PSLLD": true, "PSRLD": true, "PSRAD": true, "PMULUDQ": true, "PMADDWD": true, "PACKSSDW": true, "MOVDQA": true, "MOVDQU": true, "MASKMOVDQU": true, "LSS": true, "LGS": true, "LFS": true, "WRGSBASE": true, "RDFSBASE": true, "STR": true, "SLDT": true, "RDRAND": true, "POPF": true, "LRET": true, "FDIV": true, "FADD": true, "FRINTS": true, "FRINTM": true, "CMPPD": true, "STOSB": true, "PUSHL": true, "MOVSD_XMM": true, "CMPSD_XMM": true, "CMPPS": true, "CMPSS": true, "CMPSB": true, "MOVSQ": true, "OUTSW": true, "SLLIUW": true, "SH1ADD": true, "SH2ADDUW": true, "BSETI": true, "BEXTI": true, "CLMULR": true, "CPOPW": true, "ORCB": true, "FABS": true, "FABSS": true, "FMUL": true, "MRS": true, "LU12IW": true, "CVTTSS2SIL": true, "CVTTSD2SIL": true, "CVTTPS2DQ": true, "CVTTPD2DQ": true, "CVTSS2SIL": true, "CVTSI2SSQ": true, "CVTSI2SSL": true, "CVTSI2SDQ": true, "CVTSI2SDL": true, "CVTSD2SIL": true, "CVTPS2DQ": true, "CVTPD2DQ": true, "CVTDQ2PS": true, "CVTDQ2PD": true, } // encoderDivergences lists fixture lines whose disassembler text re-encodes // to a DIFFERENT reading, so neither the byte-exact invariant nor the // fixed-point one can hold. Every entry is a finding for the asm package's // encoder, reported and not fixed here; the disassembler text itself is the // toolchain's. amd64: CMOVLE encodes the CMOVE condition code; MOVQ to a // memory operand drops the FS segment prefix; MOVQ2DQ takes the F2 prefix // and lands in MOVDQ2Q; CRC32 with a 16-bit register widens to 32 bits. // arm64: the CRC32 forms take the wrong Rm; the register-indexed load and // store forms lose the index operand; BFXIL encodes as BFI with shifted // immediates. loong64: the SC displacement encodes unscaled. // riscv64: FSGNJXS encodes as FMIN.S; FCLASSS and FCLASSD encode as MOVF; // the AUIPC immediate loses its high bits. Keyed by the fixture text. var encoderDivergences = map[string]bool{ "CMOVLE 0(BX), DX": true, "MOVQ FS:0, DX": true, "MOVQ2DQ M2, X11": true, "CRC32 DL, R11": true, "XADDL DL, DL": true, "XCHGL DL, DL": true, "CMPL AL, $0x7": true, "CMPXCHGL DL, DL": true, "ANDL $0x7, AL": true, "SBBL $0x7, DL": true, "SBBL DL, R11": true, "SUBL $0x7, AL": true, "TESTL R11, DL": true, "BFXIL $26, R8, $16, R20": true, "CRC32B R17, R8, R16": true, "CRC32CB R19, R27, R22": true, "MOVBU (R27)(R23), R14": true, "MOVHU (R5)(R25.SXTW), R15": true, "MOVB (R5)(R15), R16": true, "MOVD R27, (R5)(R15.UXTW<<3)": true, "MOVH R11, (R27)(R14.SXTW<<1)": true, "SC R4, 1024(R5)": true, "FSGNJXS F1, F0, F2": true, "FCLASSS F0, X5": true, "FCLASSD F0, X5": true, "AUIPC $524287, X10": true, } // TestDisassemblyRoundTrip is the cheap invariant over the same slice: // dis(assemble(x)) must re-encode to x's bytes. The listing text goes back // through the parser and the encoder inside a fresh function, and the // result must match the fixture bytes exactly. // // Three documented kinds of line cannot carry the byte-exact invariant and // degrade to the weaker fixed-point check, decode(assemble(text)) == text: // position-dependent branches (relativeBranch), mnemonics the toolchain // renderer leaves outside Plan 9 vocabulary (toolchainRenderNames), and // the encoder divergences reported for asm (encoderDivergences) only for // the byte comparison. A line of none of these kinds must round-trip byte // for byte. func TestDisassemblyRoundTrip(t *testing.T) { for _, fx := range parityFixtures { lines := readFixture(t, fx.file) skipped := 0 for i, ln := range lines { if relativeBranch(ln.text) { skipped++ continue } m := mnemonic(ln.text) switch { case toolchainRenderNames[m]: // The toolchain renders a spelling the encoder cannot take. skipped++ continue case fx.arch == arch.RISCV && strings.HasPrefix(m, "V"): // The RISC-V vector slice of the corpus is the open encoder // roadmap item; its mnemonics do not assemble yet. The // decode side is fully covered by the parity test, which // never parses. skipped++ continue case m == "Unknown": // The x/arch loong64 renderer names opcodes it has no Go // spelling for "Unknown ..."; that text is the toolchain's // own and no assembler input. skipped++ continue } src := "TEXT \u00b7k(SB), NOSPLIT, $0\n\t" + ln.text + "\n\tRET\n" f, errs := parser.Parse("k.s", src) if len(errs) > 0 { t.Errorf("%s line %d (%s): parse: %v", fx.file, i+1, ln.text, errs[0]) continue } img, err := assembleFor(fx.arch, f) if err != nil { t.Errorf("%s line %d (%s): assemble: %v", fx.file, i+1, ln.text, err) continue } fn := img.Funcs[0] code := img.Code[fn.Offset : fn.Offset+fn.Size] if len(code) >= len(ln.code) && string(code[:len(ln.code)]) == string(ln.code) { continue } // Weaker invariant for the documented classes: the text must be // a fixed point of decode followed by encode, that is, the // re-encoded bytes must carry the same reading. Byte-equality // divergences outside the encoder set are real failures. back, err := Decode(fx.arch, code, 0) if err != nil || back.Text != ln.text { if encoderDivergences[ln.text] { continue } got := "undecodable" if err == nil { got = back.Text } t.Errorf("%s line %d (%s): round trip re-decodes as %q, bytes % x", fx.file, i+1, ln.text, got, code) } } if skipped > 0 { t.Logf("%s: %d of %d lines excluded (branches and toolchain-only spellings)", fx.file, skipped, len(lines)) } } } // mnemonic returns the first whitespace-free token of the text. func mnemonic(text string) string { if i := strings.IndexAny(text, " \t"); i >= 0 { return text[:i] } return text } type fixtureLine struct { code []byte text string } func readFixture(t *testing.T, name string) []fixtureLine { t.Helper() data, err := os.ReadFile(name) if err != nil { t.Fatalf("%s: %v", name, err) } var out []fixtureLine for i, line := range strings.Split(strings.TrimSuffix(string(data), "\n"), "\n") { if line == "" { continue } tab := strings.IndexByte(line, '\t') if tab < 0 { t.Fatalf("%s line %d: no tab separator", name, i+1) } code, err := hex.DecodeString(line[:tab]) if err != nil { t.Fatalf("%s line %d: bad hex: %v", name, i+1, err) } out = append(out, fixtureLine{code: code, text: line[tab+1:]}) } return out } // assemble assembles the parsed file with the encoder for a. func assembleFor(a arch.Arch, f *ast.File) (*asm.Image, error) { switch a { case arch.ARM64: return asm.AssembleFileARM64(f) case arch.RISCV: return asm.AssembleFileRISCV(f) case arch.LOONG64: return asm.AssembleFileLOONG64(f) default: return asm.AssembleFile(f) } }