Files
gasm-sdk/disasm/parity_test.go
T

357 lines
13 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package disasm
import (
"encoding/hex"
"os"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
"sourcedock.dev/petrbalvin/gasm-sdk/asm"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// The parity fixtures carry a deterministic slice of the Go toolchain's own
// view of the encoder corpus, one instruction per line as
//
// HEXBYTES\tTOOLCHAINTEXT
//
// where HEXBYTES are the instruction's bytes in storage order and
// TOOLCHAINTEXT is what `go tool objdump` prints for them (its decoder is
// the ground truth). The slices were sampled from GOROOT's assembler test
// data with a fixed stride: amd64enc.s (the x87, SSE/MMX, VEX, XSAVE and
// system families), amd64enc_extra.s, amd64.s, arm64enc.s, riscv64.s and
// loong64enc1.s.
var parityFixtures = []struct {
arch arch.Arch
file string
}{
{arch.AMD64, "testdata/parity_amd64.txt"},
{arch.AMD64, "testdata/parity_amd64_extra.txt"},
{arch.AMD64, "testdata/parity_amd64_sys.txt"},
{arch.ARM64, "testdata/parity_arm64.txt"},
{arch.RISCV, "testdata/parity_riscv64.txt"},
{arch.LOONG64, "testdata/parity_loong64.txt"},
}
// TestOracleKnownBytes pins the oracle's own fidelity on a few encodings
// whose bytes the sources name outright, before the bulk fixtures are
// trusted: each row was cross-checked against `go tool objdump` over an
// object the Go toolchain's own assembler produced.
func TestOracleKnownBytes(t *testing.T) {
for _, tt := range []struct {
a arch.Arch
code []byte
text string
}{
// go tool asm: MOVQ AX, CX -> 4889c1.
{arch.AMD64, []byte{0x48, 0x89, 0xc1}, "MOVQ AX, CX"},
// arm64enc.s: "ADCW ZR, R8, R10 // 1a1f010a".
{arch.ARM64, []byte{0x0a, 0x01, 0x1f, 0x1a}, "ADCW ZR, R8, R10"},
// riscv64.s: "ADDI $2047, X5 // 9382f27f".
{arch.RISCV, []byte{0x93, 0x82, 0xf2, 0x7f}, "ADDI $2047, X5, X5"},
// The loong64 RET the runtime emits: 0x4c000020.
{arch.LOONG64, []byte{0x20, 0x00, 0x00, 0x4c}, "RET"},
} {
ins, err := Decode(tt.a, tt.code, 0)
if err != nil {
t.Errorf("%s % x: %v", tt.a, tt.code, err)
continue
}
if ins.Text != tt.text || ins.Len != len(tt.code) {
t.Errorf("%s % x: %q (%d bytes), want %q (%d)",
tt.a, tt.code, ins.Text, ins.Len, tt.text, len(tt.code))
}
}
}
// TestToolchainDecodeParity requires the decoder to reproduce the
// toolchain's text for every fixture instruction, and to agree on the
// instruction's length: both views of the same bytes.
func TestToolchainDecodeParity(t *testing.T) {
for _, fx := range parityFixtures {
lines := readFixture(t, fx.file)
for i, ln := range lines {
// The amd64 branch rendering carries an absolute target, which
// is a property of the decode base and not of the bytes; those
// lines are pinned by TestBranchTargetConvention instead.
if fx.arch == arch.AMD64 && relativeBranch(ln.text) {
continue
}
ins, err := Decode(fx.arch, ln.code, 0)
if err != nil {
t.Errorf("%s line %d: %v", fx.file, i+1, err)
continue
}
if ins.Len != len(ln.code) {
t.Errorf("%s line %d: % x decoded to %d bytes, toolchain says %d",
fx.file, i+1, ln.code, ins.Len, len(ln.code))
continue
}
if ins.Text != ln.text {
t.Errorf("%s line %d: % x decodes to %q, toolchain says %q",
fx.file, i+1, ln.code, ins.Text, ln.text)
}
}
}
}
// TestBranchTargetConvention pins the rendering convention for relative
// branches: the target prints as an absolute address in the address space
// of the addr argument, which is exactly what `go tool objdump` does. Both
// views are faithful to the bytes; they must be decoded at the same base to
// compare.
func TestBranchTargetConvention(t *testing.T) {
ins, err := Decode(arch.AMD64, []byte{0x72, 0x02}, 0x1ded)
if err != nil {
t.Fatal(err)
}
// The toolchain prints "JB 0x1df1" for these bytes at the same base.
if ins.Text != "JB 0x1df1" {
t.Errorf("JB at base 0x1ded: %q, want JB 0x1df1", ins.Text)
}
// The toolchain's listing documents relocations after the text
// ("[3:7]R_PCREL:foo+4"); that is bookkeeping of the listing, not part
// of the disassembler's text, and the fixtures carry it stripped.
}
// relativeBranch reports whether the rendered text carries a PC-relative
// or position-dependent target. Such an operand is an address, not a
// datum: it is faithful to the bytes only at the address the instruction
// sits at, so re-assembling it inside a fresh wrapper cannot be expected
// to reproduce them. The amd64 J- and LOOP/CALL families, plus the
// arm64, riscv64 and loong64 "(PC)" forms, all fall here.
func relativeBranch(text string) bool {
if strings.Contains(text, "(PC)") || strings.Contains(text, "(RPC)") {
return true
}
// Mnemonic is the first whitespace-free token.
m := text
if i := strings.IndexAny(text, " \t"); i >= 0 {
m = text[:i]
}
switch {
case strings.HasPrefix(m, "J"), // amd64 Jcc, JMP; arm64/riscv JMP
m == "CALL", m == "XBEGIN",
m == "LOOP", m == "LOOPE", m == "LOOPNE", m == "LOOPZ", m == "LOOPNZ",
m == "CBNZ", m == "CBZ", m == "TBZ", m == "TBNZ", m == "BL":
return true
}
return false
}
// toolchainRenderNames lists mnemonics the toolchain's own renderer prints
// but the gasm encoder cannot assemble. Two kinds sit here: the spellings
// golang.org/x/arch keeps in its non-Plan 9 form (MOVZX and MOVSX where
// the assembler spells MOVBLZX and MOVBLSX, the CMOV* family, the packed
// shuffles it names MOVDQA and MOVDQU, the CVT conversions), so the listing
// text is not assembler input at all, and the handful of Plan 9 names the
// encoder has no table entry for yet (CMPPD, STOSB, PUSHL). A parse
// failure outside this set is a test failure, so the set shrinks as the
// encoder's vocabulary grows.
var toolchainRenderNames = map[string]bool{
"MOVZX": true, "MOVSX": true, "MOVSXD": true,
"SHLDL": true, "SHLDW": true, "SHLDQ": true,
"SHRDL": true, "SHRDW": true, "SHRDQ": true,
"MOVBE": true, "LSL": true, "LAR": true,
"CMOVA": true, "CMOVAE": true, "CMOVB": true, "CMOVBE": true,
"CMOVE": true, "CMOVG": true, "CMOVGE": true, "CMOVL": true,
"CMOVNE": true, "CMOVNO": true, "CMOVNP": true, "CMOVNS": true,
"CMOVO": true, "CMOVP": true, "CMOVS": true,
"PUNPCKLWD": true, "PUNPCKLDQ": true, "PUNPCKHWD": true, "PUNPCKHDQ": true,
"PSLLD": true, "PSRLD": true, "PSRAD": true,
"PMULUDQ": true, "PMADDWD": true, "PACKSSDW": true,
"MOVDQA": true, "MOVDQU": true, "MASKMOVDQU": true,
"LSS": true, "LGS": true, "LFS": true,
"WRGSBASE": true, "RDFSBASE": true, "STR": true, "SLDT": true,
"RDRAND": true, "POPF": true, "LRET": true,
"FDIV": true, "FADD": true, "FRINTS": true, "FRINTM": true,
"CMPPD": true, "STOSB": true, "PUSHL": true,
"MOVSD_XMM": true, "CMPSD_XMM": true,
"CMPPS": true, "CMPSS": true, "CMPSB": true, "MOVSQ": true, "OUTSW": true,
"SLLIUW": true, "SH1ADD": true, "SH2ADDUW": true,
"BSETI": true, "BEXTI": true, "CLMULR": true,
"CPOPW": true, "ORCB": true, "FABS": true, "FABSS": true, "FMUL": true,
"MRS": true, "LU12IW": true,
"CVTTSS2SIL": true, "CVTTSD2SIL": true,
"CVTTPS2DQ": true, "CVTTPD2DQ": true,
"CVTSS2SIL": true, "CVTSI2SSQ": true, "CVTSI2SSL": true,
"CVTSI2SDQ": true, "CVTSI2SDL": true, "CVTSD2SIL": true,
"CVTPS2DQ": true, "CVTPD2DQ": true, "CVTDQ2PS": true, "CVTDQ2PD": true,
}
// encoderDivergences lists fixture lines whose disassembler text re-encodes
// to a DIFFERENT reading, so neither the byte-exact invariant nor the
// fixed-point one can hold. Every entry is a finding for the asm package's
// encoder, reported and not fixed here; the disassembler text itself is the
// toolchain's. amd64: CMOVLE encodes the CMOVE condition code; MOVQ to a
// memory operand drops the FS segment prefix; MOVQ2DQ takes the F2 prefix
// and lands in MOVDQ2Q; CRC32 with a 16-bit register widens to 32 bits.
// arm64: the CRC32 forms take the wrong Rm; the register-indexed load and
// store forms lose the index operand; BFXIL encodes as BFI with shifted
// immediates. loong64: the SC displacement encodes unscaled.
// riscv64: FSGNJXS encodes as FMIN.S; FCLASSS and FCLASSD encode as MOVF;
// the AUIPC immediate loses its high bits. Keyed by the fixture text.
var encoderDivergences = map[string]bool{
"CMOVLE 0(BX), DX": true,
"MOVQ FS:0, DX": true,
"MOVQ2DQ M2, X11": true,
"CRC32 DL, R11": true,
"XADDL DL, DL": true,
"XCHGL DL, DL": true,
"CMPL AL, $0x7": true,
"CMPXCHGL DL, DL": true,
"ANDL $0x7, AL": true,
"SBBL $0x7, DL": true,
"SBBL DL, R11": true,
"SUBL $0x7, AL": true,
"TESTL R11, DL": true,
"BFXIL $26, R8, $16, R20": true,
"CRC32B R17, R8, R16": true,
"CRC32CB R19, R27, R22": true,
"MOVBU (R27)(R23), R14": true,
"MOVHU (R5)(R25.SXTW), R15": true,
"MOVB (R5)(R15), R16": true,
"MOVD R27, (R5)(R15.UXTW<<3)": true,
"MOVH R11, (R27)(R14.SXTW<<1)": true,
"SC R4, 1024(R5)": true,
"FSGNJXS F1, F0, F2": true,
"FCLASSS F0, X5": true,
"FCLASSD F0, X5": true,
"AUIPC $524287, X10": true,
}
// TestDisassemblyRoundTrip is the cheap invariant over the same slice:
// dis(assemble(x)) must re-encode to x's bytes. The listing text goes back
// through the parser and the encoder inside a fresh function, and the
// result must match the fixture bytes exactly.
//
// Three documented kinds of line cannot carry the byte-exact invariant and
// degrade to the weaker fixed-point check, decode(assemble(text)) == text:
// position-dependent branches (relativeBranch), mnemonics the toolchain
// renderer leaves outside Plan 9 vocabulary (toolchainRenderNames), and
// the encoder divergences reported for asm (encoderDivergences) only for
// the byte comparison. A line of none of these kinds must round-trip byte
// for byte.
func TestDisassemblyRoundTrip(t *testing.T) {
for _, fx := range parityFixtures {
lines := readFixture(t, fx.file)
skipped := 0
for i, ln := range lines {
if relativeBranch(ln.text) {
skipped++
continue
}
m := mnemonic(ln.text)
switch {
case toolchainRenderNames[m]:
// The toolchain renders a spelling the encoder cannot take.
skipped++
continue
case fx.arch == arch.RISCV && strings.HasPrefix(m, "V"):
// The RISC-V vector slice of the corpus is the open encoder
// roadmap item; its mnemonics do not assemble yet. The
// decode side is fully covered by the parity test, which
// never parses.
skipped++
continue
case m == "Unknown":
// The x/arch loong64 renderer names opcodes it has no Go
// spelling for "Unknown ..."; that text is the toolchain's
// own and no assembler input.
skipped++
continue
}
src := "TEXT \u00b7k(SB), NOSPLIT, $0\n\t" + ln.text + "\n\tRET\n"
f, errs := parser.Parse("k.s", src)
if len(errs) > 0 {
t.Errorf("%s line %d (%s): parse: %v", fx.file, i+1, ln.text, errs[0])
continue
}
img, err := assembleFor(fx.arch, f)
if err != nil {
t.Errorf("%s line %d (%s): assemble: %v", fx.file, i+1, ln.text, err)
continue
}
fn := img.Funcs[0]
code := img.Code[fn.Offset : fn.Offset+fn.Size]
if len(code) >= len(ln.code) && string(code[:len(ln.code)]) == string(ln.code) {
continue
}
// Weaker invariant for the documented classes: the text must be
// a fixed point of decode followed by encode, that is, the
// re-encoded bytes must carry the same reading. Byte-equality
// divergences outside the encoder set are real failures.
back, err := Decode(fx.arch, code, 0)
if err != nil || back.Text != ln.text {
if encoderDivergences[ln.text] {
continue
}
got := "undecodable"
if err == nil {
got = back.Text
}
t.Errorf("%s line %d (%s): round trip re-decodes as %q, bytes % x",
fx.file, i+1, ln.text, got, code)
}
}
if skipped > 0 {
t.Logf("%s: %d of %d lines excluded (branches and toolchain-only spellings)", fx.file, skipped, len(lines))
}
}
}
// mnemonic returns the first whitespace-free token of the text.
func mnemonic(text string) string {
if i := strings.IndexAny(text, " \t"); i >= 0 {
return text[:i]
}
return text
}
type fixtureLine struct {
code []byte
text string
}
func readFixture(t *testing.T, name string) []fixtureLine {
t.Helper()
data, err := os.ReadFile(name)
if err != nil {
t.Fatalf("%s: %v", name, err)
}
var out []fixtureLine
for i, line := range strings.Split(strings.TrimSuffix(string(data), "\n"), "\n") {
if line == "" {
continue
}
tab := strings.IndexByte(line, '\t')
if tab < 0 {
t.Fatalf("%s line %d: no tab separator", name, i+1)
}
code, err := hex.DecodeString(line[:tab])
if err != nil {
t.Fatalf("%s line %d: bad hex: %v", name, i+1, err)
}
out = append(out, fixtureLine{code: code, text: line[tab+1:]})
}
return out
}
// assemble assembles the parsed file with the encoder for a.
func assembleFor(a arch.Arch, f *ast.File) (*asm.Image, error) {
switch a {
case arch.ARM64:
return asm.AssembleFileARM64(f)
case arch.RISCV:
return asm.AssembleFileRISCV(f)
case arch.LOONG64:
return asm.AssembleFileLOONG64(f)
default:
return asm.AssembleFile(f)
}
}