Files
gasm-sdk/format/fuzz_test.go
T
petrbalvin 59637d35b1 test(format): fuzz the expansion round-trip
A second formatter target holds the assembly path's contract: beyond the
token view, a file the expander reads cleanly must expand to the same
statement sequence after formatting, because the macro language draws
distinctions from layout (the adjacency of a #define name and its '(',
continuation bodies) that a token count cannot see.

Assisted-by: GLM 5.3
2026-10-07 13:54:42 +02:00

312 lines
10 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package format
import (
"fmt"
"os"
"path/filepath"
"reflect"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
"sourcedock.dev/petrbalvin/gasm-sdk/lexer"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
// FuzzFormatIdempotency hammers the formatter with arbitrary input. The
// contract: formatting twice equals formatting once, the output re-lexes to
// the same tokens as the input (so formatting changes layout, never meaning),
// input that parses cleanly still parses cleanly after formatting, and its
// tree, positions aside, is unchanged. The seed corpus carries the
// repository's kernels, so a plain `go test` run replays every seed as a
// regression case and CI exercises them without any fuzzing budget.
func FuzzFormatIdempotency(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
f.Add("garbage ### ???\n")
// Line-ending whitespace at the edge of a comment: a CR followed by more
// trailing whitespace once survived the first pass and disappeared on
// re-lexing, so formatting was not idempotent.
f.Add("//\r ")
f.Add("// loop \r\t\nMOVQ AX, BX\n")
f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n")
// A block comment ahead of code on one line: the comment must not swallow
// the statement that follows it.
f.Add("/* head */ MOVQ AX, BX\n")
f.Add("/* head */ DATA d<>+0(SB)/8, $1\n")
// The exotic operand shapes the corpus taught the parser: register ranges,
// PC-relative jumps with a negative displacement and the U+2215 package
// path.
f.Add("TEXT ·f(SB), $0\n\tV4FMADDPS [Z0-Z3], Z2, Z1\n\tRET\n")
f.Add("TEXT ·f(SB), $0\n\tJMP -3(PC)\n\tRET\n")
f.Add("TEXT ·f(SB), $0\n\tCALL internal∕runtime∕atomic·Xchg(SB)\n\tRET\n")
f.Fuzz(func(t *testing.T, src string) {
once := Source(src)
twice := Source(once)
if once != twice {
t.Fatalf("formatting is not idempotent:\nfirst: %q\nsecond: %q", once, twice)
}
in, out := tokenView(src), tokenView(once)
if !sameView(in, out) {
t.Fatalf("formatting changed the token stream:\ninput: %v\noutput: %v\n%s", viewString(in), viewString(out), once)
}
file, errs := parser.Parse("in.s", src)
if len(errs) > 0 || hasStrayIllegal(src) {
// A file the parser already reports on, or one whose stray
// characters the formatter documents dropping, has no tree to
// preserve; the token view above already pinned its tokens.
return
}
reFile, errs := parser.Parse("out.s", once)
if len(errs) > 0 {
t.Fatalf("formatted output of clean input does not parse: %v\n%s", errs[0], once)
}
if got := scrub(reFile); !reflect.DeepEqual(got, scrub(file)) {
t.Fatalf("formatting changed the tree:\ninput: %s\noutput: %s\n%s", dump(scrub(file)), dump(got), once)
}
})
}
// hasStrayIllegal reports whether src lexes to an Illegal token other than
// the operand brackets, which the formatter drops by contract.
func hasStrayIllegal(src string) bool {
for _, tok := range lexer.Tokenize(src) {
if tok.Kind == token.Illegal && tok.Text != "[" && tok.Text != "]" {
return true
}
}
return false
}
// tokenView is the meaning of a source file as the assembler reads it: the
// token kinds and texts in order, ignoring line structure. Stray Illegal
// runes are dropped because the formatter documents dropping them; comment
// texts are compared trailing-whitespace-trimmed because line comments lose
// exactly that on the way through the lexer, and an unterminated block
// comment runs to the end of the file, so the formatter's mandatory final
// newline (and any line ending it lands after) is layout, not content the
// formatter destroyed.
type viewToken struct {
kind token.Kind
text string
}
func tokenView(src string) []viewToken {
var out []viewToken
for _, tok := range lexer.Tokenize(src) {
switch tok.Kind {
case token.EOF, token.Newline:
continue
case token.Illegal:
if tok.Text != "[" && tok.Text != "]" {
continue
}
case token.Comment:
text := strings.TrimRight(tok.Text, " \t\r")
if strings.HasPrefix(text, "/*") && !strings.HasSuffix(text, "*/") {
text = strings.TrimRight(text, " \t\r\n")
}
out = append(out, viewToken{tok.Kind, text})
continue
}
out = append(out, viewToken{tok.Kind, tok.Text})
}
return out
}
func sameView(a, b []viewToken) bool {
if len(a) != len(b) {
return false
}
for i := range a {
if a[i] != b[i] {
return false
}
}
return true
}
func viewString(v []viewToken) string {
var b strings.Builder
for i, t := range v {
if i > 0 {
b.WriteByte(' ')
}
b.WriteString(t.kind.String() + "(" + t.text + ")")
}
return b.String()
}
// scrub reduces a parsed file to what formatting must preserve: every field
// except source positions, the file path, and a preprocessor line's Raw text.
// Positions move with layout, the path is an input of the call, and Raw is a
// single-space join of the directive's tokens, whose input spelling the
// canonical operand spacing may legitimately re-glue ("#define A (x)" formats
// to "#define A(x)"). The token view already pins those tokens.
func scrub(f *ast.File) *ast.File {
v := reflect.ValueOf(f).Elem()
scrubValue(v)
v.FieldByName("Path").SetString("")
return f
}
// scrubValue walks v and zeroes every token.Position it reaches, so two trees
// taken from the same text at different layouts compare equal.
func scrubValue(v reflect.Value) {
switch v.Kind() {
case reflect.Pointer, reflect.Interface:
if !v.IsNil() {
scrubValue(v.Elem())
}
case reflect.Struct:
switch v.Type() {
case reflect.TypeFor[token.Position]():
v.Set(reflect.Zero(v.Type()))
return
case reflect.TypeFor[ast.Preproc]():
v.FieldByName("Raw").SetString("")
}
for _, field := range v.Fields() {
scrubValue(field)
}
case reflect.Slice, reflect.Array:
for i := 0; i < v.Len(); i++ {
scrubValue(v.Index(i))
}
case reflect.Map:
for _, k := range v.MapKeys() {
scrubValue(v.MapIndex(k))
}
}
}
// dump renders a scrubbed tree as text, following pointers, because the fmt
// verbs stop at the first address inside a slice of interfaces.
func dump(v any) string {
var b strings.Builder
dumpValue(reflect.ValueOf(v), &b)
return b.String()
}
// FuzzFormatExpansionRoundTrip hammers the formatter against the assembly
// path's reading of the file: macro expansion and include splicing. The
// token view pins every token the formatter carries over, and beyond that a
// file the expander reads cleanly must expand to exactly the same statement
// sequence after formatting, because distinctions the macro language draws
// from layout (the adjacency of a #define name and its '(', the backslash
// continuations that split a body into statements) live in spacing a pure
// token count cannot see. The seed corpus carries the repository's kernels,
// so a plain `go test` run replays every seed as a regression case.
func FuzzFormatExpansionRoundTrip(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
f.Add("#define A(x) x+1\nTEXT ·f(SB), $0\n\tA(2)\n\tRET\n")
f.Add("#define A (x) x+1\nA\n")
f.Add("#define M (BX)\nTEXT ·f(SB), $0\n\tMOVL M, AX\n\tRET\n")
f.Add("#define M() \\\n\tADD $1, R0 \\\n\tSUB $2, R1\nTEXT ·f(SB), $0\n\tM()\n\tRET\n")
f.Add("#define PAIR(v) \\\n\tADD $v, R0; \\\n\tSUB $v, R1\nTEXT ·f(SB), $0\n\tPAIR(3)\n\tRET\n")
f.Add("#define A A\nA\n") // self-reference
f.Add("#define A(x) ((x))\nA(A(A(1)))\n") // nested invocation
f.Add("#define A 1\n#define B A\nB\n") // chained object macros
f.Add("#ifdef X\n#else\n#endif\n") // conditionals
f.Add("#define A(x) x x\nA(A(A(1)))\n") // amplifying body
f.Add("#include \"textflag.h\"\n") // the never-spliced header
f.Add("#include \"no/such/header.h\"\n") // unresolvable include
f.Add("#define A(x)\nA()\n") // empty body
f.Add("TEXT ·f(SB), $0\n\tM\n\tRET\n") // undefined name, no table entry
f.Fuzz(func(t *testing.T, src string) {
out := Source(src)
in, o := tokenView(src), tokenView(out)
if !sameView(in, o) {
t.Fatalf("formatting changed the token stream:\ninput: %v\noutput: %v\n%s", viewString(in), viewString(o), out)
}
if hasStrayIllegal(src) {
// The formatter documents dropping stray runes; their files have
// no expansion to preserve, and the token view above is a weaker
// contract that already held.
return
}
before, errs := parser.ParseWithOptions("in.s", src, parser.Options{Expand: true})
if len(errs) > 0 {
return
}
after, errs := parser.ParseWithOptions("out.s", out, parser.Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("formatted output of expandable input no longer expands: %v\n%s", errs[0], out)
}
if got, want := stmtSignature(after), stmtSignature(before); !reflect.DeepEqual(got, want) {
t.Fatalf("formatting changed the expansion:\nbefore: %v\nafter: %v\n%s", want, got, out)
}
})
}
func dumpValue(v reflect.Value, b *strings.Builder) {
switch v.Kind() {
case reflect.Pointer, reflect.Interface:
if v.IsNil() {
b.WriteString("nil")
return
}
dumpValue(v.Elem(), b)
case reflect.Struct:
b.WriteString(v.Type().Name() + "{")
for i := 0; i < v.NumField(); i++ {
if i > 0 {
b.WriteString(", ")
}
b.WriteString(v.Type().Field(i).Name + ":")
dumpValue(v.Field(i), b)
}
b.WriteString("}")
case reflect.Slice:
if v.IsNil() {
b.WriteString("nil")
return
}
b.WriteString("[")
for i := 0; i < v.Len(); i++ {
if i > 0 {
b.WriteString(", ")
}
dumpValue(v.Index(i), b)
}
b.WriteString("]")
case reflect.Map:
b.WriteString("map{")
for _, k := range v.MapKeys() {
dumpValue(k, b)
b.WriteString(":")
dumpValue(v.MapIndex(k), b)
}
b.WriteString("}")
default:
b.WriteString(fmt.Sprintf("%v", v))
}
}