// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package format import ( "fmt" "os" "path/filepath" "reflect" "strings" "testing" "sourcedock.dev/petrbalvin/gasm-sdk/ast" "sourcedock.dev/petrbalvin/gasm-sdk/lexer" "sourcedock.dev/petrbalvin/gasm-sdk/parser" "sourcedock.dev/petrbalvin/gasm-sdk/token" ) // FuzzFormatIdempotency hammers the formatter with arbitrary input. The // contract: formatting twice equals formatting once, the output re-lexes to // the same tokens as the input (so formatting changes layout, never meaning), // input that parses cleanly still parses cleanly after formatting, and its // tree, positions aside, is unchanged. The seed corpus carries the // repository's kernels, so a plain `go test` run replays every seed as a // regression case and CI exercises them without any fuzzing budget. func FuzzFormatIdempotency(f *testing.F) { for _, pattern := range []string{ "../testdata/*.s", "../testdata/verify/*.s", } { files, _ := filepath.Glob(pattern) for _, path := range files { if b, err := os.ReadFile(path); err == nil { f.Add(string(b)) } } } f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n") f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n") f.Add("garbage ### ???\n") // Line-ending whitespace at the edge of a comment: a CR followed by more // trailing whitespace once survived the first pass and disappeared on // re-lexing, so formatting was not idempotent. f.Add("//\r ") f.Add("// loop \r\t\nMOVQ AX, BX\n") f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n") // A block comment ahead of code on one line: the comment must not swallow // the statement that follows it. f.Add("/* head */ MOVQ AX, BX\n") f.Add("/* head */ DATA d<>+0(SB)/8, $1\n") // The exotic operand shapes the corpus taught the parser: register ranges, // PC-relative jumps with a negative displacement and the U+2215 package // path. f.Add("TEXT ·f(SB), $0\n\tV4FMADDPS [Z0-Z3], Z2, Z1\n\tRET\n") f.Add("TEXT ·f(SB), $0\n\tJMP -3(PC)\n\tRET\n") f.Add("TEXT ·f(SB), $0\n\tCALL internal∕runtime∕atomic·Xchg(SB)\n\tRET\n") f.Fuzz(func(t *testing.T, src string) { once := Source(src) twice := Source(once) if once != twice { t.Fatalf("formatting is not idempotent:\nfirst: %q\nsecond: %q", once, twice) } in, out := tokenView(src), tokenView(once) if !sameView(in, out) { t.Fatalf("formatting changed the token stream:\ninput: %v\noutput: %v\n%s", viewString(in), viewString(out), once) } file, errs := parser.Parse("in.s", src) if len(errs) > 0 || hasStrayIllegal(src) { // A file the parser already reports on, or one whose stray // characters the formatter documents dropping, has no tree to // preserve; the token view above already pinned its tokens. return } reFile, errs := parser.Parse("out.s", once) if len(errs) > 0 { t.Fatalf("formatted output of clean input does not parse: %v\n%s", errs[0], once) } if got := scrub(reFile); !reflect.DeepEqual(got, scrub(file)) { t.Fatalf("formatting changed the tree:\ninput: %s\noutput: %s\n%s", dump(scrub(file)), dump(got), once) } }) } // hasStrayIllegal reports whether src lexes to an Illegal token other than // the operand brackets, which the formatter drops by contract. func hasStrayIllegal(src string) bool { for _, tok := range lexer.Tokenize(src) { if tok.Kind == token.Illegal && tok.Text != "[" && tok.Text != "]" { return true } } return false } // tokenView is the meaning of a source file as the assembler reads it: the // token kinds and texts in order, ignoring line structure. Stray Illegal // runes are dropped because the formatter documents dropping them; comment // texts are compared trailing-whitespace-trimmed because line comments lose // exactly that on the way through the lexer, and an unterminated block // comment runs to the end of the file, so the formatter's mandatory final // newline (and any line ending it lands after) is layout, not content the // formatter destroyed. type viewToken struct { kind token.Kind text string } func tokenView(src string) []viewToken { var out []viewToken for _, tok := range lexer.Tokenize(src) { switch tok.Kind { case token.EOF, token.Newline: continue case token.Illegal: if tok.Text != "[" && tok.Text != "]" { continue } case token.Comment: text := strings.TrimRight(tok.Text, " \t\r") if strings.HasPrefix(text, "/*") && !strings.HasSuffix(text, "*/") { text = strings.TrimRight(text, " \t\r\n") } out = append(out, viewToken{tok.Kind, text}) continue } out = append(out, viewToken{tok.Kind, tok.Text}) } return out } func sameView(a, b []viewToken) bool { if len(a) != len(b) { return false } for i := range a { if a[i] != b[i] { return false } } return true } func viewString(v []viewToken) string { var b strings.Builder for i, t := range v { if i > 0 { b.WriteByte(' ') } b.WriteString(t.kind.String() + "(" + t.text + ")") } return b.String() } // scrub reduces a parsed file to what formatting must preserve: every field // except source positions, the file path, and a preprocessor line's Raw text. // Positions move with layout, the path is an input of the call, and Raw is a // single-space join of the directive's tokens, whose input spelling the // canonical operand spacing may legitimately re-glue ("#define A (x)" formats // to "#define A(x)"). The token view already pins those tokens. func scrub(f *ast.File) *ast.File { v := reflect.ValueOf(f).Elem() scrubValue(v) v.FieldByName("Path").SetString("") return f } // scrubValue walks v and zeroes every token.Position it reaches, so two trees // taken from the same text at different layouts compare equal. func scrubValue(v reflect.Value) { switch v.Kind() { case reflect.Pointer, reflect.Interface: if !v.IsNil() { scrubValue(v.Elem()) } case reflect.Struct: switch v.Type() { case reflect.TypeFor[token.Position](): v.Set(reflect.Zero(v.Type())) return case reflect.TypeFor[ast.Preproc](): v.FieldByName("Raw").SetString("") } for _, field := range v.Fields() { scrubValue(field) } case reflect.Slice, reflect.Array: for i := 0; i < v.Len(); i++ { scrubValue(v.Index(i)) } case reflect.Map: for _, k := range v.MapKeys() { scrubValue(v.MapIndex(k)) } } } // dump renders a scrubbed tree as text, following pointers, because the fmt // verbs stop at the first address inside a slice of interfaces. func dump(v any) string { var b strings.Builder dumpValue(reflect.ValueOf(v), &b) return b.String() } func dumpValue(v reflect.Value, b *strings.Builder) { switch v.Kind() { case reflect.Pointer, reflect.Interface: if v.IsNil() { b.WriteString("nil") return } dumpValue(v.Elem(), b) case reflect.Struct: b.WriteString(v.Type().Name() + "{") for i := 0; i < v.NumField(); i++ { if i > 0 { b.WriteString(", ") } b.WriteString(v.Type().Field(i).Name + ":") dumpValue(v.Field(i), b) } b.WriteString("}") case reflect.Slice: if v.IsNil() { b.WriteString("nil") return } b.WriteString("[") for i := 0; i < v.Len(); i++ { if i > 0 { b.WriteString(", ") } dumpValue(v.Index(i), b) } b.WriteString("]") case reflect.Map: b.WriteString("map{") for _, k := range v.MapKeys() { dumpValue(k, b) b.WriteString(":") dumpValue(v.MapIndex(k), b) } b.WriteString("}") default: b.WriteString(fmt.Sprintf("%v", v)) } }