From bc3f4487380bbd39ec1dc4f84a2f06f1ddde2f8d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sat, 19 Sep 2026 20:41:43 +0200 Subject: [PATCH] feat(format): fuzz targets for the parser and formatter Assisted-by: GLM 5.3 Flash --- format/format.go | 105 ++++++++++++++++++++++++++++++++++++++++---- format/fuzz_test.go | 47 ++++++++++++++++++++ parser/fuzz_test.go | 41 +++++++++++++++++ 3 files changed, 184 insertions(+), 9 deletions(-) create mode 100644 format/fuzz_test.go create mode 100644 parser/fuzz_test.go diff --git a/format/format.go b/format/format.go index fbc9ac0..d6f0518 100644 --- a/format/format.go +++ b/format/format.go @@ -50,12 +50,31 @@ func Source(src string) string { } case len(line) >= 2 && line[1].Kind == token.Colon: inf.kind = kLabel + // Peel stacked labels exactly as the render pass does; the + // instruction after the last one is rendered at the + // function's alignment width, so its mnemonic counts here. + rest := line[2:] + for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon && + !isDirective(rest[0].Text) { + rest = rest[2:] + } + if len(rest) > 0 && rest[0].Kind == token.Ident && !isDirective(rest[0].Text) { + inf.mnemLen = len(rest[0].Text) + if funcID >= 0 && inf.mnemLen > maxWidth[funcID] { + maxWidth[funcID] = inf.mnemLen + } + } default: inf.kind = kInstr inf.funcID = funcID - inf.mnemLen = len(line[0].Text) - if funcID >= 0 && inf.mnemLen > maxWidth[funcID] { - maxWidth[funcID] = inf.mnemLen + // Only an identifier mnemonic takes the alignment width; a + // line starting with anything else renders unpadded, so its + // length must not enter the width either. + if line[0].Kind == token.Ident { + inf.mnemLen = len(line[0].Text) + if funcID >= 0 && inf.mnemLen > maxWidth[funcID] { + maxWidth[funcID] = inf.mnemLen + } } } } @@ -83,12 +102,35 @@ func Source(src string) string { out = line[0].Text + " " + renderOps(line[1:]) inBody = line[0].Text == "TEXT" case kLabel: - out = line[0].Text + ":" - // A label may share its line with an instruction; emit the - // instruction on the following line. - if rest := line[2:]; len(rest) > 0 { - out += "\n" + renderInstr(rest, maxWidth[inf.funcID]) + // Every label, and a trailing instruction, becomes its own + // output line: separate outLines keep the blank-line pass + // honest about what it is looking at. + outs = append(outs, outLine{kind: kLabel, text: line[0].Text + ":"}) + rest := line[2:] + for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon && + !isDirective(rest[0].Text) { + outs = append(outs, outLine{kind: kLabel, text: rest[0].Text + ":"}) + rest = rest[2:] } + // A label may share its line with an instruction; the canonical + // form puts the instruction on the following line. Trailing + // content that does not start an instruction (a stray operand + // token) stays on the label line: splitting it off would produce + // a line the parser rejects. + if len(rest) > 0 && rest[0].Kind == token.Ident && isDirective(rest[0].Text) { + // A bare directive cannot start a line of its own (the + // parser wants a symbol per line), so a directive sharing + // the label's line stays there. + outs[len(outs)-1].text += " " + strings.TrimRight(renderOps(rest), " \t") + } else if len(rest) > 0 && rest[0].Kind == token.Ident { + outs = append(outs, outLine{kind: kInstr, text: strings.TrimRight(renderInstr(rest, maxWidth[inf.funcID]), " \t")}) + if strings.EqualFold(rest[0].Text, "RET") { + inBody = false + } + } else if len(rest) > 0 { + outs[len(outs)-1].text += " " + renderOps(rest) + } + continue case kInstr: out = renderInstr(line, maxWidth[inf.funcID]) // A RET ends the body for indentation purposes: comments that @@ -192,6 +234,13 @@ func renderInstr(line []token.Token, width int) string { if ops == "" { return "\t" + mnem } + // Alignment is a mnemonic convention: a line that does not start with + // an identifier (a stray operand token the parser tolerates) renders + // unpadded, so that no alignment width can depend on it and the output + // stays stable across passes. + if line[0].Kind != token.Ident { + return "\t" + mnem + " " + ops + } if width < len(mnem) { width = len(mnem) } @@ -217,7 +266,19 @@ func renderPreproc(line []token.Token) string { func renderOps(toks []token.Token) string { var b strings.Builder for i, t := range toks { - if i > 0 && spaceBetween(toks[i-1], t) { + sp := i > 0 && spaceBetween(toks[i-1], t) + // The accumulated text ending in '/' must never meet a '/' or '*': + // the pair would re-lex as a comment and the next pass would see a + // different line, whatever the token boundaries were. + if !sp && i > 0 && (t.Kind == token.Slash || t.Kind == token.Star) && strings.HasSuffix(b.String(), "/") { + sp = true + } + if sp { + b.WriteByte(' ') + } else if i > 0 && wouldMerge(toks[i-1], t) { + // The tight spelling would re-lex as something else ('/' + // before '*' opens a comment), which would make the next + // formatting pass see a different line. b.WriteByte(' ') } b.WriteString(t.Text) @@ -225,8 +286,27 @@ func renderOps(toks []token.Token) string { return b.String() } +// wouldMerge reports whether writing prev immediately before cur would +// re-lex as something other than those two tokens: a '/' before a '*' opens +// a comment, '>' before '>' shifts, and adjacent operators regroup. +func wouldMerge(prev, cur token.Token) bool { + var kinds []token.Kind + for _, t := range lexer.Tokenize(prev.Text + cur.Text) { + if t.Kind == token.EOF { + break + } + kinds = append(kinds, t.Kind) + } + return len(kinds) != 2 || kinds[0] != prev.Kind || kinds[1] != cur.Kind +} + // spaceBetween decides whether a single space separates prev and cur. func spaceBetween(prev, cur token.Token) bool { + // '/' beside '/' or '*' would form a comment opener in the output and + // make the next pass see a different line; keep them separated. + if prev.Kind == token.Slash && (cur.Kind == token.Slash || cur.Kind == token.Star) { + return true + } switch cur.Kind { case token.RParen: return false @@ -274,6 +354,13 @@ func splitLines(toks []token.Token) [][]token.Token { if t.Kind == token.EOF { break } + if t.Kind == token.Illegal { + // Illegal tokens carry no canonical spelling: the parser + // reports them as errors where they matter, and the formatter + // drops them so that a stray character cannot survive into the + // output and make the next pass render a different file. + continue + } if t.Kind == token.Newline { lines = append(lines, cur) cur = nil diff --git a/format/fuzz_test.go b/format/fuzz_test.go new file mode 100644 index 0000000..ad9b118 --- /dev/null +++ b/format/fuzz_test.go @@ -0,0 +1,47 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package format + +import ( + "os" + "path/filepath" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// FuzzFormatIdempotency hammers the formatter with arbitrary input. The +// contract: formatting twice equals formatting once, and input that parses +// cleanly still parses cleanly after formatting. The seed corpus carries the +// repository's kernels, so a plain `go test` run replays every seed as a +// regression case and CI exercises them without any fuzzing budget. +func FuzzFormatIdempotency(f *testing.F) { + for _, pattern := range []string{ + "../testdata/*.s", + "../testdata/verify/*.s", + } { + files, _ := filepath.Glob(pattern) + for _, path := range files { + if b, err := os.ReadFile(path); err == nil { + f.Add(string(b)) + } + } + } + f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n") + f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n") + f.Add("garbage ### ???\n") + + f.Fuzz(func(t *testing.T, src string) { + once := Source(src) + twice := Source(once) + if once != twice { + t.Fatalf("formatting is not idempotent:\nfirst: %q\nsecond: %q", once, twice) + } + if _, errs := parser.Parse("in.s", src); len(errs) == 0 { + if _, errs := parser.Parse("out.s", once); len(errs) > 0 { + t.Fatalf("formatted output of clean input does not parse: %v\n%s", errs[0], once) + } + } + }) +} diff --git a/parser/fuzz_test.go b/parser/fuzz_test.go new file mode 100644 index 0000000..a30f44c --- /dev/null +++ b/parser/fuzz_test.go @@ -0,0 +1,41 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package parser + +import ( + "os" + "path/filepath" + "testing" +) + +// FuzzParse hammers the parser with arbitrary input. The contract: no panic, +// and always a usable file, whether or not diagnostics were reported. The +// seed corpus carries the repository's kernels, so a plain `go test` run +// replays every seed as a regression case and CI exercises them without any +// fuzzing budget. +func FuzzParse(f *testing.F) { + for _, pattern := range []string{ + "../testdata/*.s", + "../testdata/verify/*.s", + } { + files, _ := filepath.Glob(pattern) + for _, path := range files { + if b, err := os.ReadFile(path); err == nil { + f.Add(string(b)) + } + } + } + f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tRET\n") + f.Add("garbage ### ??? ::: \xff\xfe\n") + f.Add("#define A(x) x+1\nTEXT ·f(SB), $0\n\tA(2)\n\tRET\n") + f.Add("DATA t<>+0(SB)/8, $1\nGLOBL t<>(SB), RODATA, $8\n") + f.Add("TEXT ·f(SB), $0\n\tJMP (AX)\n\tCALL (BX)\n\tRET\n") + + f.Fuzz(func(t *testing.T, src string) { + file, _ := Parse("fuzz.s", src) + if file == nil { + t.Fatal("Parse returned a nil file") + } + }) +}