feat(format): fuzz targets for the parser and formatter

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-19 20:41:43 +02:00
parent f37f183577
commit bc3f448738
3 changed files with 184 additions and 9 deletions
+96 -9
View File
@@ -50,12 +50,31 @@ func Source(src string) string {
} }
case len(line) >= 2 && line[1].Kind == token.Colon: case len(line) >= 2 && line[1].Kind == token.Colon:
inf.kind = kLabel inf.kind = kLabel
// Peel stacked labels exactly as the render pass does; the
// instruction after the last one is rendered at the
// function's alignment width, so its mnemonic counts here.
rest := line[2:]
for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon &&
!isDirective(rest[0].Text) {
rest = rest[2:]
}
if len(rest) > 0 && rest[0].Kind == token.Ident && !isDirective(rest[0].Text) {
inf.mnemLen = len(rest[0].Text)
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
maxWidth[funcID] = inf.mnemLen
}
}
default: default:
inf.kind = kInstr inf.kind = kInstr
inf.funcID = funcID inf.funcID = funcID
inf.mnemLen = len(line[0].Text) // Only an identifier mnemonic takes the alignment width; a
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] { // line starting with anything else renders unpadded, so its
maxWidth[funcID] = inf.mnemLen // length must not enter the width either.
if line[0].Kind == token.Ident {
inf.mnemLen = len(line[0].Text)
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
maxWidth[funcID] = inf.mnemLen
}
} }
} }
} }
@@ -83,12 +102,35 @@ func Source(src string) string {
out = line[0].Text + " " + renderOps(line[1:]) out = line[0].Text + " " + renderOps(line[1:])
inBody = line[0].Text == "TEXT" inBody = line[0].Text == "TEXT"
case kLabel: case kLabel:
out = line[0].Text + ":" // Every label, and a trailing instruction, becomes its own
// A label may share its line with an instruction; emit the // output line: separate outLines keep the blank-line pass
// instruction on the following line. // honest about what it is looking at.
if rest := line[2:]; len(rest) > 0 { outs = append(outs, outLine{kind: kLabel, text: line[0].Text + ":"})
out += "\n" + renderInstr(rest, maxWidth[inf.funcID]) rest := line[2:]
for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon &&
!isDirective(rest[0].Text) {
outs = append(outs, outLine{kind: kLabel, text: rest[0].Text + ":"})
rest = rest[2:]
} }
// A label may share its line with an instruction; the canonical
// form puts the instruction on the following line. Trailing
// content that does not start an instruction (a stray operand
// token) stays on the label line: splitting it off would produce
// a line the parser rejects.
if len(rest) > 0 && rest[0].Kind == token.Ident && isDirective(rest[0].Text) {
// A bare directive cannot start a line of its own (the
// parser wants a symbol per line), so a directive sharing
// the label's line stays there.
outs[len(outs)-1].text += " " + strings.TrimRight(renderOps(rest), " \t")
} else if len(rest) > 0 && rest[0].Kind == token.Ident {
outs = append(outs, outLine{kind: kInstr, text: strings.TrimRight(renderInstr(rest, maxWidth[inf.funcID]), " \t")})
if strings.EqualFold(rest[0].Text, "RET") {
inBody = false
}
} else if len(rest) > 0 {
outs[len(outs)-1].text += " " + renderOps(rest)
}
continue
case kInstr: case kInstr:
out = renderInstr(line, maxWidth[inf.funcID]) out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that // A RET ends the body for indentation purposes: comments that
@@ -192,6 +234,13 @@ func renderInstr(line []token.Token, width int) string {
if ops == "" { if ops == "" {
return "\t" + mnem return "\t" + mnem
} }
// Alignment is a mnemonic convention: a line that does not start with
// an identifier (a stray operand token the parser tolerates) renders
// unpadded, so that no alignment width can depend on it and the output
// stays stable across passes.
if line[0].Kind != token.Ident {
return "\t" + mnem + " " + ops
}
if width < len(mnem) { if width < len(mnem) {
width = len(mnem) width = len(mnem)
} }
@@ -217,7 +266,19 @@ func renderPreproc(line []token.Token) string {
func renderOps(toks []token.Token) string { func renderOps(toks []token.Token) string {
var b strings.Builder var b strings.Builder
for i, t := range toks { for i, t := range toks {
if i > 0 && spaceBetween(toks[i-1], t) { sp := i > 0 && spaceBetween(toks[i-1], t)
// The accumulated text ending in '/' must never meet a '/' or '*':
// the pair would re-lex as a comment and the next pass would see a
// different line, whatever the token boundaries were.
if !sp && i > 0 && (t.Kind == token.Slash || t.Kind == token.Star) && strings.HasSuffix(b.String(), "/") {
sp = true
}
if sp {
b.WriteByte(' ')
} else if i > 0 && wouldMerge(toks[i-1], t) {
// The tight spelling would re-lex as something else ('/'
// before '*' opens a comment), which would make the next
// formatting pass see a different line.
b.WriteByte(' ') b.WriteByte(' ')
} }
b.WriteString(t.Text) b.WriteString(t.Text)
@@ -225,8 +286,27 @@ func renderOps(toks []token.Token) string {
return b.String() return b.String()
} }
// wouldMerge reports whether writing prev immediately before cur would
// re-lex as something other than those two tokens: a '/' before a '*' opens
// a comment, '>' before '>' shifts, and adjacent operators regroup.
func wouldMerge(prev, cur token.Token) bool {
var kinds []token.Kind
for _, t := range lexer.Tokenize(prev.Text + cur.Text) {
if t.Kind == token.EOF {
break
}
kinds = append(kinds, t.Kind)
}
return len(kinds) != 2 || kinds[0] != prev.Kind || kinds[1] != cur.Kind
}
// spaceBetween decides whether a single space separates prev and cur. // spaceBetween decides whether a single space separates prev and cur.
func spaceBetween(prev, cur token.Token) bool { func spaceBetween(prev, cur token.Token) bool {
// '/' beside '/' or '*' would form a comment opener in the output and
// make the next pass see a different line; keep them separated.
if prev.Kind == token.Slash && (cur.Kind == token.Slash || cur.Kind == token.Star) {
return true
}
switch cur.Kind { switch cur.Kind {
case token.RParen: case token.RParen:
return false return false
@@ -274,6 +354,13 @@ func splitLines(toks []token.Token) [][]token.Token {
if t.Kind == token.EOF { if t.Kind == token.EOF {
break break
} }
if t.Kind == token.Illegal {
// Illegal tokens carry no canonical spelling: the parser
// reports them as errors where they matter, and the formatter
// drops them so that a stray character cannot survive into the
// output and make the next pass render a different file.
continue
}
if t.Kind == token.Newline { if t.Kind == token.Newline {
lines = append(lines, cur) lines = append(lines, cur)
cur = nil cur = nil
+47
View File
@@ -0,0 +1,47 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package format
import (
"os"
"path/filepath"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// FuzzFormatIdempotency hammers the formatter with arbitrary input. The
// contract: formatting twice equals formatting once, and input that parses
// cleanly still parses cleanly after formatting. The seed corpus carries the
// repository's kernels, so a plain `go test` run replays every seed as a
// regression case and CI exercises them without any fuzzing budget.
func FuzzFormatIdempotency(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
f.Add("garbage ### ???\n")
f.Fuzz(func(t *testing.T, src string) {
once := Source(src)
twice := Source(once)
if once != twice {
t.Fatalf("formatting is not idempotent:\nfirst: %q\nsecond: %q", once, twice)
}
if _, errs := parser.Parse("in.s", src); len(errs) == 0 {
if _, errs := parser.Parse("out.s", once); len(errs) > 0 {
t.Fatalf("formatted output of clean input does not parse: %v\n%s", errs[0], once)
}
}
})
}
+41
View File
@@ -0,0 +1,41 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package parser
import (
"os"
"path/filepath"
"testing"
)
// FuzzParse hammers the parser with arbitrary input. The contract: no panic,
// and always a usable file, whether or not diagnostics were reported. The
// seed corpus carries the repository's kernels, so a plain `go test` run
// replays every seed as a regression case and CI exercises them without any
// fuzzing budget.
func FuzzParse(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tRET\n")
f.Add("garbage ### ??? ::: \xff\xfe\n")
f.Add("#define A(x) x+1\nTEXT ·f(SB), $0\n\tA(2)\n\tRET\n")
f.Add("DATA t<>+0(SB)/8, $1\nGLOBL t<>(SB), RODATA, $8\n")
f.Add("TEXT ·f(SB), $0\n\tJMP (AX)\n\tCALL (BX)\n\tRET\n")
f.Fuzz(func(t *testing.T, src string) {
file, _ := Parse("fuzz.s", src)
if file == nil {
t.Fatal("Parse returned a nil file")
}
})
}