test(lexer): fuzz the token stream invariants

Assisted-by: GLM 5.3
This commit is contained in:
2026-10-02 00:40:20 +02:00
parent ca887d3927
commit 96f2dd65b4
+108
View File
@@ -0,0 +1,108 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lexer
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
// FuzzTokenize hammers the scanner with arbitrary input. The contract: the
// scan never panics, always reaches the true end of the source (the final EOF
// sits at len(src)), positions are monotonic and stay inside the source, every
// non-EOF token consumes at least one byte, and re-lexing the token texts as
// fresh source scans cleanly again. The seed corpus carries the repository's
// kernels, so a plain `go test` run replays every seed as a regression case
// and CI exercises them without any fuzzing budget.
func FuzzTokenize(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
f.Add("\xef\xbb\xbfTEXT ·f(SB), NOSPLIT, $0\n") // BOM
f.Add("CALL internal∕runtime∕atomic·Xchg(SB)\n") // U+2215 and U+00B7
f.Add("\r\n\r\nMOVQ AX, BX\r\r\nRET\r\n") // CR, CRLF, CRLFCRLF
f.Add("// comment with stray bytes \xff\xfe\n") // invalid UTF-8
f.Add("\"unterminated\n") // string closing on a newline
f.Add("'\\n' '\xff' '\\'\n") // rune literals
// Numeric edges: the full unsigned 64-bit range, an overflow past it, and
// the base prefixes with no digits at all.
f.Add("$0xFFFFFFFFFFFFFFFF $0x10000000000000000 0x 0b 0o 1e999 1.5e-\n")
f.Add("a \\\n b\n") // line continuation
f.Add("\\ // comment \nnext\n") // continuation through a comment
f.Add("MOVQ \x00 AX /* block \n */\n") // NUL rune and a multi-line comment
f.Add("((((")
f.Add(")))))")
f.Add("<<->>-") // operator soup
f.Add("//\r ") // comment at odd line endings
f.Add("\"\\\\\"'\\''") // escapes
f.Add("/*/*//**/") // comment-like operator runs
f.Add("/ /")
f.Fuzz(func(t *testing.T, src string) {
toks := Tokenize(src)
if len(toks) == 0 || toks[len(toks)-1].Kind != token.EOF {
t.Fatalf("token stream does not end in EOF: %v", toks)
}
if got := toks[len(toks)-1].Pos.Offset; got != len(src) {
t.Fatalf("EOF offset = %d, want len(src) = %d", got, len(src))
}
// Every token consumes at least one byte of source, so a stream longer
// than the source means the scan is not making progress.
if len(toks) > len(src)+1 {
t.Fatalf("%d tokens out of %d bytes: the scan cannot be progressing", len(toks), len(src))
}
var prevEnd token.Position
for i, tok := range toks {
if tok.Pos.Offset < 0 || tok.End.Offset < tok.Pos.Offset || tok.End.Offset > len(src) {
t.Fatalf("token %d %v: byte offsets out of range", i, tok)
}
if i > 0 && posLess(tok.Pos, prevEnd) {
t.Fatalf("token %d %v starts before the previous token ends at %v", i, tok, prevEnd)
}
if posLess(tok.End, tok.Pos) {
t.Fatalf("token %d %v ends before it starts", i, tok)
}
if tok.Kind != token.EOF && tok.Text == "" {
t.Fatalf("token %d %v: non-EOF token with empty text", i, tok)
}
prevEnd = tok.End
}
// Round-trip over the token stream: the literal text the scanner
// produced must itself scan cleanly, so what the stream says about the
// source is never poison for the next consumer.
var b strings.Builder
for _, tok := range toks {
if tok.Kind == token.EOF {
continue
}
b.WriteString(tok.Text)
b.WriteByte('\n')
}
rt := Tokenize(b.String())
if len(rt) == 0 || rt[len(rt)-1].Kind != token.EOF {
t.Fatalf("re-scan of the token texts does not end in EOF")
}
})
}
// posLess orders positions the way the source orders them: by line, then by
// column.
func posLess(a, b token.Position) bool {
if a.Line != b.Line {
return a.Line < b.Line
}
return a.Column < b.Column
}