test(lexer): fuzz the token stream invariants
Assisted-by: GLM 5.3
This commit is contained in:
@@ -0,0 +1,108 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
package lexer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"sourcedock.dev/petrbalvin/gasm-sdk/token"
|
||||||
|
)
|
||||||
|
|
||||||
|
// FuzzTokenize hammers the scanner with arbitrary input. The contract: the
|
||||||
|
// scan never panics, always reaches the true end of the source (the final EOF
|
||||||
|
// sits at len(src)), positions are monotonic and stay inside the source, every
|
||||||
|
// non-EOF token consumes at least one byte, and re-lexing the token texts as
|
||||||
|
// fresh source scans cleanly again. The seed corpus carries the repository's
|
||||||
|
// kernels, so a plain `go test` run replays every seed as a regression case
|
||||||
|
// and CI exercises them without any fuzzing budget.
|
||||||
|
func FuzzTokenize(f *testing.F) {
|
||||||
|
for _, pattern := range []string{
|
||||||
|
"../testdata/*.s",
|
||||||
|
"../testdata/verify/*.s",
|
||||||
|
} {
|
||||||
|
files, _ := filepath.Glob(pattern)
|
||||||
|
for _, path := range files {
|
||||||
|
if b, err := os.ReadFile(path); err == nil {
|
||||||
|
f.Add(string(b))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
f.Add("\xef\xbb\xbfTEXT ·f(SB), NOSPLIT, $0\n") // BOM
|
||||||
|
f.Add("CALL internal∕runtime∕atomic·Xchg(SB)\n") // U+2215 and U+00B7
|
||||||
|
f.Add("\r\n\r\nMOVQ AX, BX\r\r\nRET\r\n") // CR, CRLF, CRLFCRLF
|
||||||
|
f.Add("// comment with stray bytes \xff\xfe\n") // invalid UTF-8
|
||||||
|
f.Add("\"unterminated\n") // string closing on a newline
|
||||||
|
f.Add("'\\n' '\xff' '\\'\n") // rune literals
|
||||||
|
// Numeric edges: the full unsigned 64-bit range, an overflow past it, and
|
||||||
|
// the base prefixes with no digits at all.
|
||||||
|
f.Add("$0xFFFFFFFFFFFFFFFF $0x10000000000000000 0x 0b 0o 1e999 1.5e-\n")
|
||||||
|
f.Add("a \\\n b\n") // line continuation
|
||||||
|
f.Add("\\ // comment \nnext\n") // continuation through a comment
|
||||||
|
f.Add("MOVQ \x00 AX /* block \n */\n") // NUL rune and a multi-line comment
|
||||||
|
f.Add("((((")
|
||||||
|
f.Add(")))))")
|
||||||
|
f.Add("<<->>-") // operator soup
|
||||||
|
f.Add("//\r ") // comment at odd line endings
|
||||||
|
f.Add("\"\\\\\"'\\''") // escapes
|
||||||
|
f.Add("/*/*//**/") // comment-like operator runs
|
||||||
|
f.Add("/ /")
|
||||||
|
|
||||||
|
f.Fuzz(func(t *testing.T, src string) {
|
||||||
|
toks := Tokenize(src)
|
||||||
|
if len(toks) == 0 || toks[len(toks)-1].Kind != token.EOF {
|
||||||
|
t.Fatalf("token stream does not end in EOF: %v", toks)
|
||||||
|
}
|
||||||
|
if got := toks[len(toks)-1].Pos.Offset; got != len(src) {
|
||||||
|
t.Fatalf("EOF offset = %d, want len(src) = %d", got, len(src))
|
||||||
|
}
|
||||||
|
// Every token consumes at least one byte of source, so a stream longer
|
||||||
|
// than the source means the scan is not making progress.
|
||||||
|
if len(toks) > len(src)+1 {
|
||||||
|
t.Fatalf("%d tokens out of %d bytes: the scan cannot be progressing", len(toks), len(src))
|
||||||
|
}
|
||||||
|
var prevEnd token.Position
|
||||||
|
for i, tok := range toks {
|
||||||
|
if tok.Pos.Offset < 0 || tok.End.Offset < tok.Pos.Offset || tok.End.Offset > len(src) {
|
||||||
|
t.Fatalf("token %d %v: byte offsets out of range", i, tok)
|
||||||
|
}
|
||||||
|
if i > 0 && posLess(tok.Pos, prevEnd) {
|
||||||
|
t.Fatalf("token %d %v starts before the previous token ends at %v", i, tok, prevEnd)
|
||||||
|
}
|
||||||
|
if posLess(tok.End, tok.Pos) {
|
||||||
|
t.Fatalf("token %d %v ends before it starts", i, tok)
|
||||||
|
}
|
||||||
|
if tok.Kind != token.EOF && tok.Text == "" {
|
||||||
|
t.Fatalf("token %d %v: non-EOF token with empty text", i, tok)
|
||||||
|
}
|
||||||
|
prevEnd = tok.End
|
||||||
|
}
|
||||||
|
// Round-trip over the token stream: the literal text the scanner
|
||||||
|
// produced must itself scan cleanly, so what the stream says about the
|
||||||
|
// source is never poison for the next consumer.
|
||||||
|
var b strings.Builder
|
||||||
|
for _, tok := range toks {
|
||||||
|
if tok.Kind == token.EOF {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b.WriteString(tok.Text)
|
||||||
|
b.WriteByte('\n')
|
||||||
|
}
|
||||||
|
rt := Tokenize(b.String())
|
||||||
|
if len(rt) == 0 || rt[len(rt)-1].Kind != token.EOF {
|
||||||
|
t.Fatalf("re-scan of the token texts does not end in EOF")
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// posLess orders positions the way the source orders them: by line, then by
|
||||||
|
// column.
|
||||||
|
func posLess(a, b token.Position) bool {
|
||||||
|
if a.Line != b.Line {
|
||||||
|
return a.Line < b.Line
|
||||||
|
}
|
||||||
|
return a.Column < b.Column
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user