diff --git a/lexer/fuzz_test.go b/lexer/fuzz_test.go new file mode 100644 index 0000000..d393511 --- /dev/null +++ b/lexer/fuzz_test.go @@ -0,0 +1,108 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package lexer + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-sdk/token" +) + +// FuzzTokenize hammers the scanner with arbitrary input. The contract: the +// scan never panics, always reaches the true end of the source (the final EOF +// sits at len(src)), positions are monotonic and stay inside the source, every +// non-EOF token consumes at least one byte, and re-lexing the token texts as +// fresh source scans cleanly again. The seed corpus carries the repository's +// kernels, so a plain `go test` run replays every seed as a regression case +// and CI exercises them without any fuzzing budget. +func FuzzTokenize(f *testing.F) { + for _, pattern := range []string{ + "../testdata/*.s", + "../testdata/verify/*.s", + } { + files, _ := filepath.Glob(pattern) + for _, path := range files { + if b, err := os.ReadFile(path); err == nil { + f.Add(string(b)) + } + } + } + f.Add("\xef\xbb\xbfTEXT ·f(SB), NOSPLIT, $0\n") // BOM + f.Add("CALL internal∕runtime∕atomic·Xchg(SB)\n") // U+2215 and U+00B7 + f.Add("\r\n\r\nMOVQ AX, BX\r\r\nRET\r\n") // CR, CRLF, CRLFCRLF + f.Add("// comment with stray bytes \xff\xfe\n") // invalid UTF-8 + f.Add("\"unterminated\n") // string closing on a newline + f.Add("'\\n' '\xff' '\\'\n") // rune literals + // Numeric edges: the full unsigned 64-bit range, an overflow past it, and + // the base prefixes with no digits at all. + f.Add("$0xFFFFFFFFFFFFFFFF $0x10000000000000000 0x 0b 0o 1e999 1.5e-\n") + f.Add("a \\\n b\n") // line continuation + f.Add("\\ // comment \nnext\n") // continuation through a comment + f.Add("MOVQ \x00 AX /* block \n */\n") // NUL rune and a multi-line comment + f.Add("((((") + f.Add(")))))") + f.Add("<<->>-") // operator soup + f.Add("//\r ") // comment at odd line endings + f.Add("\"\\\\\"'\\''") // escapes + f.Add("/*/*//**/") // comment-like operator runs + f.Add("/ /") + + f.Fuzz(func(t *testing.T, src string) { + toks := Tokenize(src) + if len(toks) == 0 || toks[len(toks)-1].Kind != token.EOF { + t.Fatalf("token stream does not end in EOF: %v", toks) + } + if got := toks[len(toks)-1].Pos.Offset; got != len(src) { + t.Fatalf("EOF offset = %d, want len(src) = %d", got, len(src)) + } + // Every token consumes at least one byte of source, so a stream longer + // than the source means the scan is not making progress. + if len(toks) > len(src)+1 { + t.Fatalf("%d tokens out of %d bytes: the scan cannot be progressing", len(toks), len(src)) + } + var prevEnd token.Position + for i, tok := range toks { + if tok.Pos.Offset < 0 || tok.End.Offset < tok.Pos.Offset || tok.End.Offset > len(src) { + t.Fatalf("token %d %v: byte offsets out of range", i, tok) + } + if i > 0 && posLess(tok.Pos, prevEnd) { + t.Fatalf("token %d %v starts before the previous token ends at %v", i, tok, prevEnd) + } + if posLess(tok.End, tok.Pos) { + t.Fatalf("token %d %v ends before it starts", i, tok) + } + if tok.Kind != token.EOF && tok.Text == "" { + t.Fatalf("token %d %v: non-EOF token with empty text", i, tok) + } + prevEnd = tok.End + } + // Round-trip over the token stream: the literal text the scanner + // produced must itself scan cleanly, so what the stream says about the + // source is never poison for the next consumer. + var b strings.Builder + for _, tok := range toks { + if tok.Kind == token.EOF { + continue + } + b.WriteString(tok.Text) + b.WriteByte('\n') + } + rt := Tokenize(b.String()) + if len(rt) == 0 || rt[len(rt)-1].Kind != token.EOF { + t.Fatalf("re-scan of the token texts does not end in EOF") + } + }) +} + +// posLess orders positions the way the source orders them: by line, then by +// column. +func posLess(a, b token.Position) bool { + if a.Line != b.Line { + return a.Line < b.Line + } + return a.Column < b.Column +}