// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package lexer import ( "os" "path/filepath" "strings" "testing" "sourcedock.dev/petrbalvin/gasm-sdk/token" ) // FuzzTokenize hammers the scanner with arbitrary input. The contract: the // scan never panics, always reaches the true end of the source (the final EOF // sits at len(src)), positions are monotonic and stay inside the source, every // non-EOF token consumes at least one byte, and re-lexing the token texts as // fresh source scans cleanly again. The seed corpus carries the repository's // kernels, so a plain `go test` run replays every seed as a regression case // and CI exercises them without any fuzzing budget. func FuzzTokenize(f *testing.F) { for _, pattern := range []string{ "../testdata/*.s", "../testdata/verify/*.s", } { files, _ := filepath.Glob(pattern) for _, path := range files { if b, err := os.ReadFile(path); err == nil { f.Add(string(b)) } } } f.Add("\xef\xbb\xbfTEXT ·f(SB), NOSPLIT, $0\n") // BOM f.Add("CALL internal∕runtime∕atomic·Xchg(SB)\n") // U+2215 and U+00B7 f.Add("\r\n\r\nMOVQ AX, BX\r\r\nRET\r\n") // CR, CRLF, CRLFCRLF f.Add("// comment with stray bytes \xff\xfe\n") // invalid UTF-8 f.Add("\"unterminated\n") // string closing on a newline f.Add("'\\n' '\xff' '\\'\n") // rune literals // Numeric edges: the full unsigned 64-bit range, an overflow past it, and // the base prefixes with no digits at all. f.Add("$0xFFFFFFFFFFFFFFFF $0x10000000000000000 0x 0b 0o 1e999 1.5e-\n") f.Add("a \\\n b\n") // line continuation f.Add("\\ // comment \nnext\n") // continuation through a comment f.Add("MOVQ \x00 AX /* block \n */\n") // NUL rune and a multi-line comment f.Add("((((") f.Add(")))))") f.Add("<<->>-") // operator soup f.Add("//\r ") // comment at odd line endings f.Add("\"\\\\\"'\\''") // escapes f.Add("/*/*//**/") // comment-like operator runs f.Add("/ /") f.Fuzz(func(t *testing.T, src string) { toks := Tokenize(src) if len(toks) == 0 || toks[len(toks)-1].Kind != token.EOF { t.Fatalf("token stream does not end in EOF: %v", toks) } if got := toks[len(toks)-1].Pos.Offset; got != len(src) { t.Fatalf("EOF offset = %d, want len(src) = %d", got, len(src)) } // Every token consumes at least one byte of source, so a stream longer // than the source means the scan is not making progress. if len(toks) > len(src)+1 { t.Fatalf("%d tokens out of %d bytes: the scan cannot be progressing", len(toks), len(src)) } var prevEnd token.Position for i, tok := range toks { if tok.Pos.Offset < 0 || tok.End.Offset < tok.Pos.Offset || tok.End.Offset > len(src) { t.Fatalf("token %d %v: byte offsets out of range", i, tok) } if i > 0 && posLess(tok.Pos, prevEnd) { t.Fatalf("token %d %v starts before the previous token ends at %v", i, tok, prevEnd) } if posLess(tok.End, tok.Pos) { t.Fatalf("token %d %v ends before it starts", i, tok) } if tok.Kind != token.EOF && tok.Text == "" { t.Fatalf("token %d %v: non-EOF token with empty text", i, tok) } prevEnd = tok.End } // Round-trip over the token stream: the literal text the scanner // produced must itself scan cleanly, so what the stream says about the // source is never poison for the next consumer. var b strings.Builder for _, tok := range toks { if tok.Kind == token.EOF { continue } b.WriteString(tok.Text) b.WriteByte('\n') } rt := Tokenize(b.String()) if len(rt) == 0 || rt[len(rt)-1].Kind != token.EOF { t.Fatalf("re-scan of the token texts does not end in EOF") } }) } // posLess orders positions the way the source orders them: by line, then by // column. func posLess(a, b token.Position) bool { if a.Line != b.Line { return a.Line < b.Line } return a.Column < b.Column }