Files
gasm-sdk/lexer/fuzz_test.go
T
2026-10-02 00:40:20 +02:00

109 lines
3.9 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package lexer
import (
"os"
"path/filepath"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
// FuzzTokenize hammers the scanner with arbitrary input. The contract: the
// scan never panics, always reaches the true end of the source (the final EOF
// sits at len(src)), positions are monotonic and stay inside the source, every
// non-EOF token consumes at least one byte, and re-lexing the token texts as
// fresh source scans cleanly again. The seed corpus carries the repository's
// kernels, so a plain `go test` run replays every seed as a regression case
// and CI exercises them without any fuzzing budget.
func FuzzTokenize(f *testing.F) {
for _, pattern := range []string{
"../testdata/*.s",
"../testdata/verify/*.s",
} {
files, _ := filepath.Glob(pattern)
for _, path := range files {
if b, err := os.ReadFile(path); err == nil {
f.Add(string(b))
}
}
}
f.Add("\xef\xbb\xbfTEXT ·f(SB), NOSPLIT, $0\n") // BOM
f.Add("CALL internal∕runtime∕atomic·Xchg(SB)\n") // U+2215 and U+00B7
f.Add("\r\n\r\nMOVQ AX, BX\r\r\nRET\r\n") // CR, CRLF, CRLFCRLF
f.Add("// comment with stray bytes \xff\xfe\n") // invalid UTF-8
f.Add("\"unterminated\n") // string closing on a newline
f.Add("'\\n' '\xff' '\\'\n") // rune literals
// Numeric edges: the full unsigned 64-bit range, an overflow past it, and
// the base prefixes with no digits at all.
f.Add("$0xFFFFFFFFFFFFFFFF $0x10000000000000000 0x 0b 0o 1e999 1.5e-\n")
f.Add("a \\\n b\n") // line continuation
f.Add("\\ // comment \nnext\n") // continuation through a comment
f.Add("MOVQ \x00 AX /* block \n */\n") // NUL rune and a multi-line comment
f.Add("((((")
f.Add(")))))")
f.Add("<<->>-") // operator soup
f.Add("//\r ") // comment at odd line endings
f.Add("\"\\\\\"'\\''") // escapes
f.Add("/*/*//**/") // comment-like operator runs
f.Add("/ /")
f.Fuzz(func(t *testing.T, src string) {
toks := Tokenize(src)
if len(toks) == 0 || toks[len(toks)-1].Kind != token.EOF {
t.Fatalf("token stream does not end in EOF: %v", toks)
}
if got := toks[len(toks)-1].Pos.Offset; got != len(src) {
t.Fatalf("EOF offset = %d, want len(src) = %d", got, len(src))
}
// Every token consumes at least one byte of source, so a stream longer
// than the source means the scan is not making progress.
if len(toks) > len(src)+1 {
t.Fatalf("%d tokens out of %d bytes: the scan cannot be progressing", len(toks), len(src))
}
var prevEnd token.Position
for i, tok := range toks {
if tok.Pos.Offset < 0 || tok.End.Offset < tok.Pos.Offset || tok.End.Offset > len(src) {
t.Fatalf("token %d %v: byte offsets out of range", i, tok)
}
if i > 0 && posLess(tok.Pos, prevEnd) {
t.Fatalf("token %d %v starts before the previous token ends at %v", i, tok, prevEnd)
}
if posLess(tok.End, tok.Pos) {
t.Fatalf("token %d %v ends before it starts", i, tok)
}
if tok.Kind != token.EOF && tok.Text == "" {
t.Fatalf("token %d %v: non-EOF token with empty text", i, tok)
}
prevEnd = tok.End
}
// Round-trip over the token stream: the literal text the scanner
// produced must itself scan cleanly, so what the stream says about the
// source is never poison for the next consumer.
var b strings.Builder
for _, tok := range toks {
if tok.Kind == token.EOF {
continue
}
b.WriteString(tok.Text)
b.WriteByte('\n')
}
rt := Tokenize(b.String())
if len(rt) == 0 || rt[len(rt)-1].Kind != token.EOF {
t.Fatalf("re-scan of the token texts does not end in EOF")
}
})
}
// posLess orders positions the way the source orders them: by line, then by
// column.
func posLess(a, b token.Position) bool {
if a.Line != b.Line {
return a.Line < b.Line
}
return a.Column < b.Column
}