fix(lexer): tokenise the flag separator and handle NUL and invalid UTF-8

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-19 23:48:47 +02:00
parent 93c47a312a
commit ac1c05c793
5 changed files with 105 additions and 18 deletions
+44 -18
View File
@@ -30,14 +30,22 @@ type Lexer struct {
// New returns a Lexer over src.
func New(src string) *Lexer {
runes := []rune(src)
off := make([]int, len(runes)+1)
b := 0
for i, r := range runes {
off[i] = b
b += utf8.RuneLen(r)
// Decode over the raw bytes rather than converting with []rune(src): a
// lone invalid byte converts to U+FFFD, whose RuneLen is three, and the
// offset table would then count three bytes where the source has one,
// inflating every later Position.Offset against the original source.
// Decoding advances by the true byte width (one for an invalid byte)
// while the rune stream still carries RuneError, so token text keeps the
// replacement character.
runes := make([]rune, 0, len(src))
off := make([]int, 0, len(src)+1)
for b := 0; b < len(src); {
r, size := utf8.DecodeRuneInString(src[b:])
runes = append(runes, r)
off = append(off, b)
b += size
}
off[len(runes)] = b
off = append(off, len(src))
return &Lexer{src: runes, off: off, line: 1, col: 1}
}
@@ -55,9 +63,16 @@ func Tokenize(src string) []token.Token {
}
}
// cur returns the current rune, or 0 at end of input.
// atEnd reports whether the scanner sits past the last rune. Only the index
// decides: a literal NUL rune in the source is a real character, not the end
// of input, even though cur() returns 0 for both.
func (l *Lexer) atEnd() bool { return l.i >= len(l.src) }
// cur returns the current rune, or 0 at end of input. A real NUL rune in the
// source is indistinguishable here; callers that must tell them apart use
// atEnd.
func (l *Lexer) cur() rune {
if l.i >= len(l.src) {
if l.atEnd() {
return 0
}
return l.src[l.i]
@@ -128,9 +143,15 @@ func (l *Lexer) Next() token.Token {
r := l.cur()
switch {
case r == 0:
case l.atEnd():
return l.make(token.EOF, start, "")
case r == 0:
// A real NUL rune (atEnd is false): fall through to punct, which
// emits it as an Illegal token and advances, so nothing after it
// is silently dropped.
return l.punct(start)
case r == '\n':
l.advance()
return l.make(token.Newline, start, "\n")
@@ -164,14 +185,16 @@ func (l *Lexer) Next() token.Token {
}
}
// lineComment consumes a // comment up to, but not including, the newline.
// lineComment consumes a // comment up to, but not including, the newline. A
// trailing \r is part of a CRLF line ending rather than comment content:
// dropping it keeps the formatter's output uniformly LF-terminated.
func (l *Lexer) lineComment(start token.Position) token.Token {
var b strings.Builder
for l.cur() != 0 && l.cur() != '\n' {
for !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
return l.make(token.Comment, start, b.String())
return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r"))
}
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
@@ -181,7 +204,7 @@ func (l *Lexer) blockComment(start token.Position) token.Token {
l.advance()
b.WriteRune(l.cur()) // '*'
l.advance()
for l.cur() != 0 {
for !l.atEnd() {
if l.cur() == '*' && l.peek(1) == '/' {
b.WriteString("*/")
l.advance()
@@ -199,12 +222,12 @@ func (l *Lexer) string(start token.Position) token.Token {
var b strings.Builder
b.WriteRune('"')
l.advance() // opening quote
for l.cur() != 0 && l.cur() != '\n' {
for !l.atEnd() && l.cur() != '\n' {
r := l.cur()
b.WriteRune(r)
l.advance()
if r == '\\' {
if l.cur() != 0 && l.cur() != '\n' {
if !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
@@ -223,12 +246,12 @@ func (l *Lexer) runeLit(start token.Position) token.Token {
var b strings.Builder
b.WriteRune('\'')
l.advance() // opening quote
for l.cur() != 0 && l.cur() != '\n' {
for !l.atEnd() && l.cur() != '\n' {
r := l.cur()
b.WriteRune(r)
l.advance()
if r == '\\' {
if l.cur() != 0 && l.cur() != '\n' {
if !l.atEnd() && l.cur() != '\n' {
b.WriteRune(l.cur())
l.advance()
}
@@ -373,6 +396,9 @@ func (l *Lexer) punct(start token.Position) token.Token {
case '#':
l.advance()
return l.make(token.Hash, start, "#")
case '|':
l.advance()
return l.make(token.Pipe, start, "|")
default:
// Unknown rune: emit it as Illegal and move on.
l.advance()