fix(lexer): tokenise the flag separator and handle NUL and invalid UTF-8

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-19 23:48:47 +02:00
parent 93c47a312a
commit ac1c05c793
5 changed files with 105 additions and 18 deletions
+51
View File
@@ -153,3 +153,54 @@ func TestOperatorVariants(t *testing.T) {
eq(t, texts("@>"), []string{"@", ">"})
eq(t, texts("a/b"), []string{"a", "/", "b"})
}
// TestPipeFlags covers the '|' that joins TEXT/GLOBL flag lists: it must scan
// as a token of its own so the formatter can preserve the bars the Go
// toolchain requires.
func TestPipeFlags(t *testing.T) {
eq(t, texts("TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0"),
[]string{"TEXT", "·f", "(", "SB", ")", ",", "NOSPLIT", "|", "NOFRAME", "|", "DUPOK", ",", "$", "0"})
}
// TestNulIsIllegal pins the difference between the end of input and a real
// NUL rune: the NUL must surface as an Illegal token and scanning must
// continue past it, so nothing after it is silently dropped.
func TestNulIsIllegal(t *testing.T) {
eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"})
}
// TestOffsetsAroundInvalidByte pins Position.Offset against the original
// bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset
// table by exactly one byte, so every later position stays a true byte
// offset. Columns count runes, so the invalid byte occupies one column like
// any other character.
func TestOffsetsAroundInvalidByte(t *testing.T) {
// bytes: 'A'=0, ' '=1, 0xff=2, ' '=3, 'B'=4, '\n'=5, 'C'=6.
toks := Tokenize("A \xff B\nC")
want := []struct {
text string
off int
}{
{"A", 0}, {"\uFFFD", 2}, {"B", 4}, {"\n", 5}, {"C", 6},
}
if len(toks) != len(want)+1 || toks[len(toks)-1].Kind != token.EOF {
t.Fatalf("tokens = %v, want %v plus EOF", toks, want)
}
for i, w := range want {
if toks[i].Text != w.text || toks[i].Pos.Offset != w.off {
t.Errorf("token %d = %q@%d, want %q@%d", i, toks[i].Text, toks[i].Pos.Offset, w.text, w.off)
}
}
if got := toks[len(toks)-1].Pos.Offset; got != 7 {
t.Errorf("EOF offset = %d, want 7 (source length)", got)
}
if toks[1].Pos.Line != 1 || toks[1].Pos.Column != 3 {
t.Errorf("invalid byte position = %v, want 1:3", toks[1].Pos)
}
if toks[3].Kind != token.Newline || toks[3].Pos.Line != 1 || toks[3].Pos.Column != 6 {
t.Errorf("newline token = %v, want 1:6", toks[3])
}
if toks[4].Pos.Line != 2 || toks[4].Pos.Column != 1 {
t.Errorf("C position = %v, want 2:1", toks[4].Pos)
}
}