From ecb203dcf5b60e56586f5f11e960a1ab01d9563b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 11:40:39 +0200 Subject: [PATCH] fix(lexer): treat trailing CR as line end so comment text is idempotent Assisted-by: GLM 5.3 Flash --- format/fuzz_test.go | 6 ++++++ .../fuzz/FuzzFormatIdempotency/f1f73934513efdb3 | 2 ++ lexer/lexer.go | 10 +++++++--- lexer/lexer_test.go | 13 +++++++++++++ 4 files changed, 28 insertions(+), 3 deletions(-) create mode 100644 format/testdata/fuzz/FuzzFormatIdempotency/f1f73934513efdb3 diff --git a/format/fuzz_test.go b/format/fuzz_test.go index ad9b118..f3834d7 100644 --- a/format/fuzz_test.go +++ b/format/fuzz_test.go @@ -31,6 +31,12 @@ func FuzzFormatIdempotency(f *testing.F) { f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n") f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n") f.Add("garbage ### ???\n") + // Line-ending whitespace at the edge of a comment: a CR followed by more + // trailing whitespace once survived the first pass and disappeared on + // re-lexing, so formatting was not idempotent. + f.Add("//\r ") + f.Add("// loop \r\t\nMOVQ AX, BX\n") + f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n") f.Fuzz(func(t *testing.T, src string) { once := Source(src) diff --git a/format/testdata/fuzz/FuzzFormatIdempotency/f1f73934513efdb3 b/format/testdata/fuzz/FuzzFormatIdempotency/f1f73934513efdb3 new file mode 100644 index 0000000..c586110 --- /dev/null +++ b/format/testdata/fuzz/FuzzFormatIdempotency/f1f73934513efdb3 @@ -0,0 +1,2 @@ +go test fuzz v1 +string("//\r ") diff --git a/lexer/lexer.go b/lexer/lexer.go index e6b5f2a..0fe7e3c 100644 --- a/lexer/lexer.go +++ b/lexer/lexer.go @@ -186,15 +186,19 @@ func (l *Lexer) Next() token.Token { } // lineComment consumes a // comment up to, but not including, the newline. A -// trailing \r is part of a CRLF line ending rather than comment content: -// dropping it keeps the formatter's output uniformly LF-terminated. +// trailing run of \r, spaces and tabs is line-ending whitespace rather than +// comment content, so it never enters the token text. Trimming only a \r +// directly before the token's end would make the text depend on what follows +// the comment (a newline or the end of the input): "//x\r " would carry the +// "\r " while "//x\r\n" would not, and a formatter that terminates the line +// with \n would then re-lex its own output to a shorter comment. func (l *Lexer) lineComment(start token.Position) token.Token { var b strings.Builder for !l.atEnd() && l.cur() != '\n' { b.WriteRune(l.cur()) l.advance() } - return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r")) + return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r")) } // blockComment consumes a /* ... */ comment, tolerating an unterminated one. diff --git a/lexer/lexer_test.go b/lexer/lexer_test.go index 8aa7928..abcffd2 100644 --- a/lexer/lexer_test.go +++ b/lexer/lexer_test.go @@ -85,6 +85,19 @@ func TestLabelAndComment(t *testing.T) { []token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment}) } +func TestLineCommentTrailingWhitespace(t *testing.T) { + // A trailing run of CR, spaces and tabs is line-ending whitespace, not + // comment content. The token text must not depend on what follows the + // comment: before the trim covered only a CR directly before the token's + // end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the + // formatter re-lexed its own output to a shorter comment. + eq(t, texts("// loop\r"), []string{"// loop"}) + eq(t, texts("// loop\r "), []string{"// loop"}) + eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"}) + // A CR inside the comment is content and stays. + eq(t, texts("// loops\rall"), []string{"// loops\rall"}) +} + func TestAVX512Mnemonics(t *testing.T) { eq(t, texts("VFMADD231PD Z14, Z12, Z10"), []string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})