fix(lexer): treat trailing CR as line end so comment text is idempotent
Test / test (push) Successful in 2m13s
Test / test (push) Successful in 2m13s
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -31,6 +31,12 @@ func FuzzFormatIdempotency(f *testing.F) {
|
||||
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
|
||||
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
|
||||
f.Add("garbage ### ???\n")
|
||||
// Line-ending whitespace at the edge of a comment: a CR followed by more
|
||||
// trailing whitespace once survived the first pass and disappeared on
|
||||
// re-lexing, so formatting was not idempotent.
|
||||
f.Add("//\r ")
|
||||
f.Add("// loop \r\t\nMOVQ AX, BX\n")
|
||||
f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n")
|
||||
|
||||
f.Fuzz(func(t *testing.T, src string) {
|
||||
once := Source(src)
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
go test fuzz v1
|
||||
string("//\r ")
|
||||
+7
-3
@@ -186,15 +186,19 @@ func (l *Lexer) Next() token.Token {
|
||||
}
|
||||
|
||||
// lineComment consumes a // comment up to, but not including, the newline. A
|
||||
// trailing \r is part of a CRLF line ending rather than comment content:
|
||||
// dropping it keeps the formatter's output uniformly LF-terminated.
|
||||
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
|
||||
// comment content, so it never enters the token text. Trimming only a \r
|
||||
// directly before the token's end would make the text depend on what follows
|
||||
// the comment (a newline or the end of the input): "//x\r " would carry the
|
||||
// "\r " while "//x\r\n" would not, and a formatter that terminates the line
|
||||
// with \n would then re-lex its own output to a shorter comment.
|
||||
func (l *Lexer) lineComment(start token.Position) token.Token {
|
||||
var b strings.Builder
|
||||
for !l.atEnd() && l.cur() != '\n' {
|
||||
b.WriteRune(l.cur())
|
||||
l.advance()
|
||||
}
|
||||
return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r"))
|
||||
return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r"))
|
||||
}
|
||||
|
||||
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
||||
|
||||
@@ -85,6 +85,19 @@ func TestLabelAndComment(t *testing.T) {
|
||||
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
||||
}
|
||||
|
||||
func TestLineCommentTrailingWhitespace(t *testing.T) {
|
||||
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
|
||||
// comment content. The token text must not depend on what follows the
|
||||
// comment: before the trim covered only a CR directly before the token's
|
||||
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
|
||||
// formatter re-lexed its own output to a shorter comment.
|
||||
eq(t, texts("// loop\r"), []string{"// loop"})
|
||||
eq(t, texts("// loop\r "), []string{"// loop"})
|
||||
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
|
||||
// A CR inside the comment is content and stays.
|
||||
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
|
||||
}
|
||||
|
||||
func TestAVX512Mnemonics(t *testing.T) {
|
||||
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
||||
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
||||
|
||||
Reference in New Issue
Block a user