fix(lexer): treat trailing CR as line end so comment text is idempotent
Test / test (push) Successful in 2m13s
Test / test (push) Successful in 2m13s
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -31,6 +31,12 @@ func FuzzFormatIdempotency(f *testing.F) {
|
|||||||
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
|
f.Add("TEXT ·f(SB), NOSPLIT, $0\n\tMOVQ AX, BX\n\tRET\n")
|
||||||
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
|
f.Add("TEXT ·f(SB),NOSPLIT,$0\n\tMOVQ AX,BX\n\n\n\tRET\n")
|
||||||
f.Add("garbage ### ???\n")
|
f.Add("garbage ### ???\n")
|
||||||
|
// Line-ending whitespace at the edge of a comment: a CR followed by more
|
||||||
|
// trailing whitespace once survived the first pass and disappeared on
|
||||||
|
// re-lexing, so formatting was not idempotent.
|
||||||
|
f.Add("//\r ")
|
||||||
|
f.Add("// loop \r\t\nMOVQ AX, BX\n")
|
||||||
|
f.Add("TEXT ·f(SB), NOSPLIT, $0 // tail\r\n\tMOVQ AX, BX\r\n\tRET\r\n")
|
||||||
|
|
||||||
f.Fuzz(func(t *testing.T, src string) {
|
f.Fuzz(func(t *testing.T, src string) {
|
||||||
once := Source(src)
|
once := Source(src)
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
go test fuzz v1
|
||||||
|
string("//\r ")
|
||||||
+7
-3
@@ -186,15 +186,19 @@ func (l *Lexer) Next() token.Token {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// lineComment consumes a // comment up to, but not including, the newline. A
|
// lineComment consumes a // comment up to, but not including, the newline. A
|
||||||
// trailing \r is part of a CRLF line ending rather than comment content:
|
// trailing run of \r, spaces and tabs is line-ending whitespace rather than
|
||||||
// dropping it keeps the formatter's output uniformly LF-terminated.
|
// comment content, so it never enters the token text. Trimming only a \r
|
||||||
|
// directly before the token's end would make the text depend on what follows
|
||||||
|
// the comment (a newline or the end of the input): "//x\r " would carry the
|
||||||
|
// "\r " while "//x\r\n" would not, and a formatter that terminates the line
|
||||||
|
// with \n would then re-lex its own output to a shorter comment.
|
||||||
func (l *Lexer) lineComment(start token.Position) token.Token {
|
func (l *Lexer) lineComment(start token.Position) token.Token {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
for !l.atEnd() && l.cur() != '\n' {
|
for !l.atEnd() && l.cur() != '\n' {
|
||||||
b.WriteRune(l.cur())
|
b.WriteRune(l.cur())
|
||||||
l.advance()
|
l.advance()
|
||||||
}
|
}
|
||||||
return l.make(token.Comment, start, strings.TrimSuffix(b.String(), "\r"))
|
return l.make(token.Comment, start, strings.TrimRight(b.String(), " \t\r"))
|
||||||
}
|
}
|
||||||
|
|
||||||
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
// blockComment consumes a /* ... */ comment, tolerating an unterminated one.
|
||||||
|
|||||||
@@ -85,6 +85,19 @@ func TestLabelAndComment(t *testing.T) {
|
|||||||
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
[]token.Kind{token.Ident, token.Colon, token.Ident, token.Ident, token.Comment})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestLineCommentTrailingWhitespace(t *testing.T) {
|
||||||
|
// A trailing run of CR, spaces and tabs is line-ending whitespace, not
|
||||||
|
// comment content. The token text must not depend on what follows the
|
||||||
|
// comment: before the trim covered only a CR directly before the token's
|
||||||
|
// end, "// loop\r " kept the CR while "// loop\r\n" dropped it, and the
|
||||||
|
// formatter re-lexed its own output to a shorter comment.
|
||||||
|
eq(t, texts("// loop\r"), []string{"// loop"})
|
||||||
|
eq(t, texts("// loop\r "), []string{"// loop"})
|
||||||
|
eq(t, texts("// loop \r\t\nMOVQ AX, BX"), []string{"// loop", "MOVQ", "AX", ",", "BX"})
|
||||||
|
// A CR inside the comment is content and stays.
|
||||||
|
eq(t, texts("// loops\rall"), []string{"// loops\rall"})
|
||||||
|
}
|
||||||
|
|
||||||
func TestAVX512Mnemonics(t *testing.T) {
|
func TestAVX512Mnemonics(t *testing.T) {
|
||||||
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
eq(t, texts("VFMADD231PD Z14, Z12, Z10"),
|
||||||
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
[]string{"VFMADD231PD", "Z14", ",", "Z12", ",", "Z10"})
|
||||||
|
|||||||
Reference in New Issue
Block a user