fix(format): keep every token of a line in the canonical output

Assisted-by: GLM 5.3
This commit is contained in:
2026-10-02 00:40:20 +02:00
parent 4d01bb3ecf
commit 9a5217d9c1
4 changed files with 251 additions and 12 deletions
+45 -6
View File
@@ -35,7 +35,12 @@ func Source(src string) string {
inf := info{kind: kBlank, funcID: funcID}
if len(line) > 0 {
switch {
case line[0].Kind == token.Comment:
// A whole-line comment is layout of its own. A block comment
// ahead of code on the same line ("/* head */ MOVQ AX, BX") is
// legal assembly and must not swallow the statement after it, so
// only a lone comment is classified as one; anything else renders
// as an instruction line that carries the comment inline.
case line[0].Kind == token.Comment && len(line) == 1:
inf.kind = kComment
case line[0].Kind == token.Hash:
inf.kind = kPreproc
@@ -87,6 +92,10 @@ func Source(src string) string {
for i, line := range lines {
inf := infos[i]
var out string
var last token.Token
if len(line) > 0 {
last = line[len(line)-1]
}
switch inf.kind {
case kBlank:
out = ""
@@ -99,7 +108,10 @@ func Source(src string) string {
case kPreproc:
out = renderPreproc(line)
case kDirective:
out = line[0].Text + " " + renderOps(line[1:])
out = line[0].Text
if ops := renderOps(line[1:]); ops != "" {
out += " " + ops
}
inBody = line[0].Text == "TEXT"
case kLabel:
// Every label, and a trailing instruction, becomes its own
@@ -121,9 +133,9 @@ func Source(src string) string {
// A bare directive cannot start a line of its own (the
// parser wants a symbol per line), so a directive sharing
// the label's line stays there.
outs[len(outs)-1].text += " " + strings.TrimRight(renderOps(rest), " \t")
outs[len(outs)-1].text += " " + trimLineRight(renderOps(rest), rest[len(rest)-1])
} else if len(rest) > 0 && rest[0].Kind == token.Ident {
outs = append(outs, outLine{kind: kInstr, text: strings.TrimRight(renderInstr(rest, maxWidth[inf.funcID]), " \t")})
outs = append(outs, outLine{kind: kInstr, text: trimLineRight(renderInstr(rest, maxWidth[inf.funcID]), rest[len(rest)-1])})
if strings.EqualFold(rest[0].Text, "RET") {
inBody = false
}
@@ -140,11 +152,31 @@ func Source(src string) string {
inBody = false
}
}
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
if len(line) == 0 {
outs = append(outs, outLine{kind: inf.kind, text: ""})
continue
}
outs = append(outs, outLine{kind: inf.kind, text: trimLineRight(out, last)})
}
return normalizeSpacing(outs)
}
// trimLineRight removes trailing spaces and tabs from a rendered line, which
// are layout the canonical form drops. The trim never eats into the text of
// the line's final token: a string or rune literal may legitimately end in
// whitespace, and that whitespace is the token's content, not spacing between
// tokens. A comment is the exception, because the lexer itself trims a
// comment token's trailing whitespace, so removing it cannot change what the
// next pass reads. s must end with the final token's text.
func trimLineRight(s string, last token.Token) string {
start := len(s) - len(last.Text)
t := strings.TrimRight(s, " \t")
if last.Kind == token.Comment || len(t) <= start {
return t
}
return s[:start] + last.Text
}
// Line classification, shared by the formatting passes.
const (
kBlank = iota
@@ -259,7 +291,14 @@ func renderPreproc(line []token.Token) string {
// "#" directive [args]
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
line[2].Kind == token.String {
return "#include " + line[2].Text
// Whatever follows the header name is stray, but it is the file's
// stray text: it renders after the name rather than vanishing, so
// re-lexing the output sees exactly the tokens the input carried.
out := "#include " + line[2].Text
if rest := renderOps(line[3:]); rest != "" {
out += " " + rest
}
return out
}
// The body of a directive, a macro definition included, is an ordinary
// token run: rendering it through renderOps applies the same punctuation