fix(format): keep every token of a line in the canonical output
Assisted-by: GLM 5.3
This commit is contained in:
+45
-6
@@ -35,7 +35,12 @@ func Source(src string) string {
|
||||
inf := info{kind: kBlank, funcID: funcID}
|
||||
if len(line) > 0 {
|
||||
switch {
|
||||
case line[0].Kind == token.Comment:
|
||||
// A whole-line comment is layout of its own. A block comment
|
||||
// ahead of code on the same line ("/* head */ MOVQ AX, BX") is
|
||||
// legal assembly and must not swallow the statement after it, so
|
||||
// only a lone comment is classified as one; anything else renders
|
||||
// as an instruction line that carries the comment inline.
|
||||
case line[0].Kind == token.Comment && len(line) == 1:
|
||||
inf.kind = kComment
|
||||
case line[0].Kind == token.Hash:
|
||||
inf.kind = kPreproc
|
||||
@@ -87,6 +92,10 @@ func Source(src string) string {
|
||||
for i, line := range lines {
|
||||
inf := infos[i]
|
||||
var out string
|
||||
var last token.Token
|
||||
if len(line) > 0 {
|
||||
last = line[len(line)-1]
|
||||
}
|
||||
switch inf.kind {
|
||||
case kBlank:
|
||||
out = ""
|
||||
@@ -99,7 +108,10 @@ func Source(src string) string {
|
||||
case kPreproc:
|
||||
out = renderPreproc(line)
|
||||
case kDirective:
|
||||
out = line[0].Text + " " + renderOps(line[1:])
|
||||
out = line[0].Text
|
||||
if ops := renderOps(line[1:]); ops != "" {
|
||||
out += " " + ops
|
||||
}
|
||||
inBody = line[0].Text == "TEXT"
|
||||
case kLabel:
|
||||
// Every label, and a trailing instruction, becomes its own
|
||||
@@ -121,9 +133,9 @@ func Source(src string) string {
|
||||
// A bare directive cannot start a line of its own (the
|
||||
// parser wants a symbol per line), so a directive sharing
|
||||
// the label's line stays there.
|
||||
outs[len(outs)-1].text += " " + strings.TrimRight(renderOps(rest), " \t")
|
||||
outs[len(outs)-1].text += " " + trimLineRight(renderOps(rest), rest[len(rest)-1])
|
||||
} else if len(rest) > 0 && rest[0].Kind == token.Ident {
|
||||
outs = append(outs, outLine{kind: kInstr, text: strings.TrimRight(renderInstr(rest, maxWidth[inf.funcID]), " \t")})
|
||||
outs = append(outs, outLine{kind: kInstr, text: trimLineRight(renderInstr(rest, maxWidth[inf.funcID]), rest[len(rest)-1])})
|
||||
if strings.EqualFold(rest[0].Text, "RET") {
|
||||
inBody = false
|
||||
}
|
||||
@@ -140,11 +152,31 @@ func Source(src string) string {
|
||||
inBody = false
|
||||
}
|
||||
}
|
||||
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
|
||||
if len(line) == 0 {
|
||||
outs = append(outs, outLine{kind: inf.kind, text: ""})
|
||||
continue
|
||||
}
|
||||
outs = append(outs, outLine{kind: inf.kind, text: trimLineRight(out, last)})
|
||||
}
|
||||
return normalizeSpacing(outs)
|
||||
}
|
||||
|
||||
// trimLineRight removes trailing spaces and tabs from a rendered line, which
|
||||
// are layout the canonical form drops. The trim never eats into the text of
|
||||
// the line's final token: a string or rune literal may legitimately end in
|
||||
// whitespace, and that whitespace is the token's content, not spacing between
|
||||
// tokens. A comment is the exception, because the lexer itself trims a
|
||||
// comment token's trailing whitespace, so removing it cannot change what the
|
||||
// next pass reads. s must end with the final token's text.
|
||||
func trimLineRight(s string, last token.Token) string {
|
||||
start := len(s) - len(last.Text)
|
||||
t := strings.TrimRight(s, " \t")
|
||||
if last.Kind == token.Comment || len(t) <= start {
|
||||
return t
|
||||
}
|
||||
return s[:start] + last.Text
|
||||
}
|
||||
|
||||
// Line classification, shared by the formatting passes.
|
||||
const (
|
||||
kBlank = iota
|
||||
@@ -259,7 +291,14 @@ func renderPreproc(line []token.Token) string {
|
||||
// "#" directive [args]
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
|
||||
line[2].Kind == token.String {
|
||||
return "#include " + line[2].Text
|
||||
// Whatever follows the header name is stray, but it is the file's
|
||||
// stray text: it renders after the name rather than vanishing, so
|
||||
// re-lexing the output sees exactly the tokens the input carried.
|
||||
out := "#include " + line[2].Text
|
||||
if rest := renderOps(line[3:]); rest != "" {
|
||||
out += " " + rest
|
||||
}
|
||||
return out
|
||||
}
|
||||
// The body of a directive, a macro definition included, is an ordinary
|
||||
// token run: rendering it through renderOps applies the same punctuation
|
||||
|
||||
Reference in New Issue
Block a user