From b54d2b4520e7fff9dba3061035275d7697be1c90 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 13:15:34 +0200 Subject: [PATCH] fix(format): keep macro content behind a label on its line The canonical form splits a label from the instruction that follows it, but an identifier naming a macro may expand into any token at all: moved to a line of its own it no longer parses, because the parser accepts a non-mnemonic first token only behind a label. The names #define'd in the file now hold such content back, conservatively across the whole file, and the crashing input joins the corpus. Assisted-by: GLM 5.3 --- format/expansion_test.go | 18 ++++++++- format/format.go | 40 +++++++++++++------ .../e6b15df9546d86e6 | 2 + 3 files changed, 46 insertions(+), 14 deletions(-) create mode 100644 format/testdata/fuzz/FuzzFormatExpansionRoundTrip/e6b15df9546d86e6 diff --git a/format/expansion_test.go b/format/expansion_test.go index 3a200c1..6c6ffe3 100644 --- a/format/expansion_test.go +++ b/format/expansion_test.go @@ -107,8 +107,22 @@ func TestFormatPreservesExpansionSemantics(t *testing.T) { "\tRET\n", }, { - // A use before the define: the name is not a macro yet at that - // point of the file, so the label splits as any other does. + // Content after a label that names a macro: the expansion may + // start with any token, so the canonical form keeps it on the + // label line, where the parser accepts a non-mnemonic first + // token behind a label but not at the start of a line. + name: "macro behind a label", + src: "#define M 0\n" + + "TEXT ·f(SB), $0\n" + + "\tA: M\n" + + "\tRET\n", + }, + { + // A use above its define: the name is not a macro at that point + // of the file, so splitting would be safe; the whole-file set + // the formatter keeps is deliberately conservative and leaves + // the line whole, which preserves the expansion in both + // directions alike. name: "label before the define of its name", src: "TEXT ·f(SB), $0\n" + "\tM: A00\n" + diff --git a/format/format.go b/format/format.go index 91f9675..4db8579 100644 --- a/format/format.go +++ b/format/format.go @@ -19,6 +19,7 @@ import ( // Source returns the canonical formatting of src. func Source(src string) string { lines := splitLines(lexer.Tokenize(src)) + macros := macroNames(lines) // First pass: classify each line and record, for every instruction, the // index of the TEXT function it belongs to, so that mnemonic widths can be @@ -27,13 +28,12 @@ func Source(src string) string { kind int mnemLen int funcID int - macroName bool // the line's label names a macro defined above it + macroName bool // the line's label names a macro } infos := make([]info, len(lines)) funcID := -1 - maxWidth := map[int]int{} // funcID -> widest mnemonic - macros := map[string]bool{} // names #define'd so far, in file order + maxWidth := map[int]int{} // funcID -> widest mnemonic for i, line := range lines { inf := info{kind: kBlank, funcID: funcID} if len(line) > 0 { @@ -47,10 +47,6 @@ func Source(src string) string { inf.kind = kComment case line[0].Kind == token.Hash: inf.kind = kPreproc - if len(line) >= 3 && line[1].Kind == token.Ident && - line[1].Text == "define" && line[2].Kind == token.Ident { - macros[line[2].Text] = true - } case line[0].Kind == token.Ident && isDirective(line[0].Text): inf.kind = kDirective if line[0].Text == "TEXT" { @@ -72,13 +68,16 @@ func Source(src string) string { } // Peel stacked labels exactly as the render pass does; the // instruction after the last one is rendered at the - // function's alignment width, so its mnemonic counts here. + // function's alignment width, so its mnemonic counts here, + // unless it names a macro and never reaches a line of its + // own. rest := line[2:] for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon && !isDirective(rest[0].Text) { rest = rest[2:] } - if len(rest) > 0 && rest[0].Kind == token.Ident && !isDirective(rest[0].Text) { + if len(rest) > 0 && rest[0].Kind == token.Ident && !isDirective(rest[0].Text) && + !macros[rest[0].Text] { inf.mnemLen = len(rest[0].Text) if funcID >= 0 && inf.mnemLen > maxWidth[funcID] { maxWidth[funcID] = inf.mnemLen @@ -155,14 +154,15 @@ func Source(src string) string { // A label may share its line with an instruction; the canonical // form puts the instruction on the following line. Trailing // content that does not start an instruction (a stray operand - // token) stays on the label line: splitting it off would produce - // a line the parser rejects. + // token, or an identifier naming a macro whose expansion may + // start with any token at all) stays on the label line: + // splitting it off would produce a line the parser rejects. if len(rest) > 0 && rest[0].Kind == token.Ident && isDirective(rest[0].Text) { // A bare directive cannot start a line of its own (the // parser wants a symbol per line), so a directive sharing // the label's line stays there. outs[len(outs)-1].text += " " + trimLineRight(renderOps(rest), rest[len(rest)-1]) - } else if len(rest) > 0 && rest[0].Kind == token.Ident { + } else if len(rest) > 0 && rest[0].Kind == token.Ident && !macros[rest[0].Text] { outs = append(outs, outLine{kind: kInstr, text: trimLineRight(renderInstr(rest, maxWidth[inf.funcID]), rest[len(rest)-1])}) if strings.EqualFold(rest[0].Text, "RET") { inBody = false @@ -189,6 +189,22 @@ func Source(src string) string { return normalizeSpacing(outs) } +// macroNames collects the names #define'd anywhere in the file. A label or +// instruction token that names a macro may expand into anything at all, so +// the lines carrying it are never restructured; the whole-file set is +// deliberately conservative, because a use above its define splits in +// neither direction. +func macroNames(lines [][]token.Token) map[string]bool { + names := map[string]bool{} + for _, line := range lines { + if len(line) >= 3 && line[0].Kind == token.Hash && line[1].Kind == token.Ident && + line[1].Text == "define" && line[2].Kind == token.Ident { + names[line[2].Text] = true + } + } + return names +} + // trimLineRight removes trailing spaces and tabs from a rendered line, which // are layout the canonical form drops. The trim never eats into the text of // the line's final token: a string or rune literal may legitimately end in diff --git a/format/testdata/fuzz/FuzzFormatExpansionRoundTrip/e6b15df9546d86e6 b/format/testdata/fuzz/FuzzFormatExpansionRoundTrip/e6b15df9546d86e6 new file mode 100644 index 0000000..b4f5bbe --- /dev/null +++ b/format/testdata/fuzz/FuzzFormatExpansionRoundTrip/e6b15df9546d86e6 @@ -0,0 +1,2 @@ +go test fuzz v1 +string("#define M 0\nA:M")