// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // Package format implements a canonical formatter for GAsm source — the // equivalent of gofmt for Plan 9 assembly. It works on the token stream // rather than the AST so that every line (including comments and blanks) is // preserved; it only normalises indentation, operand spacing and per-function // mnemonic alignment. Formatting is idempotent. package format import ( "strings" "sourcedock.dev/petrbalvin/gasm-devkit/lexer" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) // Source returns the canonical formatting of src. func Source(src string) string { lines := splitLines(lexer.Tokenize(src)) // First pass: classify each line and record, for every instruction, the // index of the TEXT function it belongs to, so that mnemonic widths can be // aligned per function. type info struct { kind int mnemLen int funcID int } infos := make([]info, len(lines)) funcID := -1 maxWidth := map[int]int{} // funcID -> widest mnemonic for i, line := range lines { inf := info{kind: kBlank, funcID: funcID} if len(line) > 0 { switch { case line[0].Kind == token.Comment: inf.kind = kComment case line[0].Kind == token.Hash: inf.kind = kPreproc case line[0].Kind == token.Ident && isDirective(line[0].Text): inf.kind = kDirective if line[0].Text == "TEXT" { funcID++ inf.funcID = funcID } else { funcID = -1 inf.funcID = -1 } case len(line) >= 2 && line[1].Kind == token.Colon: inf.kind = kLabel default: inf.kind = kInstr inf.funcID = funcID inf.mnemLen = len(line[0].Text) if funcID >= 0 && inf.mnemLen > maxWidth[funcID] { maxWidth[funcID] = inf.mnemLen } } } infos[i] = inf } // Second pass: render each line. outs := make([]outLine, 0, len(lines)) inBody := false for i, line := range lines { inf := infos[i] var out string switch inf.kind { case kBlank: out = "" case kComment: if inBody { out = "\t" + line[0].Text } else { out = line[0].Text } case kPreproc: out = renderPreproc(line) case kDirective: out = line[0].Text + " " + renderOps(line[1:]) inBody = line[0].Text == "TEXT" case kLabel: out = line[0].Text + ":" // A label may share its line with an instruction; emit the // instruction on the following line. if rest := line[2:]; len(rest) > 0 { out += "\n" + renderInstr(rest, maxWidth[inf.funcID]) } case kInstr: out = renderInstr(line, maxWidth[inf.funcID]) // A RET ends the body for indentation purposes: comments that // follow it — typically the next function's doc comment — belong // at column 0, not inside the finished function. if strings.EqualFold(line[0].Text, "RET") { inBody = false } } outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")}) } return normalizeSpacing(outs) } // Line classification, shared by the formatting passes. const ( kBlank = iota kComment kPreproc kDirective kLabel kInstr ) // outLine is one rendered line together with its classification. type outLine struct { kind int text string } // normalizeSpacing enforces the canonical blank-line layout: runs of blank // lines collapse to one, and a new block — a label, or a TEXT or GLOBL // directive — is preceded by exactly one blank line. Comments immediately // above a block belong to it, so the blank line is inserted before them. No // blank line is forced at the top of the file, right after a TEXT (the // function's first label), or between stacked labels that share an address. func normalizeSpacing(outs []outLine) string { blockStart := func(ol outLine) bool { switch ol.kind { case kLabel: return true case kDirective: // TEXT and GLOBL open a block; DATA continues a GLOBL block. return strings.HasPrefix(ol.text, "TEXT") || strings.HasPrefix(ol.text, "GLOBL") } return false } insert := make([]bool, len(outs)) for i, ol := range outs { if !blockStart(ol) { continue } j := i for j > 0 && outs[j-1].kind == kComment { j-- } if j == 0 { continue // top of file } switch prev := outs[j-1]; { case prev.kind == kBlank, prev.kind == kLabel: continue // already separated, or stacked labels case prev.kind == kDirective && strings.HasPrefix(prev.text, "TEXT"): continue // the function's first label } insert[j] = true } var b strings.Builder prevBlank := true // also suppresses leading blanks for i, ol := range outs { if insert[i] && !prevBlank { b.WriteByte('\n') } if ol.kind == kBlank { if !prevBlank { b.WriteByte('\n') } prevBlank = true continue } b.WriteString(ol.text) b.WriteByte('\n') prevBlank = false } out := strings.TrimRight(b.String(), "\n") if out == "" { return "" } return out + "\n" } // renderInstr renders an instruction line: a tab, the mnemonic padded to the // function's alignment width, then the re-spaced operands. func renderInstr(line []token.Token, width int) string { if len(line) == 0 { return "" } mnem := line[0].Text ops := renderOps(line[1:]) if ops == "" { return "\t" + mnem } if width < len(mnem) { width = len(mnem) } return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops } // renderPreproc renders a preprocessor line such as #include "textflag.h". func renderPreproc(line []token.Token) string { // "#" directive [args] if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" && line[2].Kind == token.String { return "#include " + line[2].Text } parts := make([]string, 0, len(line)-1) for _, t := range line[1:] { parts = append(parts, t.Text) } return "#" + strings.Join(parts, " ") } // renderOps re-spaces a run of operand tokens into canonical form. It never // invents or drops token text; it only chooses the whitespace between tokens. func renderOps(toks []token.Token) string { var b strings.Builder for i, t := range toks { if i > 0 && spaceBetween(toks[i-1], t) { b.WriteByte(' ') } b.WriteString(t.Text) } return b.String() } // spaceBetween decides whether a single space separates prev and cur. func spaceBetween(prev, cur token.Token) bool { switch cur.Kind { case token.RParen: return false case token.Comma: return false case token.Star, token.Plus, token.Minus, token.Slash: return false case token.LShift, token.RShift, token.Arrow, token.At: return false case token.LAngle, token.RAngle: return false case token.LParen: // Attach '(' to a preceding name, number, ')' or '>'. switch prev.Kind { case token.Ident, token.Number, token.RParen, token.RAngle: return false default: return true } } switch prev.Kind { case token.LParen, token.Star, token.Plus, token.Minus, token.Slash: return false case token.Dollar: return false case token.LShift, token.RShift, token.Arrow, token.At: return false case token.LAngle, token.RAngle: return false case token.Comma: return true } return true } func isDirective(s string) bool { return s == "TEXT" || s == "DATA" || s == "GLOBL" } // splitLines groups tokens into lines, dropping Newline and EOF tokens. func splitLines(toks []token.Token) [][]token.Token { var lines [][]token.Token var cur []token.Token for _, t := range toks { if t.Kind == token.EOF { break } if t.Kind == token.Newline { lines = append(lines, cur) cur = nil continue } cur = append(cur, t) } if len(cur) > 0 { lines = append(lines, cur) } return lines }