Files
gasm-sdk/format/format.go
T

440 lines
14 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package format implements a canonical formatter for GAsm source, the
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
// rather than the AST so that every line (including comments and blanks) is
// preserved; it only normalises indentation, operand spacing and per-function
// mnemonic alignment. Formatting is idempotent.
package format
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-sdk/lexer"
"sourcedock.dev/petrbalvin/gasm-sdk/token"
)
// Source returns the canonical formatting of src.
func Source(src string) string {
lines := splitLines(lexer.Tokenize(src))
// First pass: classify each line and record, for every instruction, the
// index of the TEXT function it belongs to, so that mnemonic widths can be
// aligned per function.
type info struct {
kind int
mnemLen int
funcID int
}
infos := make([]info, len(lines))
funcID := -1
maxWidth := map[int]int{} // funcID -> widest mnemonic
for i, line := range lines {
inf := info{kind: kBlank, funcID: funcID}
if len(line) > 0 {
switch {
case line[0].Kind == token.Comment:
inf.kind = kComment
case line[0].Kind == token.Hash:
inf.kind = kPreproc
case line[0].Kind == token.Ident && isDirective(line[0].Text):
inf.kind = kDirective
if line[0].Text == "TEXT" {
funcID++
inf.funcID = funcID
} else {
funcID = -1
inf.funcID = -1
}
case len(line) >= 2 && line[1].Kind == token.Colon:
inf.kind = kLabel
// Peel stacked labels exactly as the render pass does; the
// instruction after the last one is rendered at the
// function's alignment width, so its mnemonic counts here.
rest := line[2:]
for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon &&
!isDirective(rest[0].Text) {
rest = rest[2:]
}
if len(rest) > 0 && rest[0].Kind == token.Ident && !isDirective(rest[0].Text) {
inf.mnemLen = len(rest[0].Text)
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
maxWidth[funcID] = inf.mnemLen
}
}
default:
inf.kind = kInstr
inf.funcID = funcID
// Only an identifier mnemonic takes the alignment width; a
// line starting with anything else renders unpadded, so its
// length must not enter the width either.
if line[0].Kind == token.Ident {
inf.mnemLen = len(line[0].Text)
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
maxWidth[funcID] = inf.mnemLen
}
}
}
}
infos[i] = inf
}
// Second pass: render each line.
outs := make([]outLine, 0, len(lines))
inBody := false
for i, line := range lines {
inf := infos[i]
var out string
switch inf.kind {
case kBlank:
out = ""
case kComment:
if inBody {
out = "\t" + line[0].Text
} else {
out = line[0].Text
}
case kPreproc:
out = renderPreproc(line)
case kDirective:
out = line[0].Text + " " + renderOps(line[1:])
inBody = line[0].Text == "TEXT"
case kLabel:
// Every label, and a trailing instruction, becomes its own
// output line: separate outLines keep the blank-line pass
// honest about what it is looking at.
outs = append(outs, outLine{kind: kLabel, text: line[0].Text + ":"})
rest := line[2:]
for len(rest) >= 2 && rest[0].Kind == token.Ident && rest[1].Kind == token.Colon &&
!isDirective(rest[0].Text) {
outs = append(outs, outLine{kind: kLabel, text: rest[0].Text + ":"})
rest = rest[2:]
}
// A label may share its line with an instruction; the canonical
// form puts the instruction on the following line. Trailing
// content that does not start an instruction (a stray operand
// token) stays on the label line: splitting it off would produce
// a line the parser rejects.
if len(rest) > 0 && rest[0].Kind == token.Ident && isDirective(rest[0].Text) {
// A bare directive cannot start a line of its own (the
// parser wants a symbol per line), so a directive sharing
// the label's line stays there.
outs[len(outs)-1].text += " " + strings.TrimRight(renderOps(rest), " \t")
} else if len(rest) > 0 && rest[0].Kind == token.Ident {
outs = append(outs, outLine{kind: kInstr, text: strings.TrimRight(renderInstr(rest, maxWidth[inf.funcID]), " \t")})
if strings.EqualFold(rest[0].Text, "RET") {
inBody = false
}
} else if len(rest) > 0 {
outs[len(outs)-1].text += " " + renderOps(rest)
}
continue
case kInstr:
out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that
// follow it, typically the next function's doc comment, belong
// at column 0, not inside the finished function.
if strings.EqualFold(line[0].Text, "RET") {
inBody = false
}
}
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
}
return normalizeSpacing(outs)
}
// Line classification, shared by the formatting passes.
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
// outLine is one rendered line together with its classification.
type outLine struct {
kind int
text string
}
// normalizeSpacing enforces the canonical blank-line layout: runs of blank
// lines collapse to one, and a new block, a label, or a TEXT or GLOBL
// directive, is preceded by exactly one blank line. Comments immediately
// above a block belong to it, so the blank line is inserted before them. No
// blank line is forced at the top of the file, right after a TEXT (the
// function's first label), or between stacked labels that share an address.
func normalizeSpacing(outs []outLine) string {
blockStart := func(ol outLine) bool {
switch ol.kind {
case kLabel:
return true
case kDirective:
// TEXT and GLOBL open a block; DATA continues a GLOBL block.
return strings.HasPrefix(ol.text, "TEXT") || strings.HasPrefix(ol.text, "GLOBL")
}
return false
}
insert := make([]bool, len(outs))
for i, ol := range outs {
if !blockStart(ol) {
continue
}
j := i
for j > 0 && outs[j-1].kind == kComment {
j--
}
if j == 0 {
continue // top of file
}
switch prev := outs[j-1]; {
case prev.kind == kBlank, prev.kind == kLabel:
continue // already separated, or stacked labels
case prev.kind == kDirective && strings.HasPrefix(prev.text, "TEXT"):
continue // the function's first label
}
insert[j] = true
}
var b strings.Builder
prevBlank := true // also suppresses leading blanks
for i, ol := range outs {
if insert[i] && !prevBlank {
b.WriteByte('\n')
}
if ol.kind == kBlank {
if !prevBlank {
b.WriteByte('\n')
}
prevBlank = true
continue
}
b.WriteString(ol.text)
b.WriteByte('\n')
prevBlank = false
}
out := strings.TrimRight(b.String(), "\n")
if out == "" {
return ""
}
return out + "\n"
}
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
// function's alignment width, then the re-spaced operands.
func renderInstr(line []token.Token, width int) string {
if len(line) == 0 {
return ""
}
mnem := line[0].Text
ops := renderOps(line[1:])
if ops == "" {
return "\t" + mnem
}
// Alignment is a mnemonic convention: a line that does not start with
// an identifier (a stray operand token the parser tolerates) renders
// unpadded, so that no alignment width can depend on it and the output
// stays stable across passes.
if line[0].Kind != token.Ident {
return "\t" + mnem + " " + ops
}
// A statement separator belongs to the statement it ends: when the
// operands open with a ';', the alignment padding would land between
// the mnemonic and its own separator (REP ; MOVSQ), so such a line
// renders with a single space whatever the function's width.
if strings.HasPrefix(ops, ";") {
return "\t" + mnem + " " + ops
}
if width < len(mnem) {
width = len(mnem)
}
return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops
}
// renderPreproc renders a preprocessor line such as #include "textflag.h".
func renderPreproc(line []token.Token) string {
// "#" directive [args]
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
line[2].Kind == token.String {
return "#include " + line[2].Text
}
// The body of a directive, a macro definition included, is an ordinary
// token run: rendering it through renderOps applies the same punctuation
// rules as everywhere else, so a macro body keeps its canonical spelling
// ($v, (a, b), the ';' separators between statements) instead of being
// spread with a space between every token.
return "#" + renderOps(line[1:])
}
// renderOps re-spaces a run of operand tokens into canonical form. It never
// invents or drops token text; it only chooses the whitespace between tokens.
func renderOps(toks []token.Token) string {
var b strings.Builder
for i, t := range toks {
sp := i > 0 && spaceBetween(toks[i-1], t)
// The accumulated text ending in '/' must never meet a '/' or '*':
// the pair would re-lex as a comment and the next pass would see a
// different line, whatever the token boundaries were.
if !sp && i > 0 && (t.Kind == token.Slash || t.Kind == token.Star) && strings.HasSuffix(b.String(), "/") {
sp = true
}
if sp {
b.WriteByte(' ')
} else if i > 0 && wouldMerge(toks[i-1], t) {
// The tight spelling would re-lex as something else ('/'
// before '*' opens a comment), which would make the next
// formatting pass see a different line.
b.WriteByte(' ')
}
b.WriteString(t.Text)
}
return b.String()
}
// wouldMerge reports whether writing prev immediately before cur would
// re-lex as something other than those two tokens: a '/' before a '*' opens
// a comment, '>' before '>' shifts, and adjacent operators regroup.
func wouldMerge(prev, cur token.Token) bool {
var kinds []token.Kind
for _, t := range lexer.Tokenize(prev.Text + cur.Text) {
if t.Kind == token.EOF {
break
}
kinds = append(kinds, t.Kind)
}
return len(kinds) != 2 || kinds[0] != prev.Kind || kinds[1] != cur.Kind
}
// isOperandBracket reports whether t is one of the square-bracket tokens the
// lexer emits, as Illegal tokens carrying their spelling, for the arm64 and
// loong64 register lists and element selectors that valid GAsm source
// contains.
func isOperandBracket(t token.Token) bool {
return t.Kind == token.Illegal && (t.Text == "[" || t.Text == "]")
}
// isOpenBracket reports whether t is the '[' of a register list or element
// selector.
func isOpenBracket(t token.Token) bool {
return t.Kind == token.Illegal && t.Text == "["
}
// isCloseBracket reports whether t is the ']' that closes a register list or
// element selector.
func isCloseBracket(t token.Token) bool {
return t.Kind == token.Illegal && t.Text == "]"
}
// spaceBetween decides whether a single space separates prev and cur.
func spaceBetween(prev, cur token.Token) bool {
// '/' beside '/' or '*' would form a comment opener in the output and
// make the next pass see a different line; keep them separated.
if prev.Kind == token.Slash && (cur.Kind == token.Slash || cur.Kind == token.Star) {
return true
}
switch cur.Kind {
case token.Illegal:
// A closing bracket always glues to the text it closes. An opening
// bracket glues to the operand it extends (V31.B[15]) but takes its
// own space after a comma, a mnemonic or an operator, exactly like
// the parenthesis rule below. Any other Illegal spelling is stray.
if isCloseBracket(cur) {
return false
}
if isOpenBracket(cur) {
switch prev.Kind {
case token.Ident, token.Number, token.RParen, token.RAngle:
return false
}
return true
}
return true
case token.RParen:
return false
case token.Comma:
return false
case token.Semicolon:
// A ';' is a statement separator on the assembly path, not an
// operand: dropping it would fuse two statements into a line the
// assembler rejects, so it must survive as punctuation. It glues
// to the statement it ends and the next statement takes one space,
// matching the toolchain's listing style.
return false
case token.Star, token.Plus, token.Minus, token.Slash, token.Pipe:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
case token.LAngle, token.RAngle:
return false
case token.LParen:
// Attach '(' to a preceding name, number, ')' or '>'.
switch prev.Kind {
case token.Ident, token.Number, token.RParen, token.RAngle:
return false
default:
return true
}
}
switch prev.Kind {
case token.Illegal:
// '[' opens a bracket group and glues to what follows; ']' closes
// one, and what comes next takes its own space.
return !isOpenBracket(prev)
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash, token.Pipe:
return false
case token.Dollar:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
case token.LAngle, token.RAngle:
return false
case token.Comma, token.Semicolon:
// The statement after a ';' separator takes its own space, exactly
// like the operand after a comma.
return true
}
return true
}
func isDirective(s string) bool {
return s == "TEXT" || s == "DATA" || s == "GLOBL"
}
// splitLines groups tokens into lines, dropping Newline and EOF tokens.
func splitLines(toks []token.Token) [][]token.Token {
var lines [][]token.Token
var cur []token.Token
for _, t := range toks {
if t.Kind == token.EOF {
break
}
if t.Kind == token.Illegal {
// Illegal tokens carry no canonical spelling: the parser
// reports them as errors where they matter, and the formatter
// drops them so that a stray character cannot survive into the
// output and make the next pass render a different file. The
// square brackets of the arm64 and loong64 vector syntaxes are
// the one exception: the lexer gives them no dedicated kind,
// but a register list [V0.B16, V1.B16] and an element selector
// V0.B[3] are valid, load-bearing source, so their tokens stay
// in the stream and renderOps glues them back where they were.
if !isOperandBracket(t) {
continue
}
}
if t.Kind == token.Newline {
lines = append(lines, cur)
cur = nil
continue
}
cur = append(cur, t)
}
if len(cur) > 0 {
lines = append(lines, cur)
}
return lines
}