Files
gasm-sdk/format/format.go
T
2026-07-11 17:36:52 +02:00

220 lines
5.6 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Package format implements a canonical formatter for GAsm source — the
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
// rather than the AST so that every line (including comments and blanks) is
// preserved; it only normalises indentation, operand spacing and per-function
// mnemonic alignment. Formatting is idempotent.
package format
import (
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
// Source returns the canonical formatting of src.
func Source(path, src string) string {
lines := splitLines(lexer.Tokenize(src))
// First pass: classify each line and record, for every instruction, the
// index of the TEXT function it belongs to, so that mnemonic widths can be
// aligned per function.
type info struct {
kind int
mnemLen int
funcID int
}
const (
kBlank = iota
kComment
kPreproc
kDirective
kLabel
kInstr
)
infos := make([]info, len(lines))
funcID := -1
maxWidth := map[int]int{} // funcID -> widest mnemonic
for i, line := range lines {
inf := info{kind: kBlank, funcID: funcID}
if len(line) > 0 {
switch {
case line[0].Kind == token.Comment:
inf.kind = kComment
case line[0].Kind == token.Hash:
inf.kind = kPreproc
case line[0].Kind == token.Ident && isDirective(line[0].Text):
inf.kind = kDirective
if line[0].Text == "TEXT" {
funcID++
inf.funcID = funcID
} else {
funcID = -1
inf.funcID = -1
}
case len(line) >= 2 && line[1].Kind == token.Colon:
inf.kind = kLabel
default:
inf.kind = kInstr
inf.funcID = funcID
inf.mnemLen = len(line[0].Text)
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
maxWidth[funcID] = inf.mnemLen
}
}
}
infos[i] = inf
}
// Second pass: render.
var b strings.Builder
inBody := false
for i, line := range lines {
inf := infos[i]
var out string
switch inf.kind {
case kBlank:
out = ""
case kComment:
if inBody {
out = "\t" + line[0].Text
} else {
out = line[0].Text
}
case kPreproc:
out = renderPreproc(line)
case kDirective:
out = line[0].Text + " " + renderOps(line[1:])
inBody = line[0].Text == "TEXT"
case kLabel:
out = line[0].Text + ":"
// A label may share its line with an instruction; emit the
// instruction on the following line.
if rest := line[2:]; len(rest) > 0 {
out += "\n" + renderInstr(rest, maxWidth[inf.funcID])
}
case kInstr:
out = renderInstr(line, maxWidth[inf.funcID])
// A RET ends the body for indentation purposes: comments that
// follow it — typically the next function's doc comment — belong
// at column 0, not inside the finished function.
if strings.EqualFold(line[0].Text, "RET") {
inBody = false
}
}
b.WriteString(strings.TrimRight(out, " \t"))
b.WriteByte('\n')
}
return b.String()
}
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
// function's alignment width, then the re-spaced operands.
func renderInstr(line []token.Token, width int) string {
if len(line) == 0 {
return ""
}
mnem := line[0].Text
ops := renderOps(line[1:])
if ops == "" {
return "\t" + mnem
}
if width < len(mnem) {
width = len(mnem)
}
return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops
}
// renderPreproc renders a preprocessor line such as #include "textflag.h".
func renderPreproc(line []token.Token) string {
// "#" directive [args]
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
line[2].Kind == token.String {
return "#include " + line[2].Text
}
parts := make([]string, 0, len(line)-1)
for _, t := range line[1:] {
parts = append(parts, t.Text)
}
return "#" + strings.Join(parts, " ")
}
// renderOps re-spaces a run of operand tokens into canonical form. It never
// invents or drops token text; it only chooses the whitespace between tokens.
func renderOps(toks []token.Token) string {
var b strings.Builder
for i, t := range toks {
if i > 0 && spaceBetween(toks[i-1], t) {
b.WriteByte(' ')
}
b.WriteString(t.Text)
}
return b.String()
}
// spaceBetween decides whether a single space separates prev and cur.
func spaceBetween(prev, cur token.Token) bool {
switch cur.Kind {
case token.RParen:
return false
case token.Comma:
return false
case token.Star, token.Plus, token.Minus, token.Slash:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
case token.LAngle, token.RAngle:
return false
case token.LParen:
// Attach '(' to a preceding name, number, ')' or '>'.
switch prev.Kind {
case token.Ident, token.Number, token.RParen, token.RAngle:
return false
default:
return true
}
}
switch prev.Kind {
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash:
return false
case token.Dollar:
return false
case token.LShift, token.RShift, token.Arrow, token.At:
return false
case token.LAngle, token.RAngle:
return false
case token.Comma:
return true
}
return true
}
func isDirective(s string) bool {
return s == "TEXT" || s == "DATA" || s == "GLOBL"
}
// splitLines groups tokens into lines, dropping Newline and EOF tokens.
func splitLines(toks []token.Token) [][]token.Token {
var lines [][]token.Token
var cur []token.Token
for _, t := range toks {
if t.Kind == token.EOF {
break
}
if t.Kind == token.Newline {
lines = append(lines, cur)
cur = nil
continue
}
cur = append(cur, t)
}
if len(cur) > 0 {
lines = append(lines, cur)
}
return lines
}