2026-07-06 09:49:50 +02:00
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
|
|
2026-09-14 18:22:18 +02:00
|
|
|
// Package format implements a canonical formatter for GAsm source, the
|
2026-07-06 09:49:50 +02:00
|
|
|
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
|
|
|
|
|
// rather than the AST so that every line (including comments and blanks) is
|
|
|
|
|
// preserved; it only normalises indentation, operand spacing and per-function
|
|
|
|
|
// mnemonic alignment. Formatting is idempotent.
|
|
|
|
|
package format
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
"strings"
|
|
|
|
|
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
// Source returns the canonical formatting of src.
|
2026-08-29 17:12:53 +02:00
|
|
|
func Source(src string) string {
|
2026-07-06 09:49:50 +02:00
|
|
|
lines := splitLines(lexer.Tokenize(src))
|
|
|
|
|
|
|
|
|
|
// First pass: classify each line and record, for every instruction, the
|
|
|
|
|
// index of the TEXT function it belongs to, so that mnemonic widths can be
|
|
|
|
|
// aligned per function.
|
|
|
|
|
type info struct {
|
|
|
|
|
kind int
|
|
|
|
|
mnemLen int
|
|
|
|
|
funcID int
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
infos := make([]info, len(lines))
|
|
|
|
|
funcID := -1
|
|
|
|
|
maxWidth := map[int]int{} // funcID -> widest mnemonic
|
|
|
|
|
for i, line := range lines {
|
|
|
|
|
inf := info{kind: kBlank, funcID: funcID}
|
|
|
|
|
if len(line) > 0 {
|
|
|
|
|
switch {
|
|
|
|
|
case line[0].Kind == token.Comment:
|
|
|
|
|
inf.kind = kComment
|
|
|
|
|
case line[0].Kind == token.Hash:
|
|
|
|
|
inf.kind = kPreproc
|
|
|
|
|
case line[0].Kind == token.Ident && isDirective(line[0].Text):
|
|
|
|
|
inf.kind = kDirective
|
|
|
|
|
if line[0].Text == "TEXT" {
|
|
|
|
|
funcID++
|
|
|
|
|
inf.funcID = funcID
|
|
|
|
|
} else {
|
|
|
|
|
funcID = -1
|
|
|
|
|
inf.funcID = -1
|
|
|
|
|
}
|
|
|
|
|
case len(line) >= 2 && line[1].Kind == token.Colon:
|
|
|
|
|
inf.kind = kLabel
|
|
|
|
|
default:
|
|
|
|
|
inf.kind = kInstr
|
|
|
|
|
inf.funcID = funcID
|
|
|
|
|
inf.mnemLen = len(line[0].Text)
|
|
|
|
|
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
|
|
|
|
|
maxWidth[funcID] = inf.mnemLen
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
infos[i] = inf
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-12 21:24:41 +02:00
|
|
|
// Second pass: render each line.
|
|
|
|
|
outs := make([]outLine, 0, len(lines))
|
2026-07-06 09:49:50 +02:00
|
|
|
inBody := false
|
|
|
|
|
for i, line := range lines {
|
|
|
|
|
inf := infos[i]
|
|
|
|
|
var out string
|
|
|
|
|
switch inf.kind {
|
|
|
|
|
case kBlank:
|
|
|
|
|
out = ""
|
|
|
|
|
case kComment:
|
|
|
|
|
if inBody {
|
|
|
|
|
out = "\t" + line[0].Text
|
|
|
|
|
} else {
|
|
|
|
|
out = line[0].Text
|
|
|
|
|
}
|
|
|
|
|
case kPreproc:
|
|
|
|
|
out = renderPreproc(line)
|
|
|
|
|
case kDirective:
|
|
|
|
|
out = line[0].Text + " " + renderOps(line[1:])
|
|
|
|
|
inBody = line[0].Text == "TEXT"
|
|
|
|
|
case kLabel:
|
|
|
|
|
out = line[0].Text + ":"
|
|
|
|
|
// A label may share its line with an instruction; emit the
|
|
|
|
|
// instruction on the following line.
|
|
|
|
|
if rest := line[2:]; len(rest) > 0 {
|
|
|
|
|
out += "\n" + renderInstr(rest, maxWidth[inf.funcID])
|
|
|
|
|
}
|
|
|
|
|
case kInstr:
|
|
|
|
|
out = renderInstr(line, maxWidth[inf.funcID])
|
2026-07-11 17:36:52 +02:00
|
|
|
// A RET ends the body for indentation purposes: comments that
|
2026-09-14 18:22:18 +02:00
|
|
|
// follow it, typically the next function's doc comment, belong
|
2026-07-11 17:36:52 +02:00
|
|
|
// at column 0, not inside the finished function.
|
|
|
|
|
if strings.EqualFold(line[0].Text, "RET") {
|
|
|
|
|
inBody = false
|
|
|
|
|
}
|
2026-07-06 09:49:50 +02:00
|
|
|
}
|
2026-07-12 21:24:41 +02:00
|
|
|
outs = append(outs, outLine{kind: inf.kind, text: strings.TrimRight(out, " \t")})
|
2026-07-06 09:49:50 +02:00
|
|
|
}
|
2026-07-12 21:24:41 +02:00
|
|
|
return normalizeSpacing(outs)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Line classification, shared by the formatting passes.
|
|
|
|
|
const (
|
|
|
|
|
kBlank = iota
|
|
|
|
|
kComment
|
|
|
|
|
kPreproc
|
|
|
|
|
kDirective
|
|
|
|
|
kLabel
|
|
|
|
|
kInstr
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
// outLine is one rendered line together with its classification.
|
|
|
|
|
type outLine struct {
|
|
|
|
|
kind int
|
|
|
|
|
text string
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// normalizeSpacing enforces the canonical blank-line layout: runs of blank
|
2026-09-14 18:22:18 +02:00
|
|
|
// lines collapse to one, and a new block, a label, or a TEXT or GLOBL
|
|
|
|
|
// directive, is preceded by exactly one blank line. Comments immediately
|
2026-07-12 21:24:41 +02:00
|
|
|
// above a block belong to it, so the blank line is inserted before them. No
|
|
|
|
|
// blank line is forced at the top of the file, right after a TEXT (the
|
|
|
|
|
// function's first label), or between stacked labels that share an address.
|
|
|
|
|
func normalizeSpacing(outs []outLine) string {
|
|
|
|
|
blockStart := func(ol outLine) bool {
|
|
|
|
|
switch ol.kind {
|
|
|
|
|
case kLabel:
|
|
|
|
|
return true
|
|
|
|
|
case kDirective:
|
|
|
|
|
// TEXT and GLOBL open a block; DATA continues a GLOBL block.
|
|
|
|
|
return strings.HasPrefix(ol.text, "TEXT") || strings.HasPrefix(ol.text, "GLOBL")
|
|
|
|
|
}
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
insert := make([]bool, len(outs))
|
|
|
|
|
for i, ol := range outs {
|
|
|
|
|
if !blockStart(ol) {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
j := i
|
|
|
|
|
for j > 0 && outs[j-1].kind == kComment {
|
|
|
|
|
j--
|
|
|
|
|
}
|
|
|
|
|
if j == 0 {
|
|
|
|
|
continue // top of file
|
|
|
|
|
}
|
|
|
|
|
switch prev := outs[j-1]; {
|
|
|
|
|
case prev.kind == kBlank, prev.kind == kLabel:
|
|
|
|
|
continue // already separated, or stacked labels
|
|
|
|
|
case prev.kind == kDirective && strings.HasPrefix(prev.text, "TEXT"):
|
|
|
|
|
continue // the function's first label
|
|
|
|
|
}
|
|
|
|
|
insert[j] = true
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var b strings.Builder
|
|
|
|
|
prevBlank := true // also suppresses leading blanks
|
|
|
|
|
for i, ol := range outs {
|
|
|
|
|
if insert[i] && !prevBlank {
|
|
|
|
|
b.WriteByte('\n')
|
|
|
|
|
}
|
|
|
|
|
if ol.kind == kBlank {
|
|
|
|
|
if !prevBlank {
|
|
|
|
|
b.WriteByte('\n')
|
|
|
|
|
}
|
|
|
|
|
prevBlank = true
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
b.WriteString(ol.text)
|
|
|
|
|
b.WriteByte('\n')
|
|
|
|
|
prevBlank = false
|
|
|
|
|
}
|
|
|
|
|
out := strings.TrimRight(b.String(), "\n")
|
|
|
|
|
if out == "" {
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
return out + "\n"
|
2026-07-06 09:49:50 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
|
|
|
|
|
// function's alignment width, then the re-spaced operands.
|
|
|
|
|
func renderInstr(line []token.Token, width int) string {
|
|
|
|
|
if len(line) == 0 {
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
mnem := line[0].Text
|
|
|
|
|
ops := renderOps(line[1:])
|
|
|
|
|
if ops == "" {
|
|
|
|
|
return "\t" + mnem
|
|
|
|
|
}
|
|
|
|
|
if width < len(mnem) {
|
|
|
|
|
width = len(mnem)
|
|
|
|
|
}
|
|
|
|
|
return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// renderPreproc renders a preprocessor line such as #include "textflag.h".
|
|
|
|
|
func renderPreproc(line []token.Token) string {
|
|
|
|
|
// "#" directive [args]
|
|
|
|
|
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
|
|
|
|
|
line[2].Kind == token.String {
|
|
|
|
|
return "#include " + line[2].Text
|
|
|
|
|
}
|
|
|
|
|
parts := make([]string, 0, len(line)-1)
|
|
|
|
|
for _, t := range line[1:] {
|
|
|
|
|
parts = append(parts, t.Text)
|
|
|
|
|
}
|
|
|
|
|
return "#" + strings.Join(parts, " ")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// renderOps re-spaces a run of operand tokens into canonical form. It never
|
|
|
|
|
// invents or drops token text; it only chooses the whitespace between tokens.
|
|
|
|
|
func renderOps(toks []token.Token) string {
|
|
|
|
|
var b strings.Builder
|
|
|
|
|
for i, t := range toks {
|
|
|
|
|
if i > 0 && spaceBetween(toks[i-1], t) {
|
|
|
|
|
b.WriteByte(' ')
|
|
|
|
|
}
|
|
|
|
|
b.WriteString(t.Text)
|
|
|
|
|
}
|
|
|
|
|
return b.String()
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// spaceBetween decides whether a single space separates prev and cur.
|
|
|
|
|
func spaceBetween(prev, cur token.Token) bool {
|
|
|
|
|
switch cur.Kind {
|
|
|
|
|
case token.RParen:
|
|
|
|
|
return false
|
|
|
|
|
case token.Comma:
|
|
|
|
|
return false
|
|
|
|
|
case token.Star, token.Plus, token.Minus, token.Slash:
|
|
|
|
|
return false
|
|
|
|
|
case token.LShift, token.RShift, token.Arrow, token.At:
|
|
|
|
|
return false
|
|
|
|
|
case token.LAngle, token.RAngle:
|
|
|
|
|
return false
|
|
|
|
|
case token.LParen:
|
|
|
|
|
// Attach '(' to a preceding name, number, ')' or '>'.
|
|
|
|
|
switch prev.Kind {
|
|
|
|
|
case token.Ident, token.Number, token.RParen, token.RAngle:
|
|
|
|
|
return false
|
|
|
|
|
default:
|
|
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
switch prev.Kind {
|
|
|
|
|
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash:
|
|
|
|
|
return false
|
|
|
|
|
case token.Dollar:
|
|
|
|
|
return false
|
|
|
|
|
case token.LShift, token.RShift, token.Arrow, token.At:
|
|
|
|
|
return false
|
|
|
|
|
case token.LAngle, token.RAngle:
|
|
|
|
|
return false
|
|
|
|
|
case token.Comma:
|
|
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func isDirective(s string) bool {
|
|
|
|
|
return s == "TEXT" || s == "DATA" || s == "GLOBL"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// splitLines groups tokens into lines, dropping Newline and EOF tokens.
|
|
|
|
|
func splitLines(toks []token.Token) [][]token.Token {
|
|
|
|
|
var lines [][]token.Token
|
|
|
|
|
var cur []token.Token
|
|
|
|
|
for _, t := range toks {
|
|
|
|
|
if t.Kind == token.EOF {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
if t.Kind == token.Newline {
|
|
|
|
|
lines = append(lines, cur)
|
|
|
|
|
cur = nil
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
cur = append(cur, t)
|
|
|
|
|
}
|
|
|
|
|
if len(cur) > 0 {
|
|
|
|
|
lines = append(lines, cur)
|
|
|
|
|
}
|
|
|
|
|
return lines
|
|
|
|
|
}
|