feat: gasm-devkit 0.1.0 — GAsm lexer, parser, linter, formatter, LSP and amd64 assembler
Assisted-by: Qwen 3.8 Max Preview
This commit is contained in:
@@ -0,0 +1,213 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Package format implements a canonical formatter for GAsm source — the
|
||||
// equivalent of gofmt for Plan 9 assembly. It works on the token stream
|
||||
// rather than the AST so that every line (including comments and blanks) is
|
||||
// preserved; it only normalises indentation, operand spacing and per-function
|
||||
// mnemonic alignment. Formatting is idempotent.
|
||||
package format
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
// Source returns the canonical formatting of src.
|
||||
func Source(path, src string) string {
|
||||
lines := splitLines(lexer.Tokenize(src))
|
||||
|
||||
// First pass: classify each line and record, for every instruction, the
|
||||
// index of the TEXT function it belongs to, so that mnemonic widths can be
|
||||
// aligned per function.
|
||||
type info struct {
|
||||
kind int
|
||||
mnemLen int
|
||||
funcID int
|
||||
}
|
||||
const (
|
||||
kBlank = iota
|
||||
kComment
|
||||
kPreproc
|
||||
kDirective
|
||||
kLabel
|
||||
kInstr
|
||||
)
|
||||
|
||||
infos := make([]info, len(lines))
|
||||
funcID := -1
|
||||
maxWidth := map[int]int{} // funcID -> widest mnemonic
|
||||
for i, line := range lines {
|
||||
inf := info{kind: kBlank, funcID: funcID}
|
||||
if len(line) > 0 {
|
||||
switch {
|
||||
case line[0].Kind == token.Comment:
|
||||
inf.kind = kComment
|
||||
case line[0].Kind == token.Hash:
|
||||
inf.kind = kPreproc
|
||||
case line[0].Kind == token.Ident && isDirective(line[0].Text):
|
||||
inf.kind = kDirective
|
||||
if line[0].Text == "TEXT" {
|
||||
funcID++
|
||||
inf.funcID = funcID
|
||||
} else {
|
||||
funcID = -1
|
||||
inf.funcID = -1
|
||||
}
|
||||
case len(line) >= 2 && line[1].Kind == token.Colon:
|
||||
inf.kind = kLabel
|
||||
default:
|
||||
inf.kind = kInstr
|
||||
inf.funcID = funcID
|
||||
inf.mnemLen = len(line[0].Text)
|
||||
if funcID >= 0 && inf.mnemLen > maxWidth[funcID] {
|
||||
maxWidth[funcID] = inf.mnemLen
|
||||
}
|
||||
}
|
||||
}
|
||||
infos[i] = inf
|
||||
}
|
||||
|
||||
// Second pass: render.
|
||||
var b strings.Builder
|
||||
inBody := false
|
||||
for i, line := range lines {
|
||||
inf := infos[i]
|
||||
var out string
|
||||
switch inf.kind {
|
||||
case kBlank:
|
||||
out = ""
|
||||
case kComment:
|
||||
if inBody {
|
||||
out = "\t" + line[0].Text
|
||||
} else {
|
||||
out = line[0].Text
|
||||
}
|
||||
case kPreproc:
|
||||
out = renderPreproc(line)
|
||||
case kDirective:
|
||||
out = line[0].Text + " " + renderOps(line[1:])
|
||||
inBody = line[0].Text == "TEXT"
|
||||
case kLabel:
|
||||
out = line[0].Text + ":"
|
||||
// A label may share its line with an instruction; emit the
|
||||
// instruction on the following line.
|
||||
if rest := line[2:]; len(rest) > 0 {
|
||||
out += "\n" + renderInstr(rest, maxWidth[inf.funcID])
|
||||
}
|
||||
case kInstr:
|
||||
out = renderInstr(line, maxWidth[inf.funcID])
|
||||
}
|
||||
b.WriteString(strings.TrimRight(out, " \t"))
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// renderInstr renders an instruction line: a tab, the mnemonic padded to the
|
||||
// function's alignment width, then the re-spaced operands.
|
||||
func renderInstr(line []token.Token, width int) string {
|
||||
if len(line) == 0 {
|
||||
return ""
|
||||
}
|
||||
mnem := line[0].Text
|
||||
ops := renderOps(line[1:])
|
||||
if ops == "" {
|
||||
return "\t" + mnem
|
||||
}
|
||||
if width < len(mnem) {
|
||||
width = len(mnem)
|
||||
}
|
||||
return "\t" + mnem + strings.Repeat(" ", width-len(mnem)) + " " + ops
|
||||
}
|
||||
|
||||
// renderPreproc renders a preprocessor line such as #include "textflag.h".
|
||||
func renderPreproc(line []token.Token) string {
|
||||
// "#" directive [args]
|
||||
if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" &&
|
||||
line[2].Kind == token.String {
|
||||
return "#include " + line[2].Text
|
||||
}
|
||||
parts := make([]string, 0, len(line)-1)
|
||||
for _, t := range line[1:] {
|
||||
parts = append(parts, t.Text)
|
||||
}
|
||||
return "#" + strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
// renderOps re-spaces a run of operand tokens into canonical form. It never
|
||||
// invents or drops token text; it only chooses the whitespace between tokens.
|
||||
func renderOps(toks []token.Token) string {
|
||||
var b strings.Builder
|
||||
for i, t := range toks {
|
||||
if i > 0 && spaceBetween(toks[i-1], t) {
|
||||
b.WriteByte(' ')
|
||||
}
|
||||
b.WriteString(t.Text)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// spaceBetween decides whether a single space separates prev and cur.
|
||||
func spaceBetween(prev, cur token.Token) bool {
|
||||
switch cur.Kind {
|
||||
case token.RParen:
|
||||
return false
|
||||
case token.Comma:
|
||||
return false
|
||||
case token.Star, token.Plus, token.Minus, token.Slash:
|
||||
return false
|
||||
case token.LShift, token.RShift, token.Arrow, token.At:
|
||||
return false
|
||||
case token.LAngle, token.RAngle:
|
||||
return false
|
||||
case token.LParen:
|
||||
// Attach '(' to a preceding name, number, ')' or '>'.
|
||||
switch prev.Kind {
|
||||
case token.Ident, token.Number, token.RParen, token.RAngle:
|
||||
return false
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
switch prev.Kind {
|
||||
case token.LParen, token.Star, token.Plus, token.Minus, token.Slash:
|
||||
return false
|
||||
case token.Dollar:
|
||||
return false
|
||||
case token.LShift, token.RShift, token.Arrow, token.At:
|
||||
return false
|
||||
case token.LAngle, token.RAngle:
|
||||
return false
|
||||
case token.Comma:
|
||||
return true
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func isDirective(s string) bool {
|
||||
return s == "TEXT" || s == "DATA" || s == "GLOBL"
|
||||
}
|
||||
|
||||
// splitLines groups tokens into lines, dropping Newline and EOF tokens.
|
||||
func splitLines(toks []token.Token) [][]token.Token {
|
||||
var lines [][]token.Token
|
||||
var cur []token.Token
|
||||
for _, t := range toks {
|
||||
if t.Kind == token.EOF {
|
||||
break
|
||||
}
|
||||
if t.Kind == token.Newline {
|
||||
lines = append(lines, cur)
|
||||
cur = nil
|
||||
continue
|
||||
}
|
||||
cur = append(cur, t)
|
||||
}
|
||||
if len(cur) > 0 {
|
||||
lines = append(lines, cur)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
package format
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
||||
)
|
||||
|
||||
func TestGolden(t *testing.T) {
|
||||
in := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"MOVQ swin_base+0(FP), SI\n" +
|
||||
"LEAQ (SI)(BX*4), R9\n" +
|
||||
"ANDQ $-8, R10\n" +
|
||||
"VFMADD231PD Z14, Z12, Z10\n" +
|
||||
"RET\n"
|
||||
|
||||
want := "#include \"textflag.h\"\n" +
|
||||
"\n" +
|
||||
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
||||
"\tMOVQ swin_base+0(FP), SI\n" +
|
||||
"\tLEAQ (SI)(BX*4), R9\n" +
|
||||
"\tANDQ $-8, R10\n" +
|
||||
"\tVFMADD231PD Z14, Z12, Z10\n" +
|
||||
"\tRET\n"
|
||||
|
||||
got := Source("f_amd64.s", in)
|
||||
if got != want {
|
||||
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperandSpacing(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"4(SI)": "4(SI)",
|
||||
"(SI)(BX*4)": "(SI)(BX*4)",
|
||||
"$-8": "$-8",
|
||||
"$0x80020100": "$0x80020100",
|
||||
"swin_base+0(FP)": "swin_base+0(FP)",
|
||||
"mask24<>(SB)": "mask24<>(SB)",
|
||||
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
|
||||
}
|
||||
for in, want := range cases {
|
||||
toks := lexOperands(in)
|
||||
if got := renderOps(toks); got != want {
|
||||
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// lexOperands lexes a single operand string and drops the EOF token.
|
||||
func lexOperands(s string) []token.Token {
|
||||
toks := lexer.Tokenize(s)
|
||||
return toks[:len(toks)-1] // drop trailing EOF
|
||||
}
|
||||
|
||||
func TestIdempotent(t *testing.T) {
|
||||
src, err := os.ReadFile("../testdata/sample_amd64.s")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
once := Source("sample_amd64.s", string(src))
|
||||
twice := Source("sample_amd64.s", once)
|
||||
if once != twice {
|
||||
t.Fatal("formatting is not idempotent on the fixture")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRoundTrip checks that formatting produces source that still parses
|
||||
// cleanly, on the fixture and on the real go-flac kernels when present.
|
||||
func TestRoundTrip(t *testing.T) {
|
||||
files := []string{"../testdata/sample_amd64.s"}
|
||||
real, _ := filepath.Glob("../../go-libraries/go-*/*.s")
|
||||
files = append(files, real...)
|
||||
for _, path := range files {
|
||||
src, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
formatted := Source(path, string(src))
|
||||
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
|
||||
t.Errorf("formatted %s no longer parses: %v", path, errs)
|
||||
}
|
||||
if strings.TrimSpace(formatted) == "" {
|
||||
t.Errorf("formatted %s is empty", path)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user