Files
gasm-sdk/format/format_test.go
T

518 lines
17 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package format
import (
"os"
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
"sourcedock.dev/petrbalvin/gasm-devkit/token"
)
func TestGolden(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"MOVQ swin_base+0(FP), SI\n" +
"LEAQ (SI)(BX*4), R9\n" +
"ANDQ $-8, R10\n" +
"VFMADD231PD Z14, Z12, Z10\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tMOVQ swin_base+0(FP), SI\n" +
"\tLEAQ (SI)(BX*4), R9\n" +
"\tANDQ $-8, R10\n" +
"\tVFMADD231PD Z14, Z12, Z10\n" +
"\tRET\n"
got := Source(in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
}
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
// sits at column 0 even when another function (ending in RET) precedes it;
// the RET must terminate the previous body for indentation purposes.
func TestDocCommentIndent(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"// func first()\n" +
"TEXT ·first(SB), NOSPLIT, $0\n" +
"XORQ AX, AX\n" +
"RET\n" +
"\n" +
"// func second()\n" +
"TEXT ·second(SB), NOSPLIT, $0\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"// func first()\n" +
"TEXT ·first(SB), NOSPLIT, $0\n" +
"\tXORQ AX, AX\n" +
"\tRET\n" +
"\n" +
"// func second()\n" +
"TEXT ·second(SB), NOSPLIT, $0\n" +
"\tRET\n"
got := Source(in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
// Body comments stay indented.
body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n"
gotBody := Source(body)
if !strings.Contains(gotBody, "\t// inside the body\n") {
t.Fatalf("body comment must stay indented:\n%q", gotBody)
}
}
// TestBlankLines checks the blank-line canonicalisation: exactly one blank
// line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed
// to one, and no blank forced after TEXT, between stacked labels, or at the
// top of the file. Leading comments belong to the block they precede.
func TestBlankLines(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" + // first label: no blank after TEXT
"XORQ AX, AX\n" +
"JMP next\n" + // unlabelled glue: fmt inserts a blank before next:
"next:\n" +
"stacked:\n" + // stacked labels share an address: no blank between
"INCQ AX\n" +
"\n" +
"\n" + // two blanks collapse to one
"// separated block\n" + // comment belongs to the label below
"later:\n" +
"RET\n" +
"// func g()\n" + // doc comment: blank goes before it
"TEXT ·g(SB), NOSPLIT, $0\n" +
"RET\n" +
"GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL…
"DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA
"\n" +
"\n" +
"\n" // trailing blanks dropped
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"first:\n" +
"\tXORQ AX, AX\n" +
"\tJMP next\n" +
"\n" +
"next:\n" +
"stacked:\n" +
"\tINCQ AX\n" +
"\n" +
"\t// separated block\n" + // body comment before a label stays indented
"later:\n" +
"\tRET\n" +
"\n" +
"// func g()\n" +
"TEXT ·g(SB), NOSPLIT, $0\n" +
"\tRET\n" +
"\n" +
"GLOBL ·mask(SB), RODATA, $8\n" +
"DATA ·mask+0(SB)/4, $1\n"
got := Source(in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if again := Source(got); again != got {
t.Fatalf("not idempotent:\n%q", again)
}
}
func TestOperandSpacing(t *testing.T) {
cases := map[string]string{
"4(SI)": "4(SI)",
"(SI)(BX*4)": "(SI)(BX*4)",
"$-8": "$-8",
"$0x80020100": "$0x80020100",
"swin_base+0(FP)": "swin_base+0(FP)",
"mask24<>(SB)": "mask24<>(SB)",
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
"NOSPLIT|DUPOK": "NOSPLIT|DUPOK",
}
for in, want := range cases {
toks := lexOperands(in)
if got := renderOps(toks); got != want {
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
}
}
}
// TestVectorBracketSpacing pins the square-bracket operand forms of the
// arm64 and loong64 vector syntaxes. The lexer emits '[' and ']' as Illegal
// tokens carrying their spelling, and renderOps must glue them back exactly
// where they were: a register list and an element selector are load-bearing
// operands the assembler reads out of the operand text, so no bracket may be
// dropped, and the canonical spelling inside the brackets is tight.
func TestVectorBracketSpacing(t *testing.T) {
cases := map[string]string{
// Register lists of one to four registers.
"[V21.B16]": "[V21.B16]",
"[V17.B16, V18.B16]": "[V17.B16, V18.B16]",
"[V18.D1, V19.D1, V20.D1]": "[V18.D1, V19.D1, V20.D1]",
"[V14.B16, V15.B16, V16.B16, V17.B16]": "[V14.B16, V15.B16, V16.B16, V17.B16]",
// Element selectors.
"V31.B[15]": "V31.B[15]",
"V19.S[0]": "V19.S[0]",
"V1.D[1]": "V1.D[1]",
"V11.B[11], V16.B[12]": "V11.B[11], V16.B[12]",
// Lists beside address operands, on either side.
"32(R1), [V2.B16, V3.B16]": "32(R1), [V2.B16, V3.B16]",
"[V2.S4, V3.S4], (R14)": "[V2.S4, V3.S4], (R14)",
"(R24), [V18.D1, V19.D1]": "(R24), [V18.D1, V19.D1]",
// A spaced spelling canonicalises to the tight one.
"[ V21.B16 ]": "[V21.B16]",
"V31.B [15]": "V31.B[15]",
}
for in, want := range cases {
toks := lexOperands(in)
if got := renderOps(toks); got != want {
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
}
}
}
// TestSIMDBracketRoundTrip formats whole functions carrying the bracket
// shapes of the arm64 vector kernels and pins the output byte for byte. The
// brackets are load-bearing: formatting must not change what the file
// assembles to, so the formatted text keeps every bracket, re-formats to
// itself and still parses cleanly.
func TestSIMDBracketRoundTrip(t *testing.T) {
in := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"VDUP V31.B[15], R3\n" +
"VTBL V22.B16, [V28.B16], V11.B16\n" +
"VLD1 (R2), [V21.B16]\n" +
"VMOVQ $0x70, $0x80, V10\n" +
"RET\n"
want := "#include \"textflag.h\"\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"\tVDUP V31.B[15], R3\n" +
"\tVTBL V22.B16, [V28.B16], V11.B16\n" +
"\tVLD1 (R2), [V21.B16]\n" +
"\tVMOVQ $0x70, $0x80, V10\n" +
"\tRET\n"
got := Source(in)
if got != want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if again := Source(got); again != got {
t.Fatalf("not idempotent:\n%q", again)
}
if n := strings.Count(got, "["); n != 3 {
t.Errorf("output carries %d '[', want 3:\n%s", n, got)
}
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
t.Errorf("formatted output no longer parses: %v", errs)
}
}
// TestBracketFormsRoundTrip runs every bracket shape of the vector kernels
// through a full format pass as its own single-instruction function, where
// the canonical form is the line itself indented: formatting must be a no-op
// on each, so no bracket moves, vanishes or gains a space.
func TestBracketFormsRoundTrip(t *testing.T) {
for _, instr := range []string{
"VDUP V31.B[15], V18",
"VDUP V19.S[3], V18.S4",
"VDUP V1.D[1], V2.D2",
"VMOV V13.S[0], R20",
"VMOV V11.B[11], V16.B[12]",
"VMOV R20, V21.B[2]",
"VTBL V22.B16, [V28.B16], V11.B16",
"VTBL V18.B8, [V17.B16, V18.B16], V22.B8",
"VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8",
"VLD1 (R2), [V21.B16]",
"VLD1 (R24), [V18.D1, V19.D1, V20.D1]",
"VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]",
"VLD1.P 32(R1), [V2.B16, V3.B16]",
"VLD1R (R1), [V9.B8]",
"VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]",
"VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)",
"VST1.P [V2.B16], (R1)",
"VST1.P [V2.B16, V3.B16], 32(R1)",
"VMOVQ $0x70, $0x80, V10",
} {
src := "TEXT ·f(SB), NOSPLIT, $0\n" + instr + "\nRET\n"
want := "TEXT ·f(SB), NOSPLIT, $0\n\t" + instr + "\n\tRET\n"
got := Source(src)
if got != want {
t.Errorf("formatting %q:\n got %q\n want %q", instr, got, want)
continue
}
if again := Source(got); again != got {
t.Errorf("not idempotent for %q:\n%q", instr, again)
}
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
t.Errorf("formatted output of %q no longer parses: %v", instr, errs)
}
}
}
// TestFlagListRoundTrip pins the '|' flag separator and the <ABIInternal>
// marker through a full format pass: the bars the Go toolchain requires and
// the ABI bracket must survive byte for byte, on TEXT and GLOBL alike.
func TestFlagListRoundTrip(t *testing.T) {
for _, in := range []string{
"TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0\n\tRET\n",
"TEXT ·foo<ABIInternal>(SB), NOSPLIT, $-0-24\n\tRET\n",
"GLOBL ·mask(SB), RODATA|NOPTR, $8\n",
} {
if got := Source(in); got != in {
t.Fatalf("flag list did not round-trip:\n--- got ---\n%q\n--- want ---\n%q", got, in)
}
}
}
// TestCRLFInputIsNormalisedToLF checks that a CRLF file comes out with
// uniform LF endings: a // comment must not carry its line's trailing \r
// into the output.
func TestCRLFInputIsNormalisedToLF(t *testing.T) {
in := "// func f()\r\nTEXT ·f(SB), NOSPLIT, $0\r\nRET\r\n"
want := "// func f()\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n"
got := Source(in)
if got != want {
t.Fatalf("CRLF formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
}
if strings.Contains(got, "\r") {
t.Fatalf("output still contains CR: %q", got)
}
}
// TestSemicolonSeparators pins the treatment of ';' statement separators.
// The separator is load-bearing on the assembly path, where the parser reads
// semicolon-separated statements: a formatter that drops it fuses two
// statements into a line the assembler rejects, which is data corruption.
// Each row pins the canonical spelling, one space after the ';', tight
// before it, the way the toolchain's own sources and listings write it.
func TestSemicolonSeparators(t *testing.T) {
cases := []struct {
name string
in string
want string
}{
{
name: "between instructions, tight",
in: "TEXT ·f(SB), $0\nBYTE $0x48;BYTE $0xc7\nRET\n",
want: "TEXT ·f(SB), $0\n\tBYTE $0x48; BYTE $0xc7\n\tRET\n",
},
{
name: "between instructions, spaced",
in: "TEXT ·f(SB), $0\nBYTE $0x48 ; BYTE $0xc7\nRET\n",
want: "TEXT ·f(SB), $0\n\tBYTE $0x48; BYTE $0xc7\n\tRET\n",
},
{
name: "after a label",
in: "TEXT ·f(SB), $0\nlabel: BYTE $1; BYTE $2\nRET\n",
want: "TEXT ·f(SB), $0\nlabel:\n\tBYTE $1; BYTE $2\n\tRET\n",
},
{
// The continuation-spliced macro shape of the runtime sources:
// the lexer makes one logical line of the backslash continuations.
name: "inside a macro body, continued",
in: "#define MOVLTOREG(v, off) \\\n\tMOVL $v, AX; \\\n\tMOVL AX, ret+off(FP)\n",
want: "#define MOVLTOREG(v, off) MOVL $v, AX; MOVL AX, ret+off(FP)\n",
},
{
name: "inside a macro body, one line",
in: "#define PEAS BYTE $0x0a; BYTE $0x0b\n",
want: "#define PEAS BYTE $0x0a; BYTE $0x0b\n",
},
{
name: "several separators in one line",
in: "TEXT ·f(SB), $0\nBYTE $1; BYTE $2; BYTE $3\nRET\n",
want: "TEXT ·f(SB), $0\n\tBYTE $1; BYTE $2; BYTE $3\n\tRET\n",
},
{
name: "two separators back to back",
in: "TEXT ·f(SB), $0\nBYTE $1;; BYTE $2\nRET\n",
want: "TEXT ·f(SB), $0\n\tBYTE $1;; BYTE $2\n\tRET\n",
},
{
name: "inside a line comment, untouched",
in: "TEXT ·f(SB), $0\n// keep; the; separators\nBYTE $1\nRET\n",
want: "TEXT ·f(SB), $0\n\t// keep; the; separators\n\tBYTE $1\n\tRET\n",
},
{
name: "after a statement, before a comment",
in: "TEXT ·f(SB), $0\nMOVQ AX, BX; // tail\nRET\n",
want: "TEXT ·f(SB), $0\n\tMOVQ AX, BX; // tail\n\tRET\n",
},
{
name: "last character on a line",
in: "TEXT ·f(SB), $0\nBYTE $1;\nRET\n",
want: "TEXT ·f(SB), $0\n\tBYTE $1;\n\tRET\n",
},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got := Source(tc.in)
if got != tc.want {
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, tc.want)
}
if again := Source(got); again != got {
t.Fatalf("not idempotent:\n%q", again)
}
if in, out := strings.Count(tc.in, ";"), strings.Count(got, ";"); in != out {
t.Fatalf("semicolon count changed: %d -> %d\n%s", in, out, got)
}
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
t.Fatalf("formatted output no longer parses: %v", errs)
}
})
}
}
// TestSemicolonStatementRoundTrip proves the formatter's contract on the
// path where ';' separates statements: parse the source the way the
// assembler does, format it, re-parse the formatted text and compare the
// statement sequence. Raw operand texts are token-joined, so they are
// insensitive to the whitespace a format pass chooses, and the comparison
// can only fail when a token is lost: dropping a ';' fuses two statements
// into one, exactly the corruption the released formatter committed.
func TestSemicolonStatementRoundTrip(t *testing.T) {
src := "#define MOVLTOREG(v, off) \\\n" +
"\tMOVL $v, AX; \\\n" +
"\tMOVL AX, ret+off(FP)\n" +
"\n" +
"TEXT ·f(SB), NOSPLIT, $0\n" +
"BYTE $0x48; BYTE $0xc7\n" +
"first: BYTE $1; BYTE $2\n" +
"MOVLTOREG($42, 0)\n" +
"RET\n"
before, errs := parser.ParseWithOptions("in.s", src, parser.Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("source does not parse: %v", errs)
}
formatted := Source(src)
after, errs := parser.ParseWithOptions("in.s", formatted, parser.Options{Expand: true})
if len(errs) > 0 {
t.Fatalf("formatted source does not parse: %v", errs)
}
want, got := stmtSignature(before), stmtSignature(after)
if !slices.Equal(got, want) {
t.Fatalf("statement sequence changed:\n--- before ---\n%q\n--- after ---\n%q", want, got)
}
if again := Source(formatted); again != formatted {
t.Fatalf("not idempotent:\n%q", again)
}
// The two BYTE statements on the first line must stay two: one fused
// statement here is the exact defect this package once shipped.
var bytes []string
for _, stmt := range stmtSignature(after) {
if rest, ok := strings.CutPrefix(stmt, "instr BYTE "); ok {
bytes = append(bytes, rest)
}
}
if want := []string{"$ 0x48", "$ 0xc7", "$ 1", "$ 2"}; !slices.Equal(bytes, want) {
t.Fatalf("BYTE statements after expansion = %q, want %q", bytes, want)
}
}
// stmtSignature flattens a parsed file into one string per declaration and
// statement, in source order. Every component is token-derived, so the
// signature is stable across format passes and moves only when a token is
// lost or gained.
func stmtSignature(f *ast.File) []string {
var out []string
for _, d := range f.Decls {
switch d := d.(type) {
case *ast.Text:
out = append(out, "text "+d.Name.Raw)
for _, s := range d.Body {
out = append(out, stmtText(s))
}
case *ast.Globl:
out = append(out, "globl "+d.Name.Raw)
case *ast.Data:
out = append(out, "data "+d.Name.Raw)
case *ast.Include:
out = append(out, "include "+d.Header.Text)
case *ast.Preproc:
out = append(out, "preproc "+d.Raw)
}
}
for _, s := range f.Orphans {
out = append(out, stmtText(s))
}
return out
}
// stmtText renders one statement for stmtSignature.
func stmtText(s ast.Stmt) string {
switch s := s.(type) {
case *ast.Label:
return "label " + s.Name.Text
case *ast.Instr:
parts := make([]string, 0, len(s.Operands)+1)
parts = append(parts, s.Mnemonic.Text)
for _, op := range s.Operands {
parts = append(parts, op.Raw)
}
return "instr " + strings.Join(parts, " ")
default:
return "stmt"
}
}
// lexOperands lexes a single operand string and drops the EOF token.
func lexOperands(s string) []token.Token {
toks := lexer.Tokenize(s)
return toks[:len(toks)-1] // drop trailing EOF
}
func TestIdempotent(t *testing.T) {
src, err := os.ReadFile("../testdata/sample_amd64.s")
if err != nil {
t.Fatal(err)
}
once := Source(string(src))
twice := Source(once)
if once != twice {
t.Fatal("formatting is not idempotent on the fixture")
}
}
// TestRoundTrip checks that formatting produces source that still parses
// cleanly on the in-repository fixture.
func TestRoundTrip(t *testing.T) {
files := []string{"../testdata/sample_amd64.s"}
for _, path := range files {
src, err := os.ReadFile(path)
if err != nil {
t.Fatal(err)
}
formatted := Source(string(src))
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
t.Errorf("formatted %s no longer parses: %v", path, errs)
}
if strings.TrimSpace(formatted) == "" {
t.Errorf("formatted %s is empty", path)
}
}
}