2026-07-06 09:49:50 +02:00
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
|
|
|
|
|
|
package format
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
"os"
|
2026-09-20 19:14:43 +02:00
|
|
|
"slices"
|
2026-07-06 09:49:50 +02:00
|
|
|
"strings"
|
|
|
|
|
"testing"
|
|
|
|
|
|
2026-09-20 19:14:43 +02:00
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
2026-07-06 09:49:50 +02:00
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
func TestGolden(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"MOVQ swin_base+0(FP), SI\n" +
|
|
|
|
|
"LEAQ (SI)(BX*4), R9\n" +
|
|
|
|
|
"ANDQ $-8, R10\n" +
|
|
|
|
|
"VFMADD231PD Z14, Z12, Z10\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tMOVQ swin_base+0(FP), SI\n" +
|
|
|
|
|
"\tLEAQ (SI)(BX*4), R9\n" +
|
|
|
|
|
"\tANDQ $-8, R10\n" +
|
|
|
|
|
"\tVFMADD231PD Z14, Z12, Z10\n" +
|
|
|
|
|
"\tRET\n"
|
|
|
|
|
|
2026-08-29 17:12:53 +02:00
|
|
|
got := Source(in)
|
2026-07-06 09:49:50 +02:00
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-11 17:36:52 +02:00
|
|
|
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
|
2026-09-16 23:12:31 +02:00
|
|
|
// sits at column 0 even when another function (ending in RET) precedes it;
|
2026-07-11 17:36:52 +02:00
|
|
|
// the RET must terminate the previous body for indentation purposes.
|
|
|
|
|
func TestDocCommentIndent(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func first()\n" +
|
|
|
|
|
"TEXT ·first(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"XORQ AX, AX\n" +
|
|
|
|
|
"RET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func second()\n" +
|
|
|
|
|
"TEXT ·second(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func first()\n" +
|
|
|
|
|
"TEXT ·first(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tXORQ AX, AX\n" +
|
|
|
|
|
"\tRET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func second()\n" +
|
|
|
|
|
"TEXT ·second(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tRET\n"
|
|
|
|
|
|
2026-08-29 17:12:53 +02:00
|
|
|
got := Source(in)
|
2026-07-11 17:36:52 +02:00
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
// Body comments stay indented.
|
|
|
|
|
body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n"
|
2026-08-29 17:12:53 +02:00
|
|
|
gotBody := Source(body)
|
2026-07-11 17:36:52 +02:00
|
|
|
if !strings.Contains(gotBody, "\t// inside the body\n") {
|
|
|
|
|
t.Fatalf("body comment must stay indented:\n%q", gotBody)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-12 21:24:41 +02:00
|
|
|
// TestBlankLines checks the blank-line canonicalisation: exactly one blank
|
|
|
|
|
// line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed
|
|
|
|
|
// to one, and no blank forced after TEXT, between stacked labels, or at the
|
|
|
|
|
// top of the file. Leading comments belong to the block they precede.
|
|
|
|
|
func TestBlankLines(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"first:\n" + // first label: no blank after TEXT
|
|
|
|
|
"XORQ AX, AX\n" +
|
2026-09-19 23:48:47 +02:00
|
|
|
"JMP next\n" + // unlabelled glue: fmt inserts a blank before next:
|
2026-07-12 21:24:41 +02:00
|
|
|
"next:\n" +
|
|
|
|
|
"stacked:\n" + // stacked labels share an address: no blank between
|
|
|
|
|
"INCQ AX\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"\n" + // two blanks collapse to one
|
|
|
|
|
"// separated block\n" + // comment belongs to the label below
|
|
|
|
|
"later:\n" +
|
|
|
|
|
"RET\n" +
|
|
|
|
|
"// func g()\n" + // doc comment: blank goes before it
|
|
|
|
|
"TEXT ·g(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"RET\n" +
|
|
|
|
|
"GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL…
|
|
|
|
|
"DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA
|
|
|
|
|
"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"\n" // trailing blanks dropped
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"first:\n" +
|
|
|
|
|
"\tXORQ AX, AX\n" +
|
|
|
|
|
"\tJMP next\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"next:\n" +
|
|
|
|
|
"stacked:\n" +
|
|
|
|
|
"\tINCQ AX\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"\t// separated block\n" + // body comment before a label stays indented
|
|
|
|
|
"later:\n" +
|
|
|
|
|
"\tRET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func g()\n" +
|
|
|
|
|
"TEXT ·g(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tRET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"GLOBL ·mask(SB), RODATA, $8\n" +
|
|
|
|
|
"DATA ·mask+0(SB)/4, $1\n"
|
|
|
|
|
|
2026-08-29 17:12:53 +02:00
|
|
|
got := Source(in)
|
2026-07-12 21:24:41 +02:00
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
2026-08-29 17:12:53 +02:00
|
|
|
if again := Source(got); again != got {
|
2026-07-12 21:24:41 +02:00
|
|
|
t.Fatalf("not idempotent:\n%q", again)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-06 09:49:50 +02:00
|
|
|
func TestOperandSpacing(t *testing.T) {
|
|
|
|
|
cases := map[string]string{
|
|
|
|
|
"4(SI)": "4(SI)",
|
|
|
|
|
"(SI)(BX*4)": "(SI)(BX*4)",
|
|
|
|
|
"$-8": "$-8",
|
|
|
|
|
"$0x80020100": "$0x80020100",
|
|
|
|
|
"swin_base+0(FP)": "swin_base+0(FP)",
|
|
|
|
|
"mask24<>(SB)": "mask24<>(SB)",
|
|
|
|
|
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
|
2026-09-19 23:48:47 +02:00
|
|
|
"NOSPLIT|DUPOK": "NOSPLIT|DUPOK",
|
2026-07-06 09:49:50 +02:00
|
|
|
}
|
|
|
|
|
for in, want := range cases {
|
|
|
|
|
toks := lexOperands(in)
|
|
|
|
|
if got := renderOps(toks); got != want {
|
|
|
|
|
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 09:58:25 +02:00
|
|
|
// TestVectorBracketSpacing pins the square-bracket operand forms of the
|
|
|
|
|
// arm64 and loong64 vector syntaxes. The lexer emits '[' and ']' as Illegal
|
|
|
|
|
// tokens carrying their spelling, and renderOps must glue them back exactly
|
|
|
|
|
// where they were: a register list and an element selector are load-bearing
|
|
|
|
|
// operands the assembler reads out of the operand text, so no bracket may be
|
|
|
|
|
// dropped, and the canonical spelling inside the brackets is tight.
|
|
|
|
|
func TestVectorBracketSpacing(t *testing.T) {
|
|
|
|
|
cases := map[string]string{
|
|
|
|
|
// Register lists of one to four registers.
|
|
|
|
|
"[V21.B16]": "[V21.B16]",
|
|
|
|
|
"[V17.B16, V18.B16]": "[V17.B16, V18.B16]",
|
|
|
|
|
"[V18.D1, V19.D1, V20.D1]": "[V18.D1, V19.D1, V20.D1]",
|
|
|
|
|
"[V14.B16, V15.B16, V16.B16, V17.B16]": "[V14.B16, V15.B16, V16.B16, V17.B16]",
|
|
|
|
|
// Element selectors.
|
|
|
|
|
"V31.B[15]": "V31.B[15]",
|
|
|
|
|
"V19.S[0]": "V19.S[0]",
|
|
|
|
|
"V1.D[1]": "V1.D[1]",
|
|
|
|
|
"V11.B[11], V16.B[12]": "V11.B[11], V16.B[12]",
|
|
|
|
|
// Lists beside address operands, on either side.
|
|
|
|
|
"32(R1), [V2.B16, V3.B16]": "32(R1), [V2.B16, V3.B16]",
|
|
|
|
|
"[V2.S4, V3.S4], (R14)": "[V2.S4, V3.S4], (R14)",
|
|
|
|
|
"(R24), [V18.D1, V19.D1]": "(R24), [V18.D1, V19.D1]",
|
|
|
|
|
// A spaced spelling canonicalises to the tight one.
|
|
|
|
|
"[ V21.B16 ]": "[V21.B16]",
|
|
|
|
|
"V31.B [15]": "V31.B[15]",
|
|
|
|
|
}
|
|
|
|
|
for in, want := range cases {
|
|
|
|
|
toks := lexOperands(in)
|
|
|
|
|
if got := renderOps(toks); got != want {
|
|
|
|
|
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestSIMDBracketRoundTrip formats whole functions carrying the bracket
|
|
|
|
|
// shapes of the arm64 vector kernels and pins the output byte for byte. The
|
|
|
|
|
// brackets are load-bearing: formatting must not change what the file
|
|
|
|
|
// assembles to, so the formatted text keeps every bracket, re-formats to
|
|
|
|
|
// itself and still parses cleanly.
|
|
|
|
|
func TestSIMDBracketRoundTrip(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"VDUP V31.B[15], R3\n" +
|
|
|
|
|
"VTBL V22.B16, [V28.B16], V11.B16\n" +
|
|
|
|
|
"VLD1 (R2), [V21.B16]\n" +
|
|
|
|
|
"VMOVQ $0x70, $0x80, V10\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tVDUP V31.B[15], R3\n" +
|
|
|
|
|
"\tVTBL V22.B16, [V28.B16], V11.B16\n" +
|
|
|
|
|
"\tVLD1 (R2), [V21.B16]\n" +
|
|
|
|
|
"\tVMOVQ $0x70, $0x80, V10\n" +
|
|
|
|
|
"\tRET\n"
|
|
|
|
|
|
|
|
|
|
got := Source(in)
|
|
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
if again := Source(got); again != got {
|
|
|
|
|
t.Fatalf("not idempotent:\n%q", again)
|
|
|
|
|
}
|
|
|
|
|
if n := strings.Count(got, "["); n != 3 {
|
|
|
|
|
t.Errorf("output carries %d '[', want 3:\n%s", n, got)
|
|
|
|
|
}
|
|
|
|
|
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
|
|
|
|
|
t.Errorf("formatted output no longer parses: %v", errs)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestBracketFormsRoundTrip runs every bracket shape of the vector kernels
|
|
|
|
|
// through a full format pass as its own single-instruction function, where
|
|
|
|
|
// the canonical form is the line itself indented: formatting must be a no-op
|
|
|
|
|
// on each, so no bracket moves, vanishes or gains a space.
|
|
|
|
|
func TestBracketFormsRoundTrip(t *testing.T) {
|
|
|
|
|
for _, instr := range []string{
|
|
|
|
|
"VDUP V31.B[15], V18",
|
|
|
|
|
"VDUP V19.S[3], V18.S4",
|
|
|
|
|
"VDUP V1.D[1], V2.D2",
|
|
|
|
|
"VMOV V13.S[0], R20",
|
|
|
|
|
"VMOV V11.B[11], V16.B[12]",
|
|
|
|
|
"VMOV R20, V21.B[2]",
|
|
|
|
|
"VTBL V22.B16, [V28.B16], V11.B16",
|
|
|
|
|
"VTBL V18.B8, [V17.B16, V18.B16], V22.B8",
|
|
|
|
|
"VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8",
|
|
|
|
|
"VLD1 (R2), [V21.B16]",
|
|
|
|
|
"VLD1 (R24), [V18.D1, V19.D1, V20.D1]",
|
|
|
|
|
"VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]",
|
|
|
|
|
"VLD1.P 32(R1), [V2.B16, V3.B16]",
|
|
|
|
|
"VLD1R (R1), [V9.B8]",
|
|
|
|
|
"VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]",
|
|
|
|
|
"VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)",
|
|
|
|
|
"VST1.P [V2.B16], (R1)",
|
|
|
|
|
"VST1.P [V2.B16, V3.B16], 32(R1)",
|
|
|
|
|
"VMOVQ $0x70, $0x80, V10",
|
|
|
|
|
} {
|
|
|
|
|
src := "TEXT ·f(SB), NOSPLIT, $0\n" + instr + "\nRET\n"
|
|
|
|
|
want := "TEXT ·f(SB), NOSPLIT, $0\n\t" + instr + "\n\tRET\n"
|
|
|
|
|
got := Source(src)
|
|
|
|
|
if got != want {
|
|
|
|
|
t.Errorf("formatting %q:\n got %q\n want %q", instr, got, want)
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
if again := Source(got); again != got {
|
|
|
|
|
t.Errorf("not idempotent for %q:\n%q", instr, again)
|
|
|
|
|
}
|
|
|
|
|
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
|
|
|
|
|
t.Errorf("formatted output of %q no longer parses: %v", instr, errs)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-19 23:48:47 +02:00
|
|
|
// TestFlagListRoundTrip pins the '|' flag separator and the <ABIInternal>
|
|
|
|
|
// marker through a full format pass: the bars the Go toolchain requires and
|
|
|
|
|
// the ABI bracket must survive byte for byte, on TEXT and GLOBL alike.
|
|
|
|
|
func TestFlagListRoundTrip(t *testing.T) {
|
|
|
|
|
for _, in := range []string{
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0\n\tRET\n",
|
|
|
|
|
"TEXT ·foo<ABIInternal>(SB), NOSPLIT, $-0-24\n\tRET\n",
|
|
|
|
|
"GLOBL ·mask(SB), RODATA|NOPTR, $8\n",
|
|
|
|
|
} {
|
|
|
|
|
if got := Source(in); got != in {
|
|
|
|
|
t.Fatalf("flag list did not round-trip:\n--- got ---\n%q\n--- want ---\n%q", got, in)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestCRLFInputIsNormalisedToLF checks that a CRLF file comes out with
|
|
|
|
|
// uniform LF endings: a // comment must not carry its line's trailing \r
|
|
|
|
|
// into the output.
|
|
|
|
|
func TestCRLFInputIsNormalisedToLF(t *testing.T) {
|
|
|
|
|
in := "// func f()\r\nTEXT ·f(SB), NOSPLIT, $0\r\nRET\r\n"
|
|
|
|
|
want := "// func f()\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n"
|
|
|
|
|
got := Source(in)
|
|
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("CRLF formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
if strings.Contains(got, "\r") {
|
|
|
|
|
t.Fatalf("output still contains CR: %q", got)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 19:14:43 +02:00
|
|
|
// TestSemicolonSeparators pins the treatment of ';' statement separators.
|
|
|
|
|
// The separator is load-bearing on the assembly path, where the parser reads
|
|
|
|
|
// semicolon-separated statements: a formatter that drops it fuses two
|
|
|
|
|
// statements into a line the assembler rejects, which is data corruption.
|
|
|
|
|
// Each row pins the canonical spelling, one space after the ';', tight
|
|
|
|
|
// before it, the way the toolchain's own sources and listings write it.
|
|
|
|
|
func TestSemicolonSeparators(t *testing.T) {
|
|
|
|
|
cases := []struct {
|
|
|
|
|
name string
|
|
|
|
|
in string
|
|
|
|
|
want string
|
|
|
|
|
}{
|
|
|
|
|
{
|
|
|
|
|
name: "between instructions, tight",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nBYTE $0x48;BYTE $0xc7\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\tBYTE $0x48; BYTE $0xc7\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "between instructions, spaced",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nBYTE $0x48 ; BYTE $0xc7\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\tBYTE $0x48; BYTE $0xc7\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "after a label",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nlabel: BYTE $1; BYTE $2\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\nlabel:\n\tBYTE $1; BYTE $2\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
// The continuation-spliced macro shape of the runtime sources:
|
|
|
|
|
// the lexer makes one logical line of the backslash continuations.
|
|
|
|
|
name: "inside a macro body, continued",
|
|
|
|
|
in: "#define MOVLTOREG(v, off) \\\n\tMOVL $v, AX; \\\n\tMOVL AX, ret+off(FP)\n",
|
|
|
|
|
want: "#define MOVLTOREG(v, off) MOVL $v, AX; MOVL AX, ret+off(FP)\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "inside a macro body, one line",
|
|
|
|
|
in: "#define PEAS BYTE $0x0a; BYTE $0x0b\n",
|
|
|
|
|
want: "#define PEAS BYTE $0x0a; BYTE $0x0b\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "several separators in one line",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nBYTE $1; BYTE $2; BYTE $3\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\tBYTE $1; BYTE $2; BYTE $3\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "two separators back to back",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nBYTE $1;; BYTE $2\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\tBYTE $1;; BYTE $2\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "inside a line comment, untouched",
|
|
|
|
|
in: "TEXT ·f(SB), $0\n// keep; the; separators\nBYTE $1\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\t// keep; the; separators\n\tBYTE $1\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "after a statement, before a comment",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nMOVQ AX, BX; // tail\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\tMOVQ AX, BX; // tail\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
name: "last character on a line",
|
|
|
|
|
in: "TEXT ·f(SB), $0\nBYTE $1;\nRET\n",
|
|
|
|
|
want: "TEXT ·f(SB), $0\n\tBYTE $1;\n\tRET\n",
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
for _, tc := range cases {
|
|
|
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
|
|
|
got := Source(tc.in)
|
|
|
|
|
if got != tc.want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, tc.want)
|
|
|
|
|
}
|
|
|
|
|
if again := Source(got); again != got {
|
|
|
|
|
t.Fatalf("not idempotent:\n%q", again)
|
|
|
|
|
}
|
|
|
|
|
if in, out := strings.Count(tc.in, ";"), strings.Count(got, ";"); in != out {
|
|
|
|
|
t.Fatalf("semicolon count changed: %d -> %d\n%s", in, out, got)
|
|
|
|
|
}
|
|
|
|
|
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
|
|
|
|
|
t.Fatalf("formatted output no longer parses: %v", errs)
|
|
|
|
|
}
|
|
|
|
|
})
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestSemicolonStatementRoundTrip proves the formatter's contract on the
|
|
|
|
|
// path where ';' separates statements: parse the source the way the
|
|
|
|
|
// assembler does, format it, re-parse the formatted text and compare the
|
|
|
|
|
// statement sequence. Raw operand texts are token-joined, so they are
|
|
|
|
|
// insensitive to the whitespace a format pass chooses, and the comparison
|
|
|
|
|
// can only fail when a token is lost: dropping a ';' fuses two statements
|
|
|
|
|
// into one, exactly the corruption the released formatter committed.
|
|
|
|
|
func TestSemicolonStatementRoundTrip(t *testing.T) {
|
|
|
|
|
src := "#define MOVLTOREG(v, off) \\\n" +
|
|
|
|
|
"\tMOVL $v, AX; \\\n" +
|
|
|
|
|
"\tMOVL AX, ret+off(FP)\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"BYTE $0x48; BYTE $0xc7\n" +
|
|
|
|
|
"first: BYTE $1; BYTE $2\n" +
|
|
|
|
|
"MOVLTOREG($42, 0)\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
before, errs := parser.ParseWithOptions("in.s", src, parser.Options{Expand: true})
|
|
|
|
|
if len(errs) > 0 {
|
|
|
|
|
t.Fatalf("source does not parse: %v", errs)
|
|
|
|
|
}
|
|
|
|
|
formatted := Source(src)
|
|
|
|
|
after, errs := parser.ParseWithOptions("in.s", formatted, parser.Options{Expand: true})
|
|
|
|
|
if len(errs) > 0 {
|
|
|
|
|
t.Fatalf("formatted source does not parse: %v", errs)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
want, got := stmtSignature(before), stmtSignature(after)
|
|
|
|
|
if !slices.Equal(got, want) {
|
|
|
|
|
t.Fatalf("statement sequence changed:\n--- before ---\n%q\n--- after ---\n%q", want, got)
|
|
|
|
|
}
|
|
|
|
|
if again := Source(formatted); again != formatted {
|
|
|
|
|
t.Fatalf("not idempotent:\n%q", again)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// The two BYTE statements on the first line must stay two: one fused
|
|
|
|
|
// statement here is the exact defect this package once shipped.
|
|
|
|
|
var bytes []string
|
|
|
|
|
for _, stmt := range stmtSignature(after) {
|
|
|
|
|
if rest, ok := strings.CutPrefix(stmt, "instr BYTE "); ok {
|
|
|
|
|
bytes = append(bytes, rest)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if want := []string{"$ 0x48", "$ 0xc7", "$ 1", "$ 2"}; !slices.Equal(bytes, want) {
|
|
|
|
|
t.Fatalf("BYTE statements after expansion = %q, want %q", bytes, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// stmtSignature flattens a parsed file into one string per declaration and
|
|
|
|
|
// statement, in source order. Every component is token-derived, so the
|
|
|
|
|
// signature is stable across format passes and moves only when a token is
|
|
|
|
|
// lost or gained.
|
|
|
|
|
func stmtSignature(f *ast.File) []string {
|
|
|
|
|
var out []string
|
|
|
|
|
for _, d := range f.Decls {
|
|
|
|
|
switch d := d.(type) {
|
|
|
|
|
case *ast.Text:
|
|
|
|
|
out = append(out, "text "+d.Name.Raw)
|
|
|
|
|
for _, s := range d.Body {
|
|
|
|
|
out = append(out, stmtText(s))
|
|
|
|
|
}
|
|
|
|
|
case *ast.Globl:
|
|
|
|
|
out = append(out, "globl "+d.Name.Raw)
|
|
|
|
|
case *ast.Data:
|
|
|
|
|
out = append(out, "data "+d.Name.Raw)
|
|
|
|
|
case *ast.Include:
|
|
|
|
|
out = append(out, "include "+d.Header.Text)
|
|
|
|
|
case *ast.Preproc:
|
|
|
|
|
out = append(out, "preproc "+d.Raw)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
for _, s := range f.Orphans {
|
|
|
|
|
out = append(out, stmtText(s))
|
|
|
|
|
}
|
|
|
|
|
return out
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// stmtText renders one statement for stmtSignature.
|
|
|
|
|
func stmtText(s ast.Stmt) string {
|
|
|
|
|
switch s := s.(type) {
|
|
|
|
|
case *ast.Label:
|
|
|
|
|
return "label " + s.Name.Text
|
|
|
|
|
case *ast.Instr:
|
|
|
|
|
parts := make([]string, 0, len(s.Operands)+1)
|
|
|
|
|
parts = append(parts, s.Mnemonic.Text)
|
|
|
|
|
for _, op := range s.Operands {
|
|
|
|
|
parts = append(parts, op.Raw)
|
|
|
|
|
}
|
|
|
|
|
return "instr " + strings.Join(parts, " ")
|
|
|
|
|
default:
|
|
|
|
|
return "stmt"
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-06 09:49:50 +02:00
|
|
|
// lexOperands lexes a single operand string and drops the EOF token.
|
|
|
|
|
func lexOperands(s string) []token.Token {
|
|
|
|
|
toks := lexer.Tokenize(s)
|
|
|
|
|
return toks[:len(toks)-1] // drop trailing EOF
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func TestIdempotent(t *testing.T) {
|
|
|
|
|
src, err := os.ReadFile("../testdata/sample_amd64.s")
|
|
|
|
|
if err != nil {
|
|
|
|
|
t.Fatal(err)
|
|
|
|
|
}
|
2026-08-29 17:12:53 +02:00
|
|
|
once := Source(string(src))
|
|
|
|
|
twice := Source(once)
|
2026-07-06 09:49:50 +02:00
|
|
|
if once != twice {
|
|
|
|
|
t.Fatal("formatting is not idempotent on the fixture")
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestRoundTrip checks that formatting produces source that still parses
|
2026-08-07 21:28:24 +02:00
|
|
|
// cleanly on the in-repository fixture.
|
2026-07-06 09:49:50 +02:00
|
|
|
func TestRoundTrip(t *testing.T) {
|
|
|
|
|
files := []string{"../testdata/sample_amd64.s"}
|
|
|
|
|
for _, path := range files {
|
|
|
|
|
src, err := os.ReadFile(path)
|
|
|
|
|
if err != nil {
|
|
|
|
|
t.Fatal(err)
|
|
|
|
|
}
|
2026-08-29 17:12:53 +02:00
|
|
|
formatted := Source(string(src))
|
2026-07-06 09:49:50 +02:00
|
|
|
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
|
|
|
|
|
t.Errorf("formatted %s no longer parses: %v", path, errs)
|
|
|
|
|
}
|
|
|
|
|
if strings.TrimSpace(formatted) == "" {
|
|
|
|
|
t.Errorf("formatted %s is empty", path)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|