2026-07-06 09:49:50 +02:00
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
|
|
|
|
|
|
package format
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
"os"
|
|
|
|
|
"strings"
|
|
|
|
|
"testing"
|
|
|
|
|
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/lexer"
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
|
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/token"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
func TestGolden(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"MOVQ swin_base+0(FP), SI\n" +
|
|
|
|
|
"LEAQ (SI)(BX*4), R9\n" +
|
|
|
|
|
"ANDQ $-8, R10\n" +
|
|
|
|
|
"VFMADD231PD Z14, Z12, Z10\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tMOVQ swin_base+0(FP), SI\n" +
|
|
|
|
|
"\tLEAQ (SI)(BX*4), R9\n" +
|
|
|
|
|
"\tANDQ $-8, R10\n" +
|
|
|
|
|
"\tVFMADD231PD Z14, Z12, Z10\n" +
|
|
|
|
|
"\tRET\n"
|
|
|
|
|
|
2026-08-29 17:12:53 +02:00
|
|
|
got := Source(in)
|
2026-07-06 09:49:50 +02:00
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-11 17:36:52 +02:00
|
|
|
// TestDocCommentIndent checks that a doc comment preceding a TEXT directive
|
2026-09-16 23:12:31 +02:00
|
|
|
// sits at column 0 even when another function (ending in RET) precedes it;
|
2026-07-11 17:36:52 +02:00
|
|
|
// the RET must terminate the previous body for indentation purposes.
|
|
|
|
|
func TestDocCommentIndent(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func first()\n" +
|
|
|
|
|
"TEXT ·first(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"XORQ AX, AX\n" +
|
|
|
|
|
"RET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func second()\n" +
|
|
|
|
|
"TEXT ·second(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func first()\n" +
|
|
|
|
|
"TEXT ·first(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tXORQ AX, AX\n" +
|
|
|
|
|
"\tRET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func second()\n" +
|
|
|
|
|
"TEXT ·second(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tRET\n"
|
|
|
|
|
|
2026-08-29 17:12:53 +02:00
|
|
|
got := Source(in)
|
2026-07-11 17:36:52 +02:00
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
// Body comments stay indented.
|
|
|
|
|
body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n"
|
2026-08-29 17:12:53 +02:00
|
|
|
gotBody := Source(body)
|
2026-07-11 17:36:52 +02:00
|
|
|
if !strings.Contains(gotBody, "\t// inside the body\n") {
|
|
|
|
|
t.Fatalf("body comment must stay indented:\n%q", gotBody)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-12 21:24:41 +02:00
|
|
|
// TestBlankLines checks the blank-line canonicalisation: exactly one blank
|
|
|
|
|
// line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed
|
|
|
|
|
// to one, and no blank forced after TEXT, between stacked labels, or at the
|
|
|
|
|
// top of the file. Leading comments belong to the block they precede.
|
|
|
|
|
func TestBlankLines(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"first:\n" + // first label: no blank after TEXT
|
|
|
|
|
"XORQ AX, AX\n" +
|
2026-09-19 23:48:47 +02:00
|
|
|
"JMP next\n" + // unlabelled glue: fmt inserts a blank before next:
|
2026-07-12 21:24:41 +02:00
|
|
|
"next:\n" +
|
|
|
|
|
"stacked:\n" + // stacked labels share an address: no blank between
|
|
|
|
|
"INCQ AX\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"\n" + // two blanks collapse to one
|
|
|
|
|
"// separated block\n" + // comment belongs to the label below
|
|
|
|
|
"later:\n" +
|
|
|
|
|
"RET\n" +
|
|
|
|
|
"// func g()\n" + // doc comment: blank goes before it
|
|
|
|
|
"TEXT ·g(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"RET\n" +
|
|
|
|
|
"GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL…
|
|
|
|
|
"DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA
|
|
|
|
|
"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"\n" // trailing blanks dropped
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"first:\n" +
|
|
|
|
|
"\tXORQ AX, AX\n" +
|
|
|
|
|
"\tJMP next\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"next:\n" +
|
|
|
|
|
"stacked:\n" +
|
|
|
|
|
"\tINCQ AX\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"\t// separated block\n" + // body comment before a label stays indented
|
|
|
|
|
"later:\n" +
|
|
|
|
|
"\tRET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"// func g()\n" +
|
|
|
|
|
"TEXT ·g(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tRET\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"GLOBL ·mask(SB), RODATA, $8\n" +
|
|
|
|
|
"DATA ·mask+0(SB)/4, $1\n"
|
|
|
|
|
|
2026-08-29 17:12:53 +02:00
|
|
|
got := Source(in)
|
2026-07-12 21:24:41 +02:00
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
2026-08-29 17:12:53 +02:00
|
|
|
if again := Source(got); again != got {
|
2026-07-12 21:24:41 +02:00
|
|
|
t.Fatalf("not idempotent:\n%q", again)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-06 09:49:50 +02:00
|
|
|
func TestOperandSpacing(t *testing.T) {
|
|
|
|
|
cases := map[string]string{
|
|
|
|
|
"4(SI)": "4(SI)",
|
|
|
|
|
"(SI)(BX*4)": "(SI)(BX*4)",
|
|
|
|
|
"$-8": "$-8",
|
|
|
|
|
"$0x80020100": "$0x80020100",
|
|
|
|
|
"swin_base+0(FP)": "swin_base+0(FP)",
|
|
|
|
|
"mask24<>(SB)": "mask24<>(SB)",
|
|
|
|
|
"·idx16+0(SB)/4": "·idx16+0(SB)/4",
|
2026-09-19 23:48:47 +02:00
|
|
|
"NOSPLIT|DUPOK": "NOSPLIT|DUPOK",
|
2026-07-06 09:49:50 +02:00
|
|
|
}
|
|
|
|
|
for in, want := range cases {
|
|
|
|
|
toks := lexOperands(in)
|
|
|
|
|
if got := renderOps(toks); got != want {
|
|
|
|
|
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 09:58:25 +02:00
|
|
|
// TestVectorBracketSpacing pins the square-bracket operand forms of the
|
|
|
|
|
// arm64 and loong64 vector syntaxes. The lexer emits '[' and ']' as Illegal
|
|
|
|
|
// tokens carrying their spelling, and renderOps must glue them back exactly
|
|
|
|
|
// where they were: a register list and an element selector are load-bearing
|
|
|
|
|
// operands the assembler reads out of the operand text, so no bracket may be
|
|
|
|
|
// dropped, and the canonical spelling inside the brackets is tight.
|
|
|
|
|
func TestVectorBracketSpacing(t *testing.T) {
|
|
|
|
|
cases := map[string]string{
|
|
|
|
|
// Register lists of one to four registers.
|
|
|
|
|
"[V21.B16]": "[V21.B16]",
|
|
|
|
|
"[V17.B16, V18.B16]": "[V17.B16, V18.B16]",
|
|
|
|
|
"[V18.D1, V19.D1, V20.D1]": "[V18.D1, V19.D1, V20.D1]",
|
|
|
|
|
"[V14.B16, V15.B16, V16.B16, V17.B16]": "[V14.B16, V15.B16, V16.B16, V17.B16]",
|
|
|
|
|
// Element selectors.
|
|
|
|
|
"V31.B[15]": "V31.B[15]",
|
|
|
|
|
"V19.S[0]": "V19.S[0]",
|
|
|
|
|
"V1.D[1]": "V1.D[1]",
|
|
|
|
|
"V11.B[11], V16.B[12]": "V11.B[11], V16.B[12]",
|
|
|
|
|
// Lists beside address operands, on either side.
|
|
|
|
|
"32(R1), [V2.B16, V3.B16]": "32(R1), [V2.B16, V3.B16]",
|
|
|
|
|
"[V2.S4, V3.S4], (R14)": "[V2.S4, V3.S4], (R14)",
|
|
|
|
|
"(R24), [V18.D1, V19.D1]": "(R24), [V18.D1, V19.D1]",
|
|
|
|
|
// A spaced spelling canonicalises to the tight one.
|
|
|
|
|
"[ V21.B16 ]": "[V21.B16]",
|
|
|
|
|
"V31.B [15]": "V31.B[15]",
|
|
|
|
|
}
|
|
|
|
|
for in, want := range cases {
|
|
|
|
|
toks := lexOperands(in)
|
|
|
|
|
if got := renderOps(toks); got != want {
|
|
|
|
|
t.Errorf("renderOps(%q) = %q, want %q", in, got, want)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestSIMDBracketRoundTrip formats whole functions carrying the bracket
|
|
|
|
|
// shapes of the arm64 vector kernels and pins the output byte for byte. The
|
|
|
|
|
// brackets are load-bearing: formatting must not change what the file
|
|
|
|
|
// assembles to, so the formatted text keeps every bracket, re-formats to
|
|
|
|
|
// itself and still parses cleanly.
|
|
|
|
|
func TestSIMDBracketRoundTrip(t *testing.T) {
|
|
|
|
|
in := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"VDUP V31.B[15], R3\n" +
|
|
|
|
|
"VTBL V22.B16, [V28.B16], V11.B16\n" +
|
|
|
|
|
"VLD1 (R2), [V21.B16]\n" +
|
|
|
|
|
"VMOVQ $0x70, $0x80, V10\n" +
|
|
|
|
|
"RET\n"
|
|
|
|
|
|
|
|
|
|
want := "#include \"textflag.h\"\n" +
|
|
|
|
|
"\n" +
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT, $0\n" +
|
|
|
|
|
"\tVDUP V31.B[15], R3\n" +
|
|
|
|
|
"\tVTBL V22.B16, [V28.B16], V11.B16\n" +
|
|
|
|
|
"\tVLD1 (R2), [V21.B16]\n" +
|
|
|
|
|
"\tVMOVQ $0x70, $0x80, V10\n" +
|
|
|
|
|
"\tRET\n"
|
|
|
|
|
|
|
|
|
|
got := Source(in)
|
|
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
if again := Source(got); again != got {
|
|
|
|
|
t.Fatalf("not idempotent:\n%q", again)
|
|
|
|
|
}
|
|
|
|
|
if n := strings.Count(got, "["); n != 3 {
|
|
|
|
|
t.Errorf("output carries %d '[', want 3:\n%s", n, got)
|
|
|
|
|
}
|
|
|
|
|
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
|
|
|
|
|
t.Errorf("formatted output no longer parses: %v", errs)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestBracketFormsRoundTrip runs every bracket shape of the vector kernels
|
|
|
|
|
// through a full format pass as its own single-instruction function, where
|
|
|
|
|
// the canonical form is the line itself indented: formatting must be a no-op
|
|
|
|
|
// on each, so no bracket moves, vanishes or gains a space.
|
|
|
|
|
func TestBracketFormsRoundTrip(t *testing.T) {
|
|
|
|
|
for _, instr := range []string{
|
|
|
|
|
"VDUP V31.B[15], V18",
|
|
|
|
|
"VDUP V19.S[3], V18.S4",
|
|
|
|
|
"VDUP V1.D[1], V2.D2",
|
|
|
|
|
"VMOV V13.S[0], R20",
|
|
|
|
|
"VMOV V11.B[11], V16.B[12]",
|
|
|
|
|
"VMOV R20, V21.B[2]",
|
|
|
|
|
"VTBL V22.B16, [V28.B16], V11.B16",
|
|
|
|
|
"VTBL V18.B8, [V17.B16, V18.B16], V22.B8",
|
|
|
|
|
"VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8",
|
|
|
|
|
"VLD1 (R2), [V21.B16]",
|
|
|
|
|
"VLD1 (R24), [V18.D1, V19.D1, V20.D1]",
|
|
|
|
|
"VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]",
|
|
|
|
|
"VLD1.P 32(R1), [V2.B16, V3.B16]",
|
|
|
|
|
"VLD1R (R1), [V9.B8]",
|
|
|
|
|
"VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]",
|
|
|
|
|
"VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)",
|
|
|
|
|
"VST1.P [V2.B16], (R1)",
|
|
|
|
|
"VST1.P [V2.B16, V3.B16], 32(R1)",
|
|
|
|
|
"VMOVQ $0x70, $0x80, V10",
|
|
|
|
|
} {
|
|
|
|
|
src := "TEXT ·f(SB), NOSPLIT, $0\n" + instr + "\nRET\n"
|
|
|
|
|
want := "TEXT ·f(SB), NOSPLIT, $0\n\t" + instr + "\n\tRET\n"
|
|
|
|
|
got := Source(src)
|
|
|
|
|
if got != want {
|
|
|
|
|
t.Errorf("formatting %q:\n got %q\n want %q", instr, got, want)
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
if again := Source(got); again != got {
|
|
|
|
|
t.Errorf("not idempotent for %q:\n%q", instr, again)
|
|
|
|
|
}
|
|
|
|
|
if _, errs := parser.Parse("in.s", got); len(errs) > 0 {
|
|
|
|
|
t.Errorf("formatted output of %q no longer parses: %v", instr, errs)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-19 23:48:47 +02:00
|
|
|
// TestFlagListRoundTrip pins the '|' flag separator and the <ABIInternal>
|
|
|
|
|
// marker through a full format pass: the bars the Go toolchain requires and
|
|
|
|
|
// the ABI bracket must survive byte for byte, on TEXT and GLOBL alike.
|
|
|
|
|
func TestFlagListRoundTrip(t *testing.T) {
|
|
|
|
|
for _, in := range []string{
|
|
|
|
|
"TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0\n\tRET\n",
|
|
|
|
|
"TEXT ·foo<ABIInternal>(SB), NOSPLIT, $-0-24\n\tRET\n",
|
|
|
|
|
"GLOBL ·mask(SB), RODATA|NOPTR, $8\n",
|
|
|
|
|
} {
|
|
|
|
|
if got := Source(in); got != in {
|
|
|
|
|
t.Fatalf("flag list did not round-trip:\n--- got ---\n%q\n--- want ---\n%q", got, in)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestCRLFInputIsNormalisedToLF checks that a CRLF file comes out with
|
|
|
|
|
// uniform LF endings: a // comment must not carry its line's trailing \r
|
|
|
|
|
// into the output.
|
|
|
|
|
func TestCRLFInputIsNormalisedToLF(t *testing.T) {
|
|
|
|
|
in := "// func f()\r\nTEXT ·f(SB), NOSPLIT, $0\r\nRET\r\n"
|
|
|
|
|
want := "// func f()\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n"
|
|
|
|
|
got := Source(in)
|
|
|
|
|
if got != want {
|
|
|
|
|
t.Fatalf("CRLF formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want)
|
|
|
|
|
}
|
|
|
|
|
if strings.Contains(got, "\r") {
|
|
|
|
|
t.Fatalf("output still contains CR: %q", got)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-06 09:49:50 +02:00
|
|
|
// lexOperands lexes a single operand string and drops the EOF token.
|
|
|
|
|
func lexOperands(s string) []token.Token {
|
|
|
|
|
toks := lexer.Tokenize(s)
|
|
|
|
|
return toks[:len(toks)-1] // drop trailing EOF
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func TestIdempotent(t *testing.T) {
|
|
|
|
|
src, err := os.ReadFile("../testdata/sample_amd64.s")
|
|
|
|
|
if err != nil {
|
|
|
|
|
t.Fatal(err)
|
|
|
|
|
}
|
2026-08-29 17:12:53 +02:00
|
|
|
once := Source(string(src))
|
|
|
|
|
twice := Source(once)
|
2026-07-06 09:49:50 +02:00
|
|
|
if once != twice {
|
|
|
|
|
t.Fatal("formatting is not idempotent on the fixture")
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// TestRoundTrip checks that formatting produces source that still parses
|
2026-08-07 21:28:24 +02:00
|
|
|
// cleanly on the in-repository fixture.
|
2026-07-06 09:49:50 +02:00
|
|
|
func TestRoundTrip(t *testing.T) {
|
|
|
|
|
files := []string{"../testdata/sample_amd64.s"}
|
|
|
|
|
for _, path := range files {
|
|
|
|
|
src, err := os.ReadFile(path)
|
|
|
|
|
if err != nil {
|
|
|
|
|
t.Fatal(err)
|
|
|
|
|
}
|
2026-08-29 17:12:53 +02:00
|
|
|
formatted := Source(string(src))
|
2026-07-06 09:49:50 +02:00
|
|
|
if _, errs := parser.Parse(path, formatted); len(errs) > 0 {
|
|
|
|
|
t.Errorf("formatted %s no longer parses: %v", path, errs)
|
|
|
|
|
}
|
|
|
|
|
if strings.TrimSpace(formatted) == "" {
|
|
|
|
|
t.Errorf("formatted %s is empty", path)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|