// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package format import ( "os" "slices" "strings" "testing" "sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/lexer" "sourcedock.dev/petrbalvin/gasm-devkit/parser" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) func TestGolden(t *testing.T) { in := "#include \"textflag.h\"\n" + "\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "MOVQ swin_base+0(FP), SI\n" + "LEAQ (SI)(BX*4), R9\n" + "ANDQ $-8, R10\n" + "VFMADD231PD Z14, Z12, Z10\n" + "RET\n" want := "#include \"textflag.h\"\n" + "\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "\tMOVQ swin_base+0(FP), SI\n" + "\tLEAQ (SI)(BX*4), R9\n" + "\tANDQ $-8, R10\n" + "\tVFMADD231PD Z14, Z12, Z10\n" + "\tRET\n" got := Source(in) if got != want { t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want) } } // TestDocCommentIndent checks that a doc comment preceding a TEXT directive // sits at column 0 even when another function (ending in RET) precedes it; // the RET must terminate the previous body for indentation purposes. func TestDocCommentIndent(t *testing.T) { in := "#include \"textflag.h\"\n" + "\n" + "// func first()\n" + "TEXT ·first(SB), NOSPLIT, $0\n" + "XORQ AX, AX\n" + "RET\n" + "\n" + "// func second()\n" + "TEXT ·second(SB), NOSPLIT, $0\n" + "RET\n" want := "#include \"textflag.h\"\n" + "\n" + "// func first()\n" + "TEXT ·first(SB), NOSPLIT, $0\n" + "\tXORQ AX, AX\n" + "\tRET\n" + "\n" + "// func second()\n" + "TEXT ·second(SB), NOSPLIT, $0\n" + "\tRET\n" got := Source(in) if got != want { t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want) } // Body comments stay indented. body := "#include \"textflag.h\"\nTEXT ·f(SB), NOSPLIT, $0\n// inside the body\nXORQ AX, AX\nRET\n" gotBody := Source(body) if !strings.Contains(gotBody, "\t// inside the body\n") { t.Fatalf("body comment must stay indented:\n%q", gotBody) } } // TestBlankLines checks the blank-line canonicalisation: exactly one blank // line before a new block (a label, or TEXT/GLOBL), runs of blanks collapsed // to one, and no blank forced after TEXT, between stacked labels, or at the // top of the file. Leading comments belong to the block they precede. func TestBlankLines(t *testing.T) { in := "#include \"textflag.h\"\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "first:\n" + // first label: no blank after TEXT "XORQ AX, AX\n" + "JMP next\n" + // unlabelled glue: fmt inserts a blank before next: "next:\n" + "stacked:\n" + // stacked labels share an address: no blank between "INCQ AX\n" + "\n" + "\n" + // two blanks collapse to one "// separated block\n" + // comment belongs to the label below "later:\n" + "RET\n" + "// func g()\n" + // doc comment: blank goes before it "TEXT ·g(SB), NOSPLIT, $0\n" + "RET\n" + "GLOBL ·mask(SB), RODATA, $8\n" + // blank before GLOBL… "DATA ·mask+0(SB)/4, $1\n" + // …but not before DATA "\n" + "\n" + "\n" // trailing blanks dropped want := "#include \"textflag.h\"\n" + "\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "first:\n" + "\tXORQ AX, AX\n" + "\tJMP next\n" + "\n" + "next:\n" + "stacked:\n" + "\tINCQ AX\n" + "\n" + "\t// separated block\n" + // body comment before a label stays indented "later:\n" + "\tRET\n" + "\n" + "// func g()\n" + "TEXT ·g(SB), NOSPLIT, $0\n" + "\tRET\n" + "\n" + "GLOBL ·mask(SB), RODATA, $8\n" + "DATA ·mask+0(SB)/4, $1\n" got := Source(in) if got != want { t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want) } if again := Source(got); again != got { t.Fatalf("not idempotent:\n%q", again) } } func TestOperandSpacing(t *testing.T) { cases := map[string]string{ "4(SI)": "4(SI)", "(SI)(BX*4)": "(SI)(BX*4)", "$-8": "$-8", "$0x80020100": "$0x80020100", "swin_base+0(FP)": "swin_base+0(FP)", "mask24<>(SB)": "mask24<>(SB)", "·idx16+0(SB)/4": "·idx16+0(SB)/4", "NOSPLIT|DUPOK": "NOSPLIT|DUPOK", } for in, want := range cases { toks := lexOperands(in) if got := renderOps(toks); got != want { t.Errorf("renderOps(%q) = %q, want %q", in, got, want) } } } // TestVectorBracketSpacing pins the square-bracket operand forms of the // arm64 and loong64 vector syntaxes. The lexer emits '[' and ']' as Illegal // tokens carrying their spelling, and renderOps must glue them back exactly // where they were: a register list and an element selector are load-bearing // operands the assembler reads out of the operand text, so no bracket may be // dropped, and the canonical spelling inside the brackets is tight. func TestVectorBracketSpacing(t *testing.T) { cases := map[string]string{ // Register lists of one to four registers. "[V21.B16]": "[V21.B16]", "[V17.B16, V18.B16]": "[V17.B16, V18.B16]", "[V18.D1, V19.D1, V20.D1]": "[V18.D1, V19.D1, V20.D1]", "[V14.B16, V15.B16, V16.B16, V17.B16]": "[V14.B16, V15.B16, V16.B16, V17.B16]", // Element selectors. "V31.B[15]": "V31.B[15]", "V19.S[0]": "V19.S[0]", "V1.D[1]": "V1.D[1]", "V11.B[11], V16.B[12]": "V11.B[11], V16.B[12]", // Lists beside address operands, on either side. "32(R1), [V2.B16, V3.B16]": "32(R1), [V2.B16, V3.B16]", "[V2.S4, V3.S4], (R14)": "[V2.S4, V3.S4], (R14)", "(R24), [V18.D1, V19.D1]": "(R24), [V18.D1, V19.D1]", // A spaced spelling canonicalises to the tight one. "[ V21.B16 ]": "[V21.B16]", "V31.B [15]": "V31.B[15]", } for in, want := range cases { toks := lexOperands(in) if got := renderOps(toks); got != want { t.Errorf("renderOps(%q) = %q, want %q", in, got, want) } } } // TestSIMDBracketRoundTrip formats whole functions carrying the bracket // shapes of the arm64 vector kernels and pins the output byte for byte. The // brackets are load-bearing: formatting must not change what the file // assembles to, so the formatted text keeps every bracket, re-formats to // itself and still parses cleanly. func TestSIMDBracketRoundTrip(t *testing.T) { in := "#include \"textflag.h\"\n" + "\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "VDUP V31.B[15], R3\n" + "VTBL V22.B16, [V28.B16], V11.B16\n" + "VLD1 (R2), [V21.B16]\n" + "VMOVQ $0x70, $0x80, V10\n" + "RET\n" want := "#include \"textflag.h\"\n" + "\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "\tVDUP V31.B[15], R3\n" + "\tVTBL V22.B16, [V28.B16], V11.B16\n" + "\tVLD1 (R2), [V21.B16]\n" + "\tVMOVQ $0x70, $0x80, V10\n" + "\tRET\n" got := Source(in) if got != want { t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want) } if again := Source(got); again != got { t.Fatalf("not idempotent:\n%q", again) } if n := strings.Count(got, "["); n != 3 { t.Errorf("output carries %d '[', want 3:\n%s", n, got) } if _, errs := parser.Parse("in.s", got); len(errs) > 0 { t.Errorf("formatted output no longer parses: %v", errs) } } // TestBracketFormsRoundTrip runs every bracket shape of the vector kernels // through a full format pass as its own single-instruction function, where // the canonical form is the line itself indented: formatting must be a no-op // on each, so no bracket moves, vanishes or gains a space. func TestBracketFormsRoundTrip(t *testing.T) { for _, instr := range []string{ "VDUP V31.B[15], V18", "VDUP V19.S[3], V18.S4", "VDUP V1.D[1], V2.D2", "VMOV V13.S[0], R20", "VMOV V11.B[11], V16.B[12]", "VMOV R20, V21.B[2]", "VTBL V22.B16, [V28.B16], V11.B16", "VTBL V18.B8, [V17.B16, V18.B16], V22.B8", "VTBL V31.B8, [V14.B16, V15.B16, V16.B16, V17.B16], V15.B8", "VLD1 (R2), [V21.B16]", "VLD1 (R24), [V18.D1, V19.D1, V20.D1]", "VLD1 (R29), [V14.D1, V15.D1, V16.D1, V17.D1]", "VLD1.P 32(R1), [V2.B16, V3.B16]", "VLD1R (R1), [V9.B8]", "VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]", "VST1 [V2.S4, V3.S4, V4.S4, V5.S4], (R14)", "VST1.P [V2.B16], (R1)", "VST1.P [V2.B16, V3.B16], 32(R1)", "VMOVQ $0x70, $0x80, V10", } { src := "TEXT ·f(SB), NOSPLIT, $0\n" + instr + "\nRET\n" want := "TEXT ·f(SB), NOSPLIT, $0\n\t" + instr + "\n\tRET\n" got := Source(src) if got != want { t.Errorf("formatting %q:\n got %q\n want %q", instr, got, want) continue } if again := Source(got); again != got { t.Errorf("not idempotent for %q:\n%q", instr, again) } if _, errs := parser.Parse("in.s", got); len(errs) > 0 { t.Errorf("formatted output of %q no longer parses: %v", instr, errs) } } } // TestFlagListRoundTrip pins the '|' flag separator and the // marker through a full format pass: the bars the Go toolchain requires and // the ABI bracket must survive byte for byte, on TEXT and GLOBL alike. func TestFlagListRoundTrip(t *testing.T) { for _, in := range []string{ "TEXT ·f(SB), NOSPLIT|NOFRAME|DUPOK, $0\n\tRET\n", "TEXT ·foo(SB), NOSPLIT, $-0-24\n\tRET\n", "GLOBL ·mask(SB), RODATA|NOPTR, $8\n", } { if got := Source(in); got != in { t.Fatalf("flag list did not round-trip:\n--- got ---\n%q\n--- want ---\n%q", got, in) } } } // TestCRLFInputIsNormalisedToLF checks that a CRLF file comes out with // uniform LF endings: a // comment must not carry its line's trailing \r // into the output. func TestCRLFInputIsNormalisedToLF(t *testing.T) { in := "// func f()\r\nTEXT ·f(SB), NOSPLIT, $0\r\nRET\r\n" want := "// func f()\nTEXT ·f(SB), NOSPLIT, $0\n\tRET\n" got := Source(in) if got != want { t.Fatalf("CRLF formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, want) } if strings.Contains(got, "\r") { t.Fatalf("output still contains CR: %q", got) } } // TestSemicolonSeparators pins the treatment of ';' statement separators. // The separator is load-bearing on the assembly path, where the parser reads // semicolon-separated statements: a formatter that drops it fuses two // statements into a line the assembler rejects, which is data corruption. // Each row pins the canonical spelling, one space after the ';', tight // before it, the way the toolchain's own sources and listings write it. func TestSemicolonSeparators(t *testing.T) { cases := []struct { name string in string want string }{ { name: "between instructions, tight", in: "TEXT ·f(SB), $0\nBYTE $0x48;BYTE $0xc7\nRET\n", want: "TEXT ·f(SB), $0\n\tBYTE $0x48; BYTE $0xc7\n\tRET\n", }, { name: "between instructions, spaced", in: "TEXT ·f(SB), $0\nBYTE $0x48 ; BYTE $0xc7\nRET\n", want: "TEXT ·f(SB), $0\n\tBYTE $0x48; BYTE $0xc7\n\tRET\n", }, { name: "after a label", in: "TEXT ·f(SB), $0\nlabel: BYTE $1; BYTE $2\nRET\n", want: "TEXT ·f(SB), $0\nlabel:\n\tBYTE $1; BYTE $2\n\tRET\n", }, { // The continuation-spliced macro shape of the runtime sources: // the lexer makes one logical line of the backslash continuations. name: "inside a macro body, continued", in: "#define MOVLTOREG(v, off) \\\n\tMOVL $v, AX; \\\n\tMOVL AX, ret+off(FP)\n", want: "#define MOVLTOREG(v, off) MOVL $v, AX; MOVL AX, ret+off(FP)\n", }, { name: "inside a macro body, one line", in: "#define PEAS BYTE $0x0a; BYTE $0x0b\n", want: "#define PEAS BYTE $0x0a; BYTE $0x0b\n", }, { name: "several separators in one line", in: "TEXT ·f(SB), $0\nBYTE $1; BYTE $2; BYTE $3\nRET\n", want: "TEXT ·f(SB), $0\n\tBYTE $1; BYTE $2; BYTE $3\n\tRET\n", }, { name: "two separators back to back", in: "TEXT ·f(SB), $0\nBYTE $1;; BYTE $2\nRET\n", want: "TEXT ·f(SB), $0\n\tBYTE $1;; BYTE $2\n\tRET\n", }, { name: "inside a line comment, untouched", in: "TEXT ·f(SB), $0\n// keep; the; separators\nBYTE $1\nRET\n", want: "TEXT ·f(SB), $0\n\t// keep; the; separators\n\tBYTE $1\n\tRET\n", }, { name: "after a statement, before a comment", in: "TEXT ·f(SB), $0\nMOVQ AX, BX; // tail\nRET\n", want: "TEXT ·f(SB), $0\n\tMOVQ AX, BX; // tail\n\tRET\n", }, { name: "last character on a line", in: "TEXT ·f(SB), $0\nBYTE $1;\nRET\n", want: "TEXT ·f(SB), $0\n\tBYTE $1;\n\tRET\n", }, { // The REP shape: a prefix-style zero-operand statement // followed by the instruction it prefixes. The separator // belongs to the statement it ends, so the function's // alignment width (MOVSQ is the widest mnemonic here) must // not open a gap before it: one space after the mnemonic // whatever the neighbours' lengths. name: "after a prefix-style statement", in: "TEXT ·f(SB), $0\nMOVQ AX, BX\nREP; MOVSQ\nRET\n", want: "TEXT ·f(SB), $0\n\tMOVQ AX, BX\n\tREP ; MOVSQ\n\tRET\n", }, } for _, tc := range cases { t.Run(tc.name, func(t *testing.T) { got := Source(tc.in) if got != tc.want { t.Fatalf("formatting mismatch:\n--- got ---\n%q\n--- want ---\n%q", got, tc.want) } if again := Source(got); again != got { t.Fatalf("not idempotent:\n%q", again) } if in, out := strings.Count(tc.in, ";"), strings.Count(got, ";"); in != out { t.Fatalf("semicolon count changed: %d -> %d\n%s", in, out, got) } if _, errs := parser.Parse("in.s", got); len(errs) > 0 { t.Fatalf("formatted output no longer parses: %v", errs) } }) } } // TestSemicolonStatementRoundTrip proves the formatter's contract on the // path where ';' separates statements: parse the source the way the // assembler does, format it, re-parse the formatted text and compare the // statement sequence. Raw operand texts are token-joined, so they are // insensitive to the whitespace a format pass chooses, and the comparison // can only fail when a token is lost: dropping a ';' fuses two statements // into one, exactly the corruption the released formatter committed. func TestSemicolonStatementRoundTrip(t *testing.T) { src := "#define MOVLTOREG(v, off) \\\n" + "\tMOVL $v, AX; \\\n" + "\tMOVL AX, ret+off(FP)\n" + "\n" + "TEXT ·f(SB), NOSPLIT, $0\n" + "BYTE $0x48; BYTE $0xc7\n" + "first: BYTE $1; BYTE $2\n" + "MOVLTOREG($42, 0)\n" + "RET\n" before, errs := parser.ParseWithOptions("in.s", src, parser.Options{Expand: true}) if len(errs) > 0 { t.Fatalf("source does not parse: %v", errs) } formatted := Source(src) after, errs := parser.ParseWithOptions("in.s", formatted, parser.Options{Expand: true}) if len(errs) > 0 { t.Fatalf("formatted source does not parse: %v", errs) } want, got := stmtSignature(before), stmtSignature(after) if !slices.Equal(got, want) { t.Fatalf("statement sequence changed:\n--- before ---\n%q\n--- after ---\n%q", want, got) } if again := Source(formatted); again != formatted { t.Fatalf("not idempotent:\n%q", again) } // The two BYTE statements on the first line must stay two: one fused // statement here is the exact defect this package once shipped. var bytes []string for _, stmt := range stmtSignature(after) { if rest, ok := strings.CutPrefix(stmt, "instr BYTE "); ok { bytes = append(bytes, rest) } } if want := []string{"$ 0x48", "$ 0xc7", "$ 1", "$ 2"}; !slices.Equal(bytes, want) { t.Fatalf("BYTE statements after expansion = %q, want %q", bytes, want) } } // stmtSignature flattens a parsed file into one string per declaration and // statement, in source order. Every component is token-derived, so the // signature is stable across format passes and moves only when a token is // lost or gained. func stmtSignature(f *ast.File) []string { var out []string for _, d := range f.Decls { switch d := d.(type) { case *ast.Text: out = append(out, "text "+d.Name.Raw) for _, s := range d.Body { out = append(out, stmtText(s)) } case *ast.Globl: out = append(out, "globl "+d.Name.Raw) case *ast.Data: out = append(out, "data "+d.Name.Raw) case *ast.Include: out = append(out, "include "+d.Header.Text) case *ast.Preproc: out = append(out, "preproc "+d.Raw) } } for _, s := range f.Orphans { out = append(out, stmtText(s)) } return out } // stmtText renders one statement for stmtSignature. func stmtText(s ast.Stmt) string { switch s := s.(type) { case *ast.Label: return "label " + s.Name.Text case *ast.Instr: parts := make([]string, 0, len(s.Operands)+1) parts = append(parts, s.Mnemonic.Text) for _, op := range s.Operands { parts = append(parts, op.Raw) } return "instr " + strings.Join(parts, " ") default: return "stmt" } } // lexOperands lexes a single operand string and drops the EOF token. func lexOperands(s string) []token.Token { toks := lexer.Tokenize(s) return toks[:len(toks)-1] // drop trailing EOF } func TestIdempotent(t *testing.T) { src, err := os.ReadFile("../testdata/sample_amd64.s") if err != nil { t.Fatal(err) } once := Source(string(src)) twice := Source(once) if once != twice { t.Fatal("formatting is not idempotent on the fixture") } } // TestRoundTrip checks that formatting produces source that still parses // cleanly on the in-repository fixture. func TestRoundTrip(t *testing.T) { files := []string{"../testdata/sample_amd64.s"} for _, path := range files { src, err := os.ReadFile(path) if err != nil { t.Fatal(err) } formatted := Source(string(src)) if _, errs := parser.Parse(path, formatted); len(errs) > 0 { t.Errorf("formatted %s no longer parses: %v", path, errs) } if strings.TrimSpace(formatted) == "" { t.Errorf("formatted %s is empty", path) } } }