From 75e9fd771ba8dffac8f9d594809cb71f74c3758f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 19:15:35 +0200 Subject: [PATCH] feat(parser): split plain statements on semicolons in the raw parse Assisted-by: GLM 5.3 Flash --- lexer/lexer_test.go | 13 ++++++ parser/parser_test.go | 96 ++++++++++++++++++++++++++++++++++++++++++ parser/preproc.go | 2 +- parser/preproc_test.go | 6 ++- 4 files changed, 114 insertions(+), 3 deletions(-) diff --git a/lexer/lexer_test.go b/lexer/lexer_test.go index abcffd2..db20ca4 100644 --- a/lexer/lexer_test.go +++ b/lexer/lexer_test.go @@ -182,6 +182,19 @@ func TestNulIsIllegal(t *testing.T) { eq(t, texts("MOVQ \x00 AX"), []string{"MOVQ", "\x00", "AX"}) } +func TestDivisionSlashInIdentifiers(t *testing.T) { + // U+2215 DIVISION SLASH is an identifier character, the way the + // toolchain's tokenizer treats it: the package path of a symbol is + // written with it (internal∕runtime∕atomic·Xchg) and must lex as one + // name. The ordinary slash (U+002F) stays punctuation. + eq(t, texts("CALL internal∕runtime∕atomic·Xchg(SB)"), + []string{"CALL", "internal∕runtime∕atomic·Xchg", "(", "SB", ")"}) + eq(t, texts("MOVQ sync∕atomic·Align(SB), AX"), + []string{"MOVQ", "sync∕atomic·Align", "(", "SB", ")", ",", "AX"}) + // It may also begin a name, like any letter of the toolchain's rule. + eq(t, kinds("∕x"), []token.Kind{token.Ident}) +} + // TestOffsetsAroundInvalidByte pins Position.Offset against the original // bytes: an invalid UTF-8 byte decodes to RuneError but advances the offset // table by exactly one byte, so every later position stays a true byte diff --git a/parser/parser_test.go b/parser/parser_test.go index 55d4d11..501239b 100644 --- a/parser/parser_test.go +++ b/parser/parser_test.go @@ -401,3 +401,99 @@ func TestInt64MinimumImmediate(t *testing.T) { t.Errorf("imm.Float = %q, want empty", imm.Float) } } + +// TestDivisionSlashPackagePath covers the runtime's package-path spelling: +// U+2215 DIVISION SLASH separates the elements of an import path inside a +// symbol (internal∕runtime∕atomic·Xchg), and the middle dot still separates +// the package from the name. The whole spelling must reach the symbol, not +// stop at the first slash. +func TestDivisionSlashPackagePath(t *testing.T) { + file, errs := Parse("t.s", "TEXT \u00b7f(SB), $0\n\tCALL internal∕runtime∕atomic·Xchg(SB)\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse errors: %v", errs) + } + txt := file.Decls[0].(*ast.Text) + instr := txt.Body[0].(*ast.Instr) + sym := instr.Operands[0].Addr.Sym + if sym == nil { + t.Fatal("operand carries no symbol") + } + if sym.Pkg != "internal∕runtime∕atomic" { + t.Errorf("pkg = %q, want internal∕runtime∕atomic", sym.Pkg) + } + if sym.Name != "Xchg" { + t.Errorf("name = %q, want Xchg", sym.Name) + } + if sym.Raw != "internal∕runtime∕atomic·Xchg(SB)" { + t.Errorf("raw = %q", sym.Raw) + } +} + +// TestSemicolonStatements covers the plain parse path: ';' separates +// statements on one line exactly as it does inside macro expansion, and a +// ';' inside a comment is comment text. +func TestSemicolonStatements(t *testing.T) { + file, errs := Parse("t.s", "TEXT \u00b7f(SB), $0\n\tROLQ $3, DI; ROLQ $13, DI\n\tMOVQ AX, BX // note; still comment\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse errors: %v", errs) + } + txt := file.Decls[0].(*ast.Text) + if len(txt.Body) != 4 { + t.Fatalf("body = %d statements, want 4", len(txt.Body)) + } + first := txt.Body[0].(*ast.Instr) + if first.Mnemonic.Text != "ROLQ" || len(first.Operands) != 2 { + t.Errorf("first statement = %+v, want ROLQ with two operands", first.Mnemonic) + } + second := txt.Body[1].(*ast.Instr) + if second.Mnemonic.Text != "ROLQ" || len(second.Operands) != 2 { + t.Errorf("second statement = %s, want ROLQ with two operands", second.Mnemonic.Text) + } + // The trailing comment belongs to the second MOVQ, semicolon included. + third := txt.Body[2].(*ast.Instr) + if third.Mnemonic.Text != "MOVQ" || third.Comment != "note; still comment" { + t.Errorf("third = %s, comment %q", third.Mnemonic.Text, third.Comment) + } +} + +// TestSemicolonAfterLabel covers a label sharing its line with two +// statements. +func TestSemicolonAfterLabel(t *testing.T) { + file, errs := Parse("t.s", "TEXT \u00b7f(SB), $0\nloop: NOP; NOP\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse errors: %v", errs) + } + txt := file.Decls[0].(*ast.Text) + if len(txt.Body) != 4 { + t.Fatalf("body = %d statements, want 4 (label, two instructions, RET)", len(txt.Body)) + } + if _, ok := txt.Body[0].(*ast.Label); !ok { + t.Errorf("first statement = %T, want *ast.Label", txt.Body[0]) + } + for i, want := range []string{"NOP", "NOP", "RET"} { + in, ok := txt.Body[i+1].(*ast.Instr) + if !ok || in.Mnemonic.Text != want { + t.Errorf("statement %d = %v, want %s", i+1, txt.Body[i+1], want) + } + } +} + +// TestParseEqualsZeroOptions pins the contract that ParseWithOptions with +// the zero Options reproduces Parse, here for the semicolon split. +func TestParseEqualsZeroOptions(t *testing.T) { + src := "TEXT \u00b7f(SB), $0\n\tNOP; NOP\n\tRET\n" + a, errsA := Parse("t.s", src) + b, errsB := ParseWithOptions("t.s", src, Options{}) + if len(errsA) > 0 || len(errsB) > 0 { + t.Fatalf("errors: %v / %v", errsA, errsB) + } + ta, tb := texts(a), texts(b) + if len(ta) != len(tb) { + t.Fatalf("decl counts differ: %d vs %d", len(ta), len(tb)) + } + for i := range ta { + if len(ta[i].Body) != len(tb[i].Body) { + t.Fatalf("TEXT %d: body lengths differ: %d vs %d", i, len(ta[i].Body), len(tb[i].Body)) + } + } +} diff --git a/parser/preproc.go b/parser/preproc.go index 721b45e..b99be02 100644 --- a/parser/preproc.go +++ b/parser/preproc.go @@ -47,7 +47,7 @@ func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) { lines = pp.fileLines(path, tokens, token.Position{}) errs = pp.errs } else { - lines = splitLines(tokens) + lines = statementLines(tokens) } p := &state{path: path} p.parse(lines) diff --git a/parser/preproc_test.go b/parser/preproc_test.go index 3dff224..6ff01ae 100644 --- a/parser/preproc_test.go +++ b/parser/preproc_test.go @@ -462,7 +462,9 @@ TEXT ·f(SB), NOSPLIT, $0 func TestParseUnchangedWithoutExpand(t *testing.T) { // Without Expand the preprocessor must not exist: a macro invocation - // stays an unexpanded instruction line and ';' keeps the old parse. + // stays an unexpanded instruction line. The ';' statement separator is + // not part of the preprocessor: the plain parse path splits on it the + // same way the expansion path does, so both spellings agree. f, errs := Parse("t_amd64.s", ` #define TWICE ADDQ AX, AX TEXT ·f(SB), NOSPLIT, $0 @@ -480,7 +482,7 @@ TEXT ·f(SB), NOSPLIT, $0 mnemonics = append(mnemonics, in.Mnemonic.Text) } } - if strings.Join(mnemonics, " ") != "TWICE BYTE RET" { + if strings.Join(mnemonics, " ") != "TWICE BYTE BYTE RET" { t.Errorf("non-expanding parse changed: %v", mnemonics) } }