diff --git a/ast/ast.go b/ast/ast.go index e0f5718..2861f2c 100644 --- a/ast/ast.go +++ b/ast/ast.go @@ -114,6 +114,7 @@ type Symbol struct { Pkg string // package prefix before the middle dot ("" = current package) Name string // identifier without the middle dot or <> Static bool // the <> marker is present + ABI string // the ABI marker, e.g. ABIInternal ("" when absent) Pseudo string // FP, SP, SB or PC ("" for a bare name) Offset int64 HasOff bool @@ -156,5 +157,5 @@ type Address struct { Scale int // index scale; 0 when absent Offset int64 // leading displacement, from off(base) HasOff bool // a leading displacement is present - Shift string // verbatim arm64 shift suffix, e.g. "<<2" + Shift string // verbatim arm64 shift suffix, e.g. "<< 2" } diff --git a/parser/parser.go b/parser/parser.go index d5751f2..6ec3bc3 100644 --- a/parser/parser.go +++ b/parser/parser.go @@ -9,6 +9,7 @@ package parser import ( "fmt" + "math" "strconv" "strings" @@ -75,7 +76,6 @@ func splitLines(tokens []token.Token) [][]token.Token { func (p *state) parse(lines [][]token.Token) { p.file = &ast.File{Path: p.path, Macros: map[string]bool{}} for _, line := range lines { - line = trimSpace(line) if len(line) == 0 { // Blank line: a comment block ends here only if it was not // directly preceding a declaration; keep pending doc intact @@ -87,10 +87,6 @@ func (p *state) parse(lines [][]token.Token) { } } -// trimSpace is a no-op placeholder kept for symmetry; the lexer already drops -// horizontal whitespace, but this documents the intent. -func trimSpace(line []token.Token) []token.Token { return line } - func (p *state) parseLine(line []token.Token) { first := line[0] @@ -196,6 +192,10 @@ func (p *state) parseText(line []token.Token) { sym, n := parseSymbolPrefix(rest) if sym == nil { p.errorf(line[0].Pos, "TEXT missing a symbol name") + // Keep the decl in the tree with a placeholder name: the linter and + // LSP dereference Name on every parsed TEXT, so the file must stay + // usable alongside its errors. + sym = &ast.Symbol{Pos: line[0].Pos, Raw: "?", Name: "?"} } text.Name = sym rest = rest[n:] @@ -206,7 +206,7 @@ func (p *state) parseText(line []token.Token) { if rest[0].Kind == token.Ident { text.Flags = append(text.Flags, rest[0].Text) } - // Commas, '|' (Illegal) and anything else between flags is skipped. + // Commas, '|' and anything else between flags is skipped. rest = rest[1:] } @@ -307,7 +307,9 @@ var pseudoRegs = map[string]bool{"FP": true, "SP": true, "SB": true, "PC": true} // parseSymbolPrefix parses a leading symbol reference from g and returns it // together with the number of tokens consumed. It returns (nil, 0) when no -// symbol is present. +// symbol is present. Raw is the verbatim spelling, with the bracket group, +// offset and pseudo-register glued to the name the way the assembler writes +// the reference. func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) { if len(g) == 0 || g[0].Kind != token.Ident { return nil, 0 @@ -317,27 +319,44 @@ func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) { setName(g[0].Text, sym) i++ - if i+1 < len(g) && g[i].Kind == token.LAngle && g[i+1].Kind == token.RAngle { - sym.Static = true - i += 2 + var raw strings.Builder + raw.WriteString(g[0].Text) + // An optional bracket group after the name: <> marks a file-static symbol + // and selects the ABI of the reference. The ABI form is the + // standard runtime spelling (TEXT ·foo(SB)) and must be + // consumed here, or it leaks into the directive's flag list. + if i < len(g) && g[i].Kind == token.LAngle { + switch { + case i+1 < len(g) && g[i+1].Kind == token.RAngle: + sym.Static = true + raw.WriteString("<>") + i += 2 + case i+2 < len(g) && g[i+1].Kind == token.Ident && g[i+2].Kind == token.RAngle: + sym.ABI = g[i+1].Text + raw.WriteString("<" + sym.ABI + ">") + i += 3 + } } if i < len(g) && (g[i].Kind == token.Plus || g[i].Kind == token.Minus) { neg := g[i].Kind == token.Minus + raw.WriteString(g[i].Text) i++ if i < len(g) && g[i].Kind == token.Number { sym.Offset, sym.HasOff = parseInt(g[i].Text), true if neg { sym.Offset = -sym.Offset } + raw.WriteString(g[i].Text) i++ } } if i+2 < len(g) && g[i].Kind == token.LParen && g[i+1].Kind == token.Ident && pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen { sym.Pseudo = g[i+1].Text + raw.WriteString("(" + sym.Pseudo + ")") i += 3 } - sym.Raw = joinRaw(g[:i]) + sym.Raw = raw.String() return sym, i } @@ -384,9 +403,8 @@ func parseImmediate(g []token.Token) ast.Immediate { } // $sym(…) form. if findPseudoParen(g) >= 0 || (g[0].Kind == token.Ident) { - if sym, n := parseSymbolPrefix(g); sym != nil && (sym.Pseudo != "" || sym.Static) { + if sym, n := parseSymbolPrefix(g); n > 0 && (sym.Pseudo != "" || sym.Static) { imm.Sym = sym - _ = n return imm } } @@ -402,10 +420,16 @@ func parseImmediate(g []token.Token) ast.Immediate { if v, ok := tryInt(text); ok { imm.Val = v imm.HasVal = true - } else if u, err := strconv.ParseUint(text, 0, 64); err == nil && !imm.Neg { - // Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…) - // overflow int64; keep the bit pattern. - imm.Val = int64(u) + } else if u, err := strconv.ParseUint(text, 0, 64); err == nil && (!imm.Neg || u == 1<<63) { + // Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…) can + // overflow int64; keep the bit pattern. A negated magnitude of + // exactly 1<<63 is the int64 minimum: ParseInt rejects it, but + // the value is representable, so Val carries it exactly. + if imm.Neg { + imm.Val = math.MinInt64 + } else { + imm.Val = int64(u) + } imm.HasVal = true } else { imm.Float = text diff --git a/parser/parser_test.go b/parser/parser_test.go index d29ab89..55d4d11 100644 --- a/parser/parser_test.go +++ b/parser/parser_test.go @@ -4,7 +4,10 @@ package parser import ( + "math" "os" + "slices" + "strings" "testing" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -315,4 +318,86 @@ func TestSignedZeroFrame(t *testing.T) { if txt.Args == nil || !txt.Args.Imm.HasVal || txt.Args.Imm.Val != 24 { t.Errorf("args = %+v, want -24", txt.Args) } + // The marker belongs to the symbol: the pseudo-register is + // consumed, the marker is recorded, and neither leaks into the flags. + if txt.Name.Pseudo != "SB" { + t.Errorf("pseudo = %q, want SB", txt.Name.Pseudo) + } + if txt.Name.ABI != "ABIInternal" { + t.Errorf("ABI = %q, want ABIInternal", txt.Name.ABI) + } + if !strings.Contains(txt.Name.Raw, "") { + t.Errorf("Raw = %q, want it to contain ", txt.Name.Raw) + } + if want := []string{"NOSPLIT"}; !slices.Equal(txt.Flags, want) { + t.Errorf("flags = %v, want %v", txt.Flags, want) + } +} + +// TestPipedFlags covers TEXT and GLOBL flag lists joined by '|': the bars are +// their own token kind, skipped by the flag loop, and only the identifiers +// are collected as flags. +func TestPipedFlags(t *testing.T) { + file, errs := Parse("t.s", "TEXT \u00b7f(SB), NOSPLIT|NOFRAME|DUPOK, $0\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse errors: %v", errs) + } + txt := file.Decls[0].(*ast.Text) + if want := []string{"NOSPLIT", "NOFRAME", "DUPOK"}; !slices.Equal(txt.Flags, want) { + t.Errorf("flags = %v, want %v", txt.Flags, want) + } + + g, errs := Parse("t.s", "GLOBL \u00b7mask(SB), RODATA|NOPTR, $8\n") + if len(errs) > 0 { + t.Fatalf("parse errors: %v", errs) + } + gl := g.Decls[0].(*ast.Globl) + if want := []string{"RODATA", "NOPTR"}; !slices.Equal(gl.Flags, want) { + t.Errorf("flags = %v, want %v", gl.Flags, want) + } +} + +// TestTextMissingSymbolKeepsDecl covers a TEXT with no symbol at all: the +// decl must stay in the tree with a non-nil placeholder name, because the +// linter and the LSP dereference Name on every parsed TEXT. +func TestTextMissingSymbolKeepsDecl(t *testing.T) { + file, errs := Parse("t.s", "// func f(a int) int\nTEXT $0\n\tMOVQ AX, BX\n") + if len(errs) == 0 { + t.Fatal("expected a diagnostic for the missing symbol") + } + if file == nil || len(file.Decls) != 1 { + t.Fatalf("file = %v, want the TEXT decl kept", file) + } + txt := file.Decls[0].(*ast.Text) + if txt.Name == nil { + t.Fatal("Name must never be nil: downstream tools dereference it") + } + if txt.Name.Name == "" { + t.Error("placeholder name is empty") + } + if txt.Frame == nil || !txt.Frame.Imm.HasVal || txt.Frame.Imm.Val != 0 { + t.Errorf("frame = %+v, want $0", txt.Frame) + } + if len(txt.Body) != 1 { + t.Errorf("body = %d statements, want 1", len(txt.Body)) + } +} + +// TestInt64MinimumImmediate covers $-0x8000000000000000: the digits overflow +// int64 when parsed directly, but the negated magnitude is exactly the int64 +// minimum and must land in Val rather than the float fallback. +func TestInt64MinimumImmediate(t *testing.T) { + file, errs := Parse("t.s", "TEXT \u00b7f(SB), $0\n\tMOVQ $-0x8000000000000000, AX\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse errors: %v", errs) + } + txt := file.Decls[0].(*ast.Text) + instr := txt.Body[0].(*ast.Instr) + imm := instr.Operands[0].Imm + if !imm.HasVal || imm.Val != math.MinInt64 { + t.Errorf("imm = %+v, want Val = %d with HasVal set", imm, math.MinInt64) + } + if imm.Float != "" { + t.Errorf("imm.Float = %q, want empty", imm.Float) + } }