diff --git a/parser/flags.go b/parser/flags.go index 23c85c6..ee118e5 100644 --- a/parser/flags.go +++ b/parser/flags.go @@ -62,23 +62,14 @@ var flagOrder = []struct { {"ABIWRAPPER", 4096}, } -// flagsRun returns the flags operand of a TEXT or GLOBL directive: the -// tokens after the symbol up to the frame operand's '$', with comments and -// the one trailing comma that separated the operand from the '$' removed. +// flagsRun prepares one flags operand for evaluation: the comments are +// dropped, they are layout the expression never sees. func flagsRun(g []token.Token) []token.Token { - n := 0 - for n < len(g) && g[n].Kind != token.Dollar { - n++ - } - out := make([]token.Token, 0, n) - for _, t := range g[:n] { - if t.Kind == token.Comment { - continue + out := make([]token.Token, 0, len(g)) + for _, t := range g { + if t.Kind != token.Comment { + out = append(out, t) } - out = append(out, t) - } - if len(out) > 0 && out[len(out)-1].Kind == token.Comma { - out = out[:len(out)-1] } return out } diff --git a/parser/flags_test.go b/parser/flags_test.go new file mode 100644 index 0000000..85757b6 --- /dev/null +++ b/parser/flags_test.go @@ -0,0 +1,225 @@ +// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package parser + +import ( + "slices" + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-sdk/ast" +) + +// parseTextHeader parses the one TEXT directive of src, raw (no expansion). +func parseTextHeader(t *testing.T, header string) *ast.Text { + t.Helper() + f, errs := Parse("t.s", header+"\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse %q: %v", header, errs) + } + return f.Decls[0].(*ast.Text) +} + +// parseTextHeaderExpanded parses the one TEXT directive of src on the +// assembly path, where the toolchain's rejections apply. +func parseTextHeaderExpanded(t *testing.T, header string) (*ast.Text, []error) { + t.Helper() + f, errs := ParseWithOptions("t.s", header+"\n\tRET\n", Options{Expand: true}) + if len(f.Decls) != 1 { + t.Fatalf("parse %q: %d decls, want 1", header, len(f.Decls)) + } + return f.Decls[0].(*ast.Text), errs +} + +// TestTextFlagsExpression covers the flags operand as one expression: names +// joined by '|', the legacy numeric spellings, and the arithmetic forms the +// toolchain's evalInteger folds. Names written purely keep their written +// order and duplicates; an operand with literals in it expands to the +// canonical ascending name list. +func TestTextFlagsExpression(t *testing.T) { + for _, c := range []struct { + operand string + wantFlag []string + wantVal int64 + }{ + {"NOSPLIT", []string{"NOSPLIT"}, 4}, + {"NOSPLIT|NOFRAME|DUPOK", []string{"NOSPLIT", "NOFRAME", "DUPOK"}, 518}, + {"TOPFRAME|NOSPLIT", []string{"TOPFRAME", "NOSPLIT"}, 2052}, + {"NOSPLIT|NOSPLIT", []string{"NOSPLIT", "NOSPLIT"}, 4}, + {"4", []string{"NOSPLIT"}, 4}, + {"512", []string{"NOFRAME"}, 512}, + {"2|4", []string{"DUPOK", "NOSPLIT"}, 6}, + {"(NOSPLIT|NOFRAME)", []string{"NOSPLIT", "NOFRAME"}, 516}, + {"NOSPLIT|4", []string{"NOSPLIT"}, 4}, + {"NOSPLIT&DUPOK", nil, 0}, + {"$NOSPLIT", []string{"NOSPLIT"}, 4}, + } { + txt := parseTextHeader(t, "TEXT \u00b7f(SB), "+c.operand+", $0") + if !slices.Equal(txt.Flags, c.wantFlag) { + t.Errorf("%s: flags = %v, want %v", c.operand, txt.Flags, c.wantFlag) + } + if txt.FlagVal != c.wantVal { + t.Errorf("%s: FlagVal = %d, want %d", c.operand, txt.FlagVal, c.wantVal) + } + if txt.Frame == nil || !txt.Frame.Imm.HasVal || txt.Frame.Imm.Val != 0 { + t.Errorf("%s: frame = %+v, want $0", c.operand, txt.Frame) + } + } + + // No flags operand at all: the toolchain's zero. + txt := parseTextHeader(t, "TEXT \u00b7f(SB), $0") + if txt.Flags != nil || txt.FlagVal != 0 { + t.Errorf("no flags: flags = %v val = %d, want nil 0", txt.Flags, txt.FlagVal) + } +} + +// TestTextFlagValues pins the flag table against the constants +// cmd/internal/obj/textflag.go defines and the toolchain's textflag.h +// ships; the assembler, linker and compiler must all agree on them. +func TestTextFlagValues(t *testing.T) { + for _, c := range []struct { + name string + val int64 + }{ + {"NOPROF", 1}, {"DUPOK", 2}, {"NOSPLIT", 4}, {"RODATA", 8}, + {"NOPTR", 16}, {"WRAPPER", 32}, {"NEEDCTXT", 64}, {"TLSBSS", 256}, + {"NOFRAME", 512}, {"REFLECTMETHOD", 1024}, {"TOPFRAME", 2048}, + {"ABIWRAPPER", 4096}, + } { + txt := parseTextHeader(t, "TEXT \u00b7f(SB), "+c.name+", $0") + if txt.FlagVal != c.val { + t.Errorf("%s = %d, want %d", c.name, txt.FlagVal, c.val) + } + if !slices.Equal(txt.Flags, []string{c.name}) { + t.Errorf("%s: flags = %v, want [%s]", c.name, txt.Flags, c.name) + } + } +} + +// TestFlagsUnknownName covers the identifier the table does not know: the +// toolchain's assembler rejects the line ("unexpected TYPO evaluating +// expression") and so does the assembly path of this parser, while the +// tooling view stays tolerant, keeps the scanned names and carries no +// value the assembler could trust. +func TestFlagsUnknownName(t *testing.T) { + const header = "TEXT \u00b7f(SB), NOSPLIT|TYPO, $0" + + txt := parseTextHeader(t, header) + if !slices.Equal(txt.Flags, []string{"NOSPLIT", "TYPO"}) { + t.Errorf("raw flags = %v, want the scanned names", txt.Flags) + } + if txt.FlagVal != 0 { + t.Errorf("raw FlagVal = %d, want 0", txt.FlagVal) + } + + _, errs := parseTextHeaderExpanded(t, header) + if len(errs) != 1 { + t.Fatalf("assembly path: %d errors, want 1: %v", len(errs), errs) + } + perr, ok := errs[0].(Error) + if !ok { + t.Fatalf("error %v is not a parser.Error", errs[0]) + } + if !strings.Contains(perr.Msg, "unexpected TYPO evaluating expression") { + t.Errorf("message = %q, want the toolchain's rejection", perr.Msg) + } + if perr.Pos.Line != 1 || perr.Pos.Column != 22 { + t.Errorf("position = %d:%d, want 1:22", perr.Pos.Line, perr.Pos.Column) + } +} + +// TestFlagsLeftoverExpression covers the operand that never closes into +// one expression: the toolchain rejects it where the expression stops. +func TestFlagsLeftoverExpression(t *testing.T) { + _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT TYPO, $0") + if len(errs) != 1 || !strings.Contains(errs[0].Error(), "unexpected TYPO evaluating expression") { + t.Fatalf("errors = %v, want the toolchain's rejection at TYPO", errs) + } + txt := parseTextHeader(t, "TEXT \u00b7f(SB), NOSPLIT TYPO, $0") + if txt.FlagVal != 0 { + t.Errorf("raw FlagVal = %d, want 0", txt.FlagVal) + } +} + +// TestNoFrameDemandsZeroFrame covers the toolchain's frame check its arm64 +// backend raises: NOFRAME reserves no frame, so a declared positive frame +// contradicts the flag. The zero and negative frames are the flag's own +// shapes (the arm64 BSD syscall-stub pattern writes $-8). +func TestNoFrameDemandsZeroFrame(t *testing.T) { + _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT|NOFRAME, $8") + if len(errs) != 1 || !strings.Contains(errs[0].Error(), + "NOFRAME functions must have a frame size of 0, not 8") { + t.Fatalf("errors = %v, want the toolchain's frame check", errs) + } + for _, frame := range []string{"$0", "$-8", "$-0"} { + if _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT|NOFRAME, "+frame); len(errs) != 0 { + t.Errorf("%s: errors = %v, want none", frame, errs) + } + } + // The tooling view leaves the case to the linter's advisory. + txt := parseTextHeader(t, "TEXT \u00b7f(SB), NOSPLIT|NOFRAME, $8") + if txt.Frame == nil { + t.Fatal("raw parse lost the frame") + } +} + +// TestABIInternalRequiresNoSplit covers the toolchain's own TEXT check: an +// ABIInternal symbol must not need the stack the ABI wrapper would have to +// bridge, so the flag is mandatory. +func TestABIInternalRequiresNoSplit(t *testing.T) { + _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), $0") + if len(errs) != 1 || !strings.Contains(errs[0].Error(), + `TEXT "f": ABIInternal requires NOSPLIT`) { + t.Fatalf("errors = %v, want the toolchain's ABI check", errs) + } + if _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT, $0"); len(errs) != 0 { + t.Errorf("NOSPLIT present: errors = %v, want none", errs) + } + // The numeric spelling of the flag satisfies the check the same way + // the toolchain's integer test does. + if _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), 4, $0"); len(errs) != 0 { + t.Errorf("numeric NOSPLIT: errors = %v, want none", errs) + } +} + +// TestGloblFlagsOperand covers the GLOBL side: the atoms stay as written +// (the numeric combinations are what the link layer reads back) while the +// operand still folds to one value and validates on the assembly path. +func TestGloblFlagsOperand(t *testing.T) { + for _, c := range []struct { + operand string + wantFlag []string + wantVal int64 + wantSize int64 + }{ + {"RODATA|NOPTR, $8", []string{"RODATA", "NOPTR"}, 24, 8}, + {"10, $8", []string{"10"}, 10, 8}, + {"8|2, $8", []string{"8", "2"}, 10, 8}, + {"RODATA, $8", []string{"RODATA"}, 8, 8}, + {"$8", nil, 0, 8}, + } { + f, errs := Parse("t.s", "GLOBL \u00b7x(SB), "+c.operand+"\n") + if len(errs) > 0 { + t.Fatalf("parse %q: %v", c.operand, errs) + } + g := f.Decls[0].(*ast.Globl) + if !slices.Equal(g.Flags, c.wantFlag) { + t.Errorf("%s: flags = %v, want %v", c.operand, g.Flags, c.wantFlag) + } + if g.FlagVal != c.wantVal { + t.Errorf("%s: FlagVal = %d, want %d", c.operand, g.FlagVal, c.wantVal) + } + if c.wantSize != 0 && (g.Size == nil || !g.Size.Imm.HasVal || g.Size.Imm.Val != c.wantSize) { + t.Errorf("%s: size = %+v, want $%d", c.operand, g.Size, c.wantSize) + } + } + + f, errs := ParseWithOptions("t.s", "GLOBL \u00b7x(SB), RODATA|TYPO, $8\n", Options{Expand: true}) + if len(errs) != 1 || !strings.Contains(errs[0].Error(), "unexpected TYPO evaluating expression") { + t.Fatalf("errors = %v, want the toolchain's rejection", errs) + } + if f.Decls[0].(*ast.Globl).FlagVal != 0 { + t.Errorf("failed operand must carry no value") + } +} diff --git a/parser/parser.go b/parser/parser.go index f4b88eb..603080a 100644 --- a/parser/parser.go +++ b/parser/parser.go @@ -251,22 +251,35 @@ func (p *state) parseText(line []token.Token) { text.Name = sym rest = rest[n:] - // Consume the flags operand: everything between the symbol and the - // frame '$' is one operand, which the toolchain evaluates to a single - // integer (identifiers joined by '|', each a known textflag.h name, - // literals and constant arithmetic beside them). A single trailing - // comma is the separator that stood before the '$'. + // The tail splits into the toolchain's comma-separated operands: with + // two, the first is the flags expression and the last the frame; with + // three or more, the toolchain's own complaint. One operand is the + // frame alone, whatever it starts with, so a malformed frame takes the + // frame diagnostic and never reads as flags. rest = skipComma(rest) - text.Flags, text.FlagVal = p.evalFlags(flagsRun(rest), false) - for len(rest) > 0 && rest[0].Kind != token.Dollar { - rest = rest[1:] + ops := splitOperands(rest) + var frame []token.Token + switch { + case len(ops) >= 2: + text.Flags, text.FlagVal = p.evalFlags(flagsRun(ops[0]), false) + frame = ops[1] + if len(ops) > 2 { + p.errorf(line[0].Pos, "expect two or three operands for TEXT") + } + case len(ops) == 1: + frame = ops[0] + } + if len(frame) > 0 && frame[0].Kind != token.Dollar { + p.errorf(line[0].Pos, "TEXT frame size must be an immediate constant") + frame = nil } // Frame: $[-]number ; optional args: -number. The Go runtime writes // zero frames with an explicit sign ("$-0-24"), so the number may carry // one. Whatever remains after the header is the body and is parsed by // the caller. - if len(rest) > 0 && rest[0].Kind == token.Dollar { + if len(frame) > 0 && frame[0].Kind == token.Dollar { + rest = frame n := 1 neg := false if n < len(rest) && (rest[n].Kind == token.Minus || rest[n].Kind == token.Plus) { @@ -337,13 +350,15 @@ func (p *state) parseGlobl(line []token.Token) *ast.Globl { // operand as one constant expression, so every name is validated here // and the whole operand folds to g.FlagVal; Flags keeps the atoms as // written, the numeric spellings being data-side combinations the - // link layer reads back. - g.Flags, g.FlagVal = p.evalFlags(flagsRun(rest), true) - for i, t := range rest { - if t.Kind == token.Dollar { - g.Size = parseOperand(rest[i:], false) - break - } + // link layer reads back. The size is the last operand; a third + // operand's excess is the toolchain's own silence. + ops := splitOperands(rest) + switch { + case len(ops) >= 2: + g.Flags, g.FlagVal = p.evalFlags(flagsRun(ops[0]), true) + g.Size = parseOperand(ops[1], false) + case len(ops) == 1: + g.Size = parseOperand(ops[0], false) } return g }