feat(parser): read the TEXT and GLOBL operands the toolchain counts them

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 21:50:20 +02:00
1 parent cea5db6964
commit 431d0d9b0b
3 files changed
+262 -31

No files matched your search

+6 -15
View File
@@ -62,23 +62,14 @@ var flagOrder = []struct {
{"ABIWRAPPER", 4096},
}
// flagsRun returns the flags operand of a TEXT or GLOBL directive: the
// tokens after the symbol up to the frame operand's '$', with comments and
// the one trailing comma that separated the operand from the '$' removed.
// flagsRun prepares one flags operand for evaluation: the comments are
// dropped, they are layout the expression never sees.
func flagsRun(g []token.Token) []token.Token {
n := 0
for n < len(g) && g[n].Kind != token.Dollar {
n++
}
out := make([]token.Token, 0, n)
for _, t := range g[:n] {
if t.Kind == token.Comment {
continue
out := make([]token.Token, 0, len(g))
for _, t := range g {
if t.Kind != token.Comment {
out = append(out, t)
}
out = append(out, t)
}
if len(out) > 0 && out[len(out)-1].Kind == token.Comma {
out = out[:len(out)-1]
}
return out
}
+225
View File
@@ -0,0 +1,225 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package parser
import (
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
)
// parseTextHeader parses the one TEXT directive of src, raw (no expansion).
func parseTextHeader(t *testing.T, header string) *ast.Text {
t.Helper()
f, errs := Parse("t.s", header+"\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse %q: %v", header, errs)
}
return f.Decls[0].(*ast.Text)
}
// parseTextHeaderExpanded parses the one TEXT directive of src on the
// assembly path, where the toolchain's rejections apply.
func parseTextHeaderExpanded(t *testing.T, header string) (*ast.Text, []error) {
t.Helper()
f, errs := ParseWithOptions("t.s", header+"\n\tRET\n", Options{Expand: true})
if len(f.Decls) != 1 {
t.Fatalf("parse %q: %d decls, want 1", header, len(f.Decls))
}
return f.Decls[0].(*ast.Text), errs
}
// TestTextFlagsExpression covers the flags operand as one expression: names
// joined by '|', the legacy numeric spellings, and the arithmetic forms the
// toolchain's evalInteger folds. Names written purely keep their written
// order and duplicates; an operand with literals in it expands to the
// canonical ascending name list.
func TestTextFlagsExpression(t *testing.T) {
for _, c := range []struct {
operand string
wantFlag []string
wantVal int64
}{
{"NOSPLIT", []string{"NOSPLIT"}, 4},
{"NOSPLIT|NOFRAME|DUPOK", []string{"NOSPLIT", "NOFRAME", "DUPOK"}, 518},
{"TOPFRAME|NOSPLIT", []string{"TOPFRAME", "NOSPLIT"}, 2052},
{"NOSPLIT|NOSPLIT", []string{"NOSPLIT", "NOSPLIT"}, 4},
{"4", []string{"NOSPLIT"}, 4},
{"512", []string{"NOFRAME"}, 512},
{"2|4", []string{"DUPOK", "NOSPLIT"}, 6},
{"(NOSPLIT|NOFRAME)", []string{"NOSPLIT", "NOFRAME"}, 516},
{"NOSPLIT|4", []string{"NOSPLIT"}, 4},
{"NOSPLIT&DUPOK", nil, 0},
{"$NOSPLIT", []string{"NOSPLIT"}, 4},
} {
txt := parseTextHeader(t, "TEXT \u00b7f(SB), "+c.operand+", $0")
if !slices.Equal(txt.Flags, c.wantFlag) {
t.Errorf("%s: flags = %v, want %v", c.operand, txt.Flags, c.wantFlag)
}
if txt.FlagVal != c.wantVal {
t.Errorf("%s: FlagVal = %d, want %d", c.operand, txt.FlagVal, c.wantVal)
}
if txt.Frame == nil || !txt.Frame.Imm.HasVal || txt.Frame.Imm.Val != 0 {
t.Errorf("%s: frame = %+v, want $0", c.operand, txt.Frame)
}
}
// No flags operand at all: the toolchain's zero.
txt := parseTextHeader(t, "TEXT \u00b7f(SB), $0")
if txt.Flags != nil || txt.FlagVal != 0 {
t.Errorf("no flags: flags = %v val = %d, want nil 0", txt.Flags, txt.FlagVal)
}
}
// TestTextFlagValues pins the flag table against the constants
// cmd/internal/obj/textflag.go defines and the toolchain's textflag.h
// ships; the assembler, linker and compiler must all agree on them.
func TestTextFlagValues(t *testing.T) {
for _, c := range []struct {
name string
val int64
}{
{"NOPROF", 1}, {"DUPOK", 2}, {"NOSPLIT", 4}, {"RODATA", 8},
{"NOPTR", 16}, {"WRAPPER", 32}, {"NEEDCTXT", 64}, {"TLSBSS", 256},
{"NOFRAME", 512}, {"REFLECTMETHOD", 1024}, {"TOPFRAME", 2048},
{"ABIWRAPPER", 4096},
} {
txt := parseTextHeader(t, "TEXT \u00b7f(SB), "+c.name+", $0")
if txt.FlagVal != c.val {
t.Errorf("%s = %d, want %d", c.name, txt.FlagVal, c.val)
}
if !slices.Equal(txt.Flags, []string{c.name}) {
t.Errorf("%s: flags = %v, want [%s]", c.name, txt.Flags, c.name)
}
}
}
// TestFlagsUnknownName covers the identifier the table does not know: the
// toolchain's assembler rejects the line ("unexpected TYPO evaluating
// expression") and so does the assembly path of this parser, while the
// tooling view stays tolerant, keeps the scanned names and carries no
// value the assembler could trust.
func TestFlagsUnknownName(t *testing.T) {
const header = "TEXT \u00b7f(SB), NOSPLIT|TYPO, $0"
txt := parseTextHeader(t, header)
if !slices.Equal(txt.Flags, []string{"NOSPLIT", "TYPO"}) {
t.Errorf("raw flags = %v, want the scanned names", txt.Flags)
}
if txt.FlagVal != 0 {
t.Errorf("raw FlagVal = %d, want 0", txt.FlagVal)
}
_, errs := parseTextHeaderExpanded(t, header)
if len(errs) != 1 {
t.Fatalf("assembly path: %d errors, want 1: %v", len(errs), errs)
}
perr, ok := errs[0].(Error)
if !ok {
t.Fatalf("error %v is not a parser.Error", errs[0])
}
if !strings.Contains(perr.Msg, "unexpected TYPO evaluating expression") {
t.Errorf("message = %q, want the toolchain's rejection", perr.Msg)
}
if perr.Pos.Line != 1 || perr.Pos.Column != 22 {
t.Errorf("position = %d:%d, want 1:22", perr.Pos.Line, perr.Pos.Column)
}
}
// TestFlagsLeftoverExpression covers the operand that never closes into
// one expression: the toolchain rejects it where the expression stops.
func TestFlagsLeftoverExpression(t *testing.T) {
_, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT TYPO, $0")
if len(errs) != 1 || !strings.Contains(errs[0].Error(), "unexpected TYPO evaluating expression") {
t.Fatalf("errors = %v, want the toolchain's rejection at TYPO", errs)
}
txt := parseTextHeader(t, "TEXT \u00b7f(SB), NOSPLIT TYPO, $0")
if txt.FlagVal != 0 {
t.Errorf("raw FlagVal = %d, want 0", txt.FlagVal)
}
}
// TestNoFrameDemandsZeroFrame covers the toolchain's frame check its arm64
// backend raises: NOFRAME reserves no frame, so a declared positive frame
// contradicts the flag. The zero and negative frames are the flag's own
// shapes (the arm64 BSD syscall-stub pattern writes $-8).
func TestNoFrameDemandsZeroFrame(t *testing.T) {
_, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT|NOFRAME, $8")
if len(errs) != 1 || !strings.Contains(errs[0].Error(),
"NOFRAME functions must have a frame size of 0, not 8") {
t.Fatalf("errors = %v, want the toolchain's frame check", errs)
}
for _, frame := range []string{"$0", "$-8", "$-0"} {
if _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f(SB), NOSPLIT|NOFRAME, "+frame); len(errs) != 0 {
t.Errorf("%s: errors = %v, want none", frame, errs)
}
}
// The tooling view leaves the case to the linter's advisory.
txt := parseTextHeader(t, "TEXT \u00b7f(SB), NOSPLIT|NOFRAME, $8")
if txt.Frame == nil {
t.Fatal("raw parse lost the frame")
}
}
// TestABIInternalRequiresNoSplit covers the toolchain's own TEXT check: an
// ABIInternal symbol must not need the stack the ABI wrapper would have to
// bridge, so the flag is mandatory.
func TestABIInternalRequiresNoSplit(t *testing.T) {
_, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f<ABIInternal>(SB), $0")
if len(errs) != 1 || !strings.Contains(errs[0].Error(),
`TEXT "f": ABIInternal requires NOSPLIT`) {
t.Fatalf("errors = %v, want the toolchain's ABI check", errs)
}
if _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f<ABIInternal>(SB), NOSPLIT, $0"); len(errs) != 0 {
t.Errorf("NOSPLIT present: errors = %v, want none", errs)
}
// The numeric spelling of the flag satisfies the check the same way
// the toolchain's integer test does.
if _, errs := parseTextHeaderExpanded(t, "TEXT \u00b7f<ABIInternal>(SB), 4, $0"); len(errs) != 0 {
t.Errorf("numeric NOSPLIT: errors = %v, want none", errs)
}
}
// TestGloblFlagsOperand covers the GLOBL side: the atoms stay as written
// (the numeric combinations are what the link layer reads back) while the
// operand still folds to one value and validates on the assembly path.
func TestGloblFlagsOperand(t *testing.T) {
for _, c := range []struct {
operand string
wantFlag []string
wantVal int64
wantSize int64
}{
{"RODATA|NOPTR, $8", []string{"RODATA", "NOPTR"}, 24, 8},
{"10, $8", []string{"10"}, 10, 8},
{"8|2, $8", []string{"8", "2"}, 10, 8},
{"RODATA, $8", []string{"RODATA"}, 8, 8},
{"$8", nil, 0, 8},
} {
f, errs := Parse("t.s", "GLOBL \u00b7x(SB), "+c.operand+"\n")
if len(errs) > 0 {
t.Fatalf("parse %q: %v", c.operand, errs)
}
g := f.Decls[0].(*ast.Globl)
if !slices.Equal(g.Flags, c.wantFlag) {
t.Errorf("%s: flags = %v, want %v", c.operand, g.Flags, c.wantFlag)
}
if g.FlagVal != c.wantVal {
t.Errorf("%s: FlagVal = %d, want %d", c.operand, g.FlagVal, c.wantVal)
}
if c.wantSize != 0 && (g.Size == nil || !g.Size.Imm.HasVal || g.Size.Imm.Val != c.wantSize) {
t.Errorf("%s: size = %+v, want $%d", c.operand, g.Size, c.wantSize)
}
}
f, errs := ParseWithOptions("t.s", "GLOBL \u00b7x(SB), RODATA|TYPO, $8\n", Options{Expand: true})
if len(errs) != 1 || !strings.Contains(errs[0].Error(), "unexpected TYPO evaluating expression") {
t.Fatalf("errors = %v, want the toolchain's rejection", errs)
}
if f.Decls[0].(*ast.Globl).FlagVal != 0 {
t.Errorf("failed operand must carry no value")
}
}
+31 -16
View File
@@ -251,22 +251,35 @@ func (p *state) parseText(line []token.Token) {
text.Name = sym
rest = rest[n:]
// Consume the flags operand: everything between the symbol and the
// frame '$' is one operand, which the toolchain evaluates to a single
// integer (identifiers joined by '|', each a known textflag.h name,
// literals and constant arithmetic beside them). A single trailing
// comma is the separator that stood before the '$'.
// The tail splits into the toolchain's comma-separated operands: with
// two, the first is the flags expression and the last the frame; with
// three or more, the toolchain's own complaint. One operand is the
// frame alone, whatever it starts with, so a malformed frame takes the
// frame diagnostic and never reads as flags.
rest = skipComma(rest)
text.Flags, text.FlagVal = p.evalFlags(flagsRun(rest), false)
for len(rest) > 0 && rest[0].Kind != token.Dollar {
rest = rest[1:]
ops := splitOperands(rest)
var frame []token.Token
switch {
case len(ops) >= 2:
text.Flags, text.FlagVal = p.evalFlags(flagsRun(ops[0]), false)
frame = ops[1]
if len(ops) > 2 {
p.errorf(line[0].Pos, "expect two or three operands for TEXT")
}
case len(ops) == 1:
frame = ops[0]
}
if len(frame) > 0 && frame[0].Kind != token.Dollar {
p.errorf(line[0].Pos, "TEXT frame size must be an immediate constant")
frame = nil
}
// Frame: $[-]number ; optional args: -number. The Go runtime writes
// zero frames with an explicit sign ("$-0-24"), so the number may carry
// one. Whatever remains after the header is the body and is parsed by
// the caller.
if len(rest) > 0 && rest[0].Kind == token.Dollar {
if len(frame) > 0 && frame[0].Kind == token.Dollar {
rest = frame
n := 1
neg := false
if n < len(rest) && (rest[n].Kind == token.Minus || rest[n].Kind == token.Plus) {
@@ -337,13 +350,15 @@ func (p *state) parseGlobl(line []token.Token) *ast.Globl {
// operand as one constant expression, so every name is validated here
// and the whole operand folds to g.FlagVal; Flags keeps the atoms as
// written, the numeric spellings being data-side combinations the
// link layer reads back.
g.Flags, g.FlagVal = p.evalFlags(flagsRun(rest), true)
for i, t := range rest {
if t.Kind == token.Dollar {
g.Size = parseOperand(rest[i:], false)
break
}
// link layer reads back. The size is the last operand; a third
// operand's excess is the toolchain's own silence.
ops := splitOperands(rest)
switch {
case len(ops) >= 2:
g.Flags, g.FlagVal = p.evalFlags(flagsRun(ops[0]), true)
g.Size = parseOperand(ops[1], false)
case len(ops) == 1:
g.Size = parseOperand(ops[0], false)
}
return g
}