diff --git a/ast/ast.go b/ast/ast.go index a1acbdb..3cc4b5c 100644 --- a/ast/ast.go +++ b/ast/ast.go @@ -58,8 +58,14 @@ type Text struct { Keyword token.Token // the TEXT token Name *Symbol // ·funcName(SB) Flags []string // NOSPLIT, DUPOK, … - Frame *Operand // $0 - Args *Operand // the -65 part; nil when absent + // FlagVal is the flags operand as the toolchain reads it: one integer + // the operand evaluates to, the OR of the named flags and any literals + // written beside them. Zero when the TEXT carries no flags operand, and + // after an evaluation failure, where Flags still carries the scanned + // names beside the diagnostic. + FlagVal int64 + Frame *Operand // $0 + Args *Operand // the -65 part; nil when absent Body []Stmt Doc string // preceding comment block (typically the Go signature) } @@ -72,6 +78,10 @@ type Globl struct { Keyword token.Token Name *Symbol Flags []string + // FlagVal is the flags operand folded to the one integer the toolchain + // evaluates it as, exactly as on Text. Zero when the GLOBL carries no + // flags operand or the operand failed to evaluate. + FlagVal int64 Size *Operand } diff --git a/parser/flags.go b/parser/flags.go new file mode 100644 index 0000000..23c85c6 --- /dev/null +++ b/parser/flags.go @@ -0,0 +1,187 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// The TEXT and GLOBL flags operand. +// +// The toolchain reads the middle operand of a TEXT (and of a GLOBL) as one +// constant expression: identifiers joined by '|', each a name +// runtime/textflag.h defines as a number, with literals and constant +// arithmetic beside them (cmd/asm/internal/asm/asm.go, evalInteger). gasm +// consumes textflag.h natively rather than expanding it, so the names are +// evaluated here against the same table, and an identifier outside it is +// the toolchain's evaluation failure. +package parser + +import ( + "strconv" + + "sourcedock.dev/petrbalvin/gasm-sdk/token" +) + +// textFlags is the flag table of runtime/textflag.h: the names the +// assembler knows and the values they stand for. +var textFlags = map[string]int64{ + "NOPROF": 1, + "DUPOK": 2, + "NOSPLIT": 4, + "RODATA": 8, + "NOPTR": 16, + "WRAPPER": 32, + "NEEDCTXT": 64, + "TLSBSS": 256, + "NOFRAME": 512, + "REFLECTMETHOD": 1024, + "TOPFRAME": 2048, + "ABIWRAPPER": 4096, +} + +// flagNOSPLIT and flagNOFRAME single out the two flags whose value the +// parser itself reasons about, in the toolchain's own TEXT diagnostics. +const ( + flagNOSPLIT = 4 + flagNOFRAME = 512 +) + +// flagOrder lists the table ascending by value, the order the canonical +// name expansion of a folded value reads in. +var flagOrder = []struct { + name string + val int64 +}{ + {"NOPROF", 1}, + {"DUPOK", 2}, + {"NOSPLIT", 4}, + {"RODATA", 8}, + {"NOPTR", 16}, + {"WRAPPER", 32}, + {"NEEDCTXT", 64}, + {"TLSBSS", 256}, + {"NOFRAME", 512}, + {"REFLECTMETHOD", 1024}, + {"TOPFRAME", 2048}, + {"ABIWRAPPER", 4096}, +} + +// flagsRun returns the flags operand of a TEXT or GLOBL directive: the +// tokens after the symbol up to the frame operand's '$', with comments and +// the one trailing comma that separated the operand from the '$' removed. +func flagsRun(g []token.Token) []token.Token { + n := 0 + for n < len(g) && g[n].Kind != token.Dollar { + n++ + } + out := make([]token.Token, 0, n) + for _, t := range g[:n] { + if t.Kind == token.Comment { + continue + } + out = append(out, t) + } + if len(out) > 0 && out[len(out)-1].Kind == token.Comma { + out = out[:len(out)-1] + } + return out +} + +// evalFlags evaluates one flags operand. Every identifier must name a +// flag in textFlags; the toolchain rejects any other with "unexpected NAME +// evaluating expression", and so does this. The whole operand folds to +// one integer, which is returned as the value. +// +// The names returned follow the operand's own spelling: an operand written +// purely as names keeps them in written order, duplicates included, the +// toolchain accepting both silently; an operand that mixes in literals or +// arithmetic has no written names to keep, so the folded value expands to +// the canonical ascending name list. Bits outside the table (a literal +// such as 8192) live in the value alone; the names cannot spell them. +// +// After a failure the names collected so far are returned with value 0: +// the tree stays usable beside the diagnostic, and value 0 is exactly the +// flags the assembler can trust the operand for. +func (p *state) evalFlags(g []token.Token, keepWritten bool) ([]string, int64) { + if len(g) == 0 { + return nil, 0 + } + // An explicit '$' makes the operand an immediate ('$NOSPLIT' over the + // preprocessor's '$4'); the toolchain reads it as the same constant. + if g[0].Kind == token.Dollar { + g = g[1:] + if len(g) == 0 { + return nil, 0 + } + } + + sub := make([]token.Token, 0, len(g)) + var names []string + namesOnly := true + failed := false + for _, t := range g { + if t.Kind == token.Ident { + v, ok := textFlags[t.Text] + if !ok { + if p.expand { + p.errorf(t.Pos, "unexpected %s evaluating expression", t.Text) + } + failed = true + } + names = append(names, t.Text) + if ok { + sub = append(sub, numberToken(v, t)) + } + continue + } + if t.Kind != token.Pipe { + // Anything beyond names and their '|' separators gives the + // operand arithmetic the written names cannot express. + namesOnly = false + } + if keepWritten && t.Kind == token.Number { + names = append(names, t.Text) + } + sub = append(sub, t) + } + if failed { + return names, 0 + } + v, rest, ok := foldExpr(sub) + if !ok || len(rest) > 0 { + if p.expand { + bad := g[0] + if ok && len(rest) > 0 { + bad = rest[0] + } + p.errorf(bad.Pos, "unexpected %s evaluating expression", bad.Text) + } + return names, 0 + } + if keepWritten { + return names, v + } + if namesOnly { + // Written names stay exactly as written; the value folds them. + return names, v + } + return flagNames(v), v +} + +// flagNames expands a folded flags value into the canonical ascending name +// list of the table's bits it holds. A negative value expands to nothing: +// its bit pattern is not a combination the names can spell. +func flagNames(v int64) []string { + if v <= 0 { + return nil + } + var out []string + for _, f := range flagOrder { + if v&f.val != 0 { + out = append(out, f.name) + v &^= f.val + } + } + return out +} + +// numberToken rebuilds t as the decimal literal of v at t's position. +func numberToken(v int64, t token.Token) token.Token { + return token.Token{Kind: token.Number, Text: strconv.FormatInt(v, 10), Pos: t.Pos, End: t.End} +} diff --git a/parser/parser.go b/parser/parser.go index a89f5ea..f4b88eb 100644 --- a/parser/parser.go +++ b/parser/parser.go @@ -43,6 +43,12 @@ type state struct { file *ast.File errs []error + // expand records the assembly path (ParseWithOptions with Expand): the + // one view of a file that must reject what the toolchain's assembler + // rejects. The tooling view (linter, formatter, language server) stays + // tolerant and reports through its own diagnostics instead. + expand bool + curText *ast.Text // the TEXT body labels/instructions attach to pending []string // comment lines awaiting a TEXT to become its Doc } @@ -245,13 +251,14 @@ func (p *state) parseText(line []token.Token) { text.Name = sym rest = rest[n:] - // Consume flags (identifiers, possibly '|' joined) up to the frame '$'. + // Consume the flags operand: everything between the symbol and the + // frame '$' is one operand, which the toolchain evaluates to a single + // integer (identifiers joined by '|', each a known textflag.h name, + // literals and constant arithmetic beside them). A single trailing + // comma is the separator that stood before the '$'. rest = skipComma(rest) + text.Flags, text.FlagVal = p.evalFlags(flagsRun(rest), false) for len(rest) > 0 && rest[0].Kind != token.Dollar { - if rest[0].Kind == token.Ident { - text.Flags = append(text.Flags, rest[0].Text) - } - // Commas, '|' and anything else between flags is skipped. rest = rest[1:] } @@ -291,6 +298,23 @@ func (p *state) parseText(line []token.Token) { } } + // The toolchain's own TEXT diagnostics over the evaluated flags + // (cmd/asm/internal/asm/asm.go asmText, and the frame check its arm64 + // backend raises): an ABIInternal symbol must not grow the stack, and + // NOFRAME reserves no frame at all, so a declared frame contradicts it. + // Both belong to the assembly path; the tooling view reports the frame + // case through the linter's own advisory instead. + if p.expand { + if text.Name.ABI == "ABIInternal" && text.FlagVal&flagNOSPLIT == 0 { + p.errorf(line[0].Pos, "TEXT %q: ABIInternal requires NOSPLIT", text.Name.Name) + } + if text.FlagVal&flagNOFRAME != 0 && text.Frame != nil && + text.Frame.Imm.HasVal && text.Frame.Imm.Val > 0 { + p.errorf(text.Frame.Pos, "NOFRAME functions must have a frame size of 0, not %d", + text.Frame.Imm.Val) + } + } + p.file.Decls = append(p.file.Decls, text) p.curText = text } @@ -309,15 +333,17 @@ func (p *state) parseGlobl(line []token.Token) *ast.Globl { g.Name = sym rest = skipComma(rest[n:]) // Flags are identifiers (RODATA, DUPOK) or legacy numeric constants - // (2, 8, 9, 10) from runtime/textflag.h. - for len(rest) > 0 && rest[0].Kind != token.Dollar { - if rest[0].Kind == token.Ident || rest[0].Kind == token.Number { - g.Flags = append(g.Flags, rest[0].Text) + // (2, 8, 9, 10) from runtime/textflag.h. The toolchain evaluates the + // operand as one constant expression, so every name is validated here + // and the whole operand folds to g.FlagVal; Flags keeps the atoms as + // written, the numeric spellings being data-side combinations the + // link layer reads back. + g.Flags, g.FlagVal = p.evalFlags(flagsRun(rest), true) + for i, t := range rest { + if t.Kind == token.Dollar { + g.Size = parseOperand(rest[i:], false) + break } - rest = rest[1:] - } - if len(rest) > 0 && rest[0].Kind == token.Dollar { - g.Size = parseOperand(rest, false) } return g } diff --git a/parser/preproc.go b/parser/preproc.go index 5bb1adc..04804bd 100644 --- a/parser/preproc.go +++ b/parser/preproc.go @@ -60,7 +60,7 @@ func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) { } else { lines = statementLines(tokens) } - p := &state{path: path} + p := &state{path: path, expand: opts.Expand} p.parse(lines) return p.file, append(errs, p.errs...) }