// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // The preprocessor turns #define and #include directives into the token // stream the parser really sees, the way the Go toolchain's assembler does: // object and parameterised macros expand at the point of use, and an // #include splices the named file's lines in place of the directive. The // pass runs only on the assembly path (gasm asm, diff, the corpus audit), // where the result is machine code; parsing for the linter, formatter and // language server keeps the raw file so their view of #define lines, and // therefore their macro-aware behaviour, is unchanged. package parser import ( "fmt" "os" "path/filepath" "slices" "strconv" "strings" "unicode/utf8" "sourcedock.dev/petrbalvin/gasm-sdk/ast" "sourcedock.dev/petrbalvin/gasm-sdk/lexer" "sourcedock.dev/petrbalvin/gasm-sdk/token" ) // Options controls the optional preprocessing applied before a file is // parsed. The zero value reproduces Parse exactly. type Options struct { // IncludeDirs lists the -I directories searched for #include files, // in order, after the including file's own directory. IncludeDirs []string // Expand enables macro expansion, include splicing and the // statement-separator reading of ';' that the expanded bodies rely on. Expand bool // Predefines names the macros defined before the file is read. The // go command drives go tool asm with -D GOOS_ -D GOARCH_, // and GOROOT's own headers (go_tls.h, asm_riscv64.h) select their // platform blocks with #ifdef on exactly those names, so an assembler // without them cannot see the platform definitions at all. Predefines map[string]string } // ParseWithOptions parses src like Parse, optionally preprocessing it first. // The returned file is usable even when errors is non-empty. func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) { tokens := lexer.Tokenize(src) var lines [][]token.Token var errs []error if opts.Expand { pp := &preproc{opts: opts, macros: map[string]*macroDef{}} for name, value := range opts.Predefines { body := lexer.Tokenize(value) pp.macros[name] = ¯oDef{name: name, body: body} pp.bodyTokens += len(body) } lines = pp.fileLines(path, tokens, token.Position{}) errs = pp.errs } else { lines = statementLines(tokens) } p := &state{path: path} p.parse(lines) return p.file, append(errs, p.errs...) } // maxExpansionDepth bounds recursive macro expansion; the toolchain's // assembler gives up after 100 nested invocations without producing a token. const maxExpansionDepth = 100 // expandWorkFactor and expandWorkFloor size the per-line expansion work // budget: a generous multiple of the line and of everything the macro table // can inject into it. No real file approaches the budget, because real // macros multiply their input by a modest factor, while amplification that // grows exponentially with nesting pays for every token it produces and // stops at the budget instead of running the machine for hours. const ( expandWorkFactor = 64 expandWorkFloor = 4096 ) // textflagHeader names the one header gasm does not splice: its flag macros // (NOSPLIT, RODATA, …) are consumed by name throughout gasm's parser, // encoders and linter, and expanding them to their numeric constants would // leave every consumer blind to them. const textflagHeader = "textflag.h" // macroDef is one #define. A nil args slice is an object macro; a non-nil // (possibly empty) one is parameterised, the C distinction between // "#define A(x)" and "#define A (x)". type macroDef struct { name string args []string body []token.Token } // preproc carries the state of one expansion pass: the live macro table, the // chain of files currently being read, for cycle detection, and the // conditional-inclusion stack of #ifdef regions. type preproc struct { opts Options macros map[string]*macroDef errs []error stack []string // absolute paths of files being read, innermost last ifdefStack []bool // one entry per open #ifdef/#ifndef, its truth bodyTokens int // total length of every defined macro body } // enabled reports whether the position being read is inside a live // conditional branch. Directives inside a disabled branch contribute // nothing, and its content lines are dropped, exactly as the toolchain's // input stack does. func (pp *preproc) enabled() bool { return len(pp.ifdefStack) == 0 || pp.ifdefStack[len(pp.ifdefStack)-1] } func (pp *preproc) errorf(pos token.Position, format string, args ...any) { pp.errs = append(pp.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)}) } // fileLines tokenizes and preprocesses one file into logical lines. // Directive lines are kept (the parser records them for the tooling); // #include lines are replaced by the included file's lines. includePos is // the position of the #include that pulled this file in, zero for the // top-level file, and only serves cycle diagnostics. func (pp *preproc) fileLines(path string, tokens []token.Token, includePos token.Position) [][]token.Token { abs, err := filepath.Abs(path) if err != nil { abs = filepath.Clean(path) } if slices.Contains(pp.stack, abs) { if includePos.IsValid() { pp.errorf(includePos, "#include %q: include cycle (%s is already being read)", path, filepath.Base(path)) } return nil } pp.stack = append(pp.stack, abs) var out [][]token.Token for _, line := range splitLines(tokens) { if len(line) == 0 { out = append(out, line) continue } if line[0].Kind == token.Hash { out = append(out, pp.directive(line, filepath.Dir(path))...) continue } if !pp.enabled() { continue } out = append(out, splitOnSemicolons(pp.expandTokens(line))...) } pp.stack = pp.stack[:len(pp.stack)-1] if len(pp.stack) == 0 && len(pp.ifdefStack) > 0 { // The stack is per-input, shared across includes, so only the // top-level file's end can decide the input was left unclosed. pp.errorf(token.Position{Line: 1, Column: 1}, "unclosed #ifdef or #ifndef") } return out } // directive processes one '#' line and returns the lines to keep in the // stream: every directive line is kept as-is for the parser (which records // it), except #include, which is replaced by the spliced content. // Conditionals are tracked on every line; every other directive is inert // inside a disabled branch. func (pp *preproc) directive(line []token.Token, dir string) [][]token.Token { if len(line) < 2 || line[1].Kind != token.Ident { return [][]token.Token{line} } switch line[1].Text { case "ifdef", "ifndef": pp.ifdef(line, line[1].Text == "ifndef") case "else": pp.elseBranch(line) case "endif": pp.endif(line) case "define": if pp.enabled() { pp.define(line) } case "undef": if pp.enabled() { pp.undef(line) } case "include": if pp.enabled() { return pp.include(line, dir) } default: // #line and unknown directives are recorded but not interpreted: // conservative support keeps the parser's view intact and files // using them fail on their content, not silently. } return [][]token.Token{line} } // ifdef handles "#ifdef NAME" and "#ifndef NAME", pushing the branch's truth // onto the conditional stack. A branch opened inside a disabled region is // itself disabled, however the name resolves. func (pp *preproc) ifdef(line []token.Token, inverted bool) { truth := false if len(line) >= 3 && line[2].Kind == token.Ident { _, defined := pp.macros[line[2].Text] truth = defined != inverted } else { pp.errorf(line[0].Pos, "expected identifier after #%s", line[1].Text) } if !pp.enabled() { truth = false } pp.ifdefStack = append(pp.ifdefStack, truth) } // elseBranch flips the innermost conditional's truth, but only when the // region enclosing it is itself live: the toolchain keeps outer overrides. func (pp *preproc) elseBranch(line []token.Token) { if len(pp.ifdefStack) == 0 { pp.errorf(line[0].Pos, "unmatched #else") return } if len(pp.ifdefStack) == 1 || pp.ifdefStack[len(pp.ifdefStack)-2] { pp.ifdefStack[len(pp.ifdefStack)-1] = !pp.ifdefStack[len(pp.ifdefStack)-1] } } // endif closes the innermost conditional. func (pp *preproc) endif(line []token.Token) { if len(pp.ifdefStack) == 0 { pp.errorf(line[0].Pos, "unmatched #endif") return } pp.ifdefStack = pp.ifdefStack[:len(pp.ifdefStack)-1] } // define parses "#define NAME[(formals)] body" into the macro table. The // body runs to the end of the logical line (the lexer has already spliced // backslash continuations) and stops at a comment, which never expands. func (pp *preproc) define(line []token.Token) { if len(line) < 3 || line[2].Kind != token.Ident { return } name := line[2] args := []string(nil) body := line[3:] // The definition is parameterised only when '(' follows the name // directly; the toolchain separates "#define A(x)" from // "#define A (x)" by adjacency, and so does the column check here. if len(body) > 0 && body[0].Kind == token.LParen && body[0].Pos.Column == name.Pos.Column+utf8.RuneCountInString(name.Text) { args = []string{} i := 1 for i < len(body) && body[i].Kind != token.RParen { if body[i].Kind == token.Ident { args = append(args, body[i].Text) } i++ } if i < len(body) { body = body[i+1:] } else { body = nil } } if i := slices.IndexFunc(body, func(t token.Token) bool { return t.Kind == token.Comment }); i >= 0 { body = body[:i] } if old, exists := pp.macros[name.Text]; exists { // The toolchain refuses redefinition, so a file the oracle accepts // never redefines; failing here keeps that contract visible. pp.errorf(name.Pos, "redefinition of macro %s", name.Text) pp.bodyTokens -= len(old.body) } stored := pp.bodyWithBreaks(body) pp.bodyTokens += len(stored) pp.macros[name.Text] = ¯oDef{name: name.Text, args: args, body: stored} } // bodyWithBreaks records the statement boundaries the continuations carry. // The lexer splices backslash-continued lines into one logical line, but the // toolchain keeps the newline as a token in the stored body, which is how a // multi-instruction body without semicolons (the arm64 style) still splits // into statements on expansion. A line change inside the logical line is // exactly a continuation, so the boundary is restored from the positions. func (pp *preproc) bodyWithBreaks(body []token.Token) []token.Token { out := make([]token.Token, 0, len(body)) for i, t := range body { if i > 0 && t.Pos.Line != body[i-1].Pos.Line { out = append(out, token.Token{Kind: token.Newline, Text: "\n", Pos: t.Pos, End: t.Pos}) } out = append(out, t) } return out } // undef handles "#undef NAME", which the toolchain honours and requires to // name a defined macro. func (pp *preproc) undef(line []token.Token) { if len(line) < 3 || line[2].Kind != token.Ident { return } def, ok := pp.macros[line[2].Text] if !ok { pp.errorf(line[2].Pos, "#undef for undefined macro %s", line[2].Text) return } pp.bodyTokens -= len(def.body) delete(pp.macros, line[2].Text) } // include resolves and splices "#include \"file\"". A header that cannot be // read keeps the directive line in the stream, with a diagnostic. func (pp *preproc) include(line []token.Token, dir string) [][]token.Token { if len(line) < 3 || line[2].Kind != token.String { return [][]token.Token{line} } header := line[2] name, err := strconv.Unquote(header.Text) if err != nil { pp.errorf(header.Pos, "unquoting include file name: %v", err) return [][]token.Token{line} } if filepath.Base(name) == textflagHeader { // Flag macros are handled natively (see textflagHeader); the // directive stays so tools still see the include. return [][]token.Token{line} } resolved, ok := pp.resolve(name, dir) if !ok { searched := append([]string{dir}, pp.opts.IncludeDirs...) pp.errorf(header.Pos, "#include %q: file not found (searched %s)", name, strings.Join(searched, ", ")) return [][]token.Token{line} } src, err := os.ReadFile(resolved) if err != nil { pp.errorf(header.Pos, "#include %q: %v", name, err) return [][]token.Token{line} } return pp.fileLines(resolved, lexer.Tokenize(string(src)), header.Pos) } // resolve looks an include name up the way the toolchain does: as written // (relative to the working directory), then relative to the including // file's directory, then in each -I directory in order. func (pp *preproc) resolve(name, dir string) (string, bool) { candidates := []string{name} if !filepath.IsAbs(name) { candidates = append(candidates, filepath.Join(dir, name)) for _, d := range pp.opts.IncludeDirs { candidates = append(candidates, filepath.Join(d, name)) } } for _, c := range candidates { if st, err := os.Stat(c); err == nil && !st.IsDir() { return c, true } } return "", false } // tokenFrame is one layer of the pending token stream during expansion: a // slice of tokens plus how far it has been read. type tokenFrame struct { toks []token.Token i int } // expandTokens expands every macro invocation in a token sequence, // recursively, with a depth guard, the way the toolchain's input stack // re-reads pushed tokens: an object macro may name a parameterised one, and // the argument list of the expansion may then come from the tokens that // follow. The pending stream is a stack of frames, so an expansion pushes // its body as the next thing to read instead of splicing it into one flat // slice: a long run of invocations costs work proportional to what it // produces, never the square of the line. Work is billed against a budget, // because a body that repeats its argument multiplies every nesting level; // amplification that outgrows the budget is an error, not an hours-long // machine commitment. func (pp *preproc) expandTokens(in []token.Token) []token.Token { stack := []tokenFrame{{toks: in}} out := make([]token.Token, 0, len(in)) consecutive := 0 budget := expandWorkFactor*(len(in)+pp.bodyTokens) + expandWorkFloor spent := func(n int) bool { budget -= n return budget < 0 } // nextAfter returns the first unread token behind the invocation at the // top of the stack, together with the frame and offset it sits at. nextAfter := func() (fi, off int, tok token.Token, ok bool) { for j, s := range slices.Backward(stack) { start := s.i if j == len(stack)-1 { start++ } if start < len(s.toks) { return j, start, s.toks[start], true } } return 0, 0, token.Token{}, false } for len(stack) > 0 { top := &stack[len(stack)-1] if top.i >= len(top.toks) { stack = stack[:len(stack)-1] continue } t := top.toks[top.i] if t.Kind != token.Ident { out = append(out, t) top.i++ consecutive = 0 if spent(1) { break } continue } def, suffix := pp.macroFor(t.Text) if def == nil { out = append(out, t) top.i++ consecutive = 0 if spent(1) { break } continue } // The guard mirrors the toolchain's: 100 nested invocations in a // row without a plain token between them means recursion. consecutive++ if consecutive > maxExpansionDepth { pp.errorf(t.Pos, "recursive macro invocation (deeper than %d levels)", maxExpansionDepth) return nil } if def.args == nil { top.i++ body := restamp(def.body, t.Pos) if suffix != "" { // The macro was reached only through a compound spelling // (ACC0.B16 over "#define ACC0 V8"), so the selector has // to travel with the expansion. body = appendSelector(body, suffix, t.Pos) } stack = append(stack, tokenFrame{toks: body}) if spent(len(body)) { break } continue } // A parameterised macro invoked without its parentheses stands // unexpanded, naming itself, as in the toolchain. The parenthesis // may sit past the end of this body, in the pending frames behind // it, exactly where the toolchain's input stack would find it. if _, _, nxt, ok := nextAfter(); !ok || nxt.Kind != token.LParen { out = append(out, t) top.i++ consecutive = 0 if spent(1) { break } continue } args, fi, off, ok := pp.collectArgs(stack, t) if !ok { return nil } if spent(collectedTokens(args)) { break } // A zero-argument macro may be invoked as NAME(). if len(def.args) == 0 && len(args) == 1 && len(args[0]) == 0 { args = nil } if len(args) != len(def.args) { pp.errorf(t.Pos, "wrong arg count for macro %s: got %d, want %d", t.Text, len(args), len(def.args)) // Skip the invocation: drop the frames it consumed and resume // right after its closing parenthesis. stack = stack[:fi+1] stack[fi].i = off consecutive = 0 continue } sub := make([]token.Token, 0, len(def.body)) for _, bt := range def.body { if bt.Kind == token.Ident { if k := slices.Index(def.args, bt.Text); k >= 0 { sub = append(sub, restamp(args[k], t.Pos)...) continue } // A parameter used with an element or lane selector: the // lexer folds A.S4 into one identifier, so the whole-token // match above cannot see the parameter. The toolchain // lexes the period separately and substitutes the name // alone; splitting at the FIRST period and pasting the // argument back in front of the selector is the equivalent // for this lexer. if k, sel := parameterSelector(bt.Text, def.args); k >= 0 { sub = append(sub, restamp(pasteSelector(args[k], sel), t.Pos)...) continue } } sub = append(sub, bt) } // The invocation consumed every frame down to its closing // parenthesis; resume there, with the substitution read first. stack = stack[:fi+1] stack[fi].i = off stack = append(stack, tokenFrame{toks: sub}) if spent(len(sub)) { break } } if budget < 0 { pp.errorf(in[0].Pos, "macro expansion exceeds the work budget of %d tokens", expandWorkFactor*(len(in)+pp.bodyTokens)+expandWorkFloor) return nil } return out } // collectedTokens counts the tokens an argument list carried off the pending // stream, so the expansion's work budget pays for reading them. func collectedTokens(args [][]token.Token) int { n := 0 for _, a := range args { n += len(a) } return n } // macroFor finds the macro a use names. The lexer folds NAME.selector into // one identifier token, so a macro written behind a selector suffix // (ACC0.B16 over "#define ACC0 V8") never matches a whole-token table // lookup; the toolchain splits on the period and reads the two halves, so // the prefix before the FIRST period is tried here as well and the caller // re-attaches the suffix to whatever the macro expands to. Only a whole // name counts: AB.S4 does not reach a macro named A, and a parameterised // macro is not hidden behind a selector, because its invocation would need // the parentheses to follow the bare name. func (pp *preproc) macroFor(text string) (*macroDef, string) { if def := pp.macros[text]; def != nil { return def, "" } if j := strings.IndexByte(text, '.'); j > 0 { if def := pp.macros[text[:j]]; def != nil && def.args == nil { return def, text[j:] } } return nil, "" } // appendSelector glues a selector suffix onto an object macro's expansion: // the selector binds to the identifier the expansion ends with, the way the // toolchain's operand parser reads V0 and .B16 back as one register // spelling. An expansion that does not end in an identifier carries the // selector as its own token, which the parser then reports where it cannot // parse it. func appendSelector(body []token.Token, suffix string, pos token.Position) []token.Token { if n := len(body); n > 0 && body[n-1].Kind == token.Ident { body[n-1].Text += suffix return body } return append(body, token.Token{Kind: token.Ident, Text: suffix, Pos: pos, End: pos}) } // parameterSelector reports the argument a compound body token names: the // parameter whose whole name occupies the text before the token's FIRST // period, with the selector that follows. k is negative when no parameter // matches, which leaves tokens like AB.S4 untouched even though a parameter // A is bound. func parameterSelector(text string, args []string) (int, string) { j := strings.IndexByte(text, '.') if j <= 0 { return -1, "" } if k := slices.Index(args, text[:j]); k >= 0 { return k, text[j:] } return -1, "" } // pasteSelector joins an argument with the selector a compound body token // carries, textually: the selector binds to the identifier the argument // ends with, so A.S4 over the argument V0.B16 spells V0.B16.S4, exactly the // operand the toolchain's split-then-substitute leaves behind. An argument // with no trailing identifier carries the selector as a separate token, // which the parser then reports where it cannot parse it. func pasteSelector(val []token.Token, suffix string) []token.Token { if len(val) == 0 { return []token.Token{{Kind: token.Ident, Text: suffix}} } out := slices.Clone(val) if n := len(out); out[n-1].Kind == token.Ident { out[n-1].Text += suffix return out } return append(out, token.Token{Kind: token.Ident, Text: suffix}) } // collectArgs reads the actual argument tokens of an invocation at the top // of the stack, whose opening parenthesis is the first unread token behind // the invocation. Commas separate arguments except inside nested // parentheses, and the list may run on into the pending frames below, the // way the toolchain's input stack keeps reading pushed-back tokens. It // returns the arguments together with the frame and offset of the first // token after the closing parenthesis; ok is false when the list never // closes, which is a diagnostic. func (pp *preproc) collectArgs(stack []tokenFrame, name token.Token) (args [][]token.Token, fi, off int, ok bool) { var cur []token.Token nesting := 0 started := false for j, s := range slices.Backward(stack) { start := s.i if j == len(stack)-1 { start++ // past the invocation's name } for k := start; k < len(s.toks); k++ { t := s.toks[k] if !started { // The opening parenthesis itself. started = true continue } switch t.Kind { case token.LParen: nesting++ cur = append(cur, t) case token.RParen: if nesting == 0 { return append(args, cur), j, k + 1, true } nesting-- cur = append(cur, t) case token.Comma: if nesting == 0 { args = append(args, cur) cur = nil continue } cur = append(cur, t) case token.Comment: pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text) return nil, 0, 0, false default: cur = append(cur, t) } } } pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text) return nil, 0, 0, false } // restamp copies body tokens to the invocation's position, so diagnostics // and the line table point where the macro was used, as the toolchain's // input stack does. func restamp(body []token.Token, pos token.Position) []token.Token { out := make([]token.Token, len(body)) for i, t := range body { t.Pos, t.End = pos, pos out[i] = t } return out } // splitOnSemicolons breaks a token sequence at ';' statement separators and // at the Newline markers that record continuation boundaries inside macro // bodies, producing the logical lines the parser expects. The separators // carry no meaning beyond the break, so the pieces are exactly what the same // statements on separate lines would produce. func splitOnSemicolons(ts []token.Token) [][]token.Token { var out [][]token.Token start := 0 for i, t := range ts { if t.Kind == token.Semicolon || t.Kind == token.Newline { if i > start { out = append(out, ts[start:i]) } start = i + 1 } } if start < len(ts) { out = append(out, ts[start:]) } return out }