// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // The preprocessor turns #define and #include directives into the token // stream the parser really sees, the way the Go toolchain's assembler does: // object and parameterised macros expand at the point of use, and an // #include splices the named file's lines in place of the directive. The // pass runs only on the assembly path (gasm asm, diff, the corpus audit), // where the result is machine code; parsing for the linter, formatter and // language server keeps the raw file so their view of #define lines, and // therefore their macro-aware behaviour, is unchanged. package parser import ( "fmt" "os" "path/filepath" "slices" "strconv" "strings" "unicode/utf8" "sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/lexer" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) // Options controls the optional preprocessing applied before a file is // parsed. The zero value reproduces Parse exactly. type Options struct { // IncludeDirs lists the -I directories searched for #include files, // in order, after the including file's own directory. IncludeDirs []string // Expand enables macro expansion, include splicing and the // statement-separator reading of ';' that the expanded bodies rely on. Expand bool } // ParseWithOptions parses src like Parse, optionally preprocessing it first. // The returned file is usable even when errors is non-empty. func ParseWithOptions(path, src string, opts Options) (*ast.File, []error) { tokens := lexer.Tokenize(src) var lines [][]token.Token var errs []error if opts.Expand { pp := &preproc{opts: opts, macros: map[string]*macroDef{}} lines = pp.fileLines(path, tokens, token.Position{}) errs = pp.errs } else { lines = statementLines(tokens) } p := &state{path: path} p.parse(lines) return p.file, append(errs, p.errs...) } // maxExpansionDepth bounds recursive macro expansion; the toolchain's // assembler gives up after 100 nested invocations without producing a token. const maxExpansionDepth = 100 // textflagHeader names the one header gasm does not splice: its flag macros // (NOSPLIT, RODATA, …) are consumed by name throughout gasm's parser, // encoders and linter, and expanding them to their numeric constants would // leave every consumer blind to them. const textflagHeader = "textflag.h" // macroDef is one #define. A nil args slice is an object macro; a non-nil // (possibly empty) one is parameterised, the C distinction between // "#define A(x)" and "#define A (x)". type macroDef struct { name string args []string body []token.Token } // preproc carries the state of one expansion pass: the live macro table, the // chain of files currently being read, for cycle detection, and the // conditional-inclusion stack of #ifdef regions. type preproc struct { opts Options macros map[string]*macroDef errs []error stack []string // absolute paths of files being read, innermost last ifdefStack []bool // one entry per open #ifdef/#ifndef, its truth } // enabled reports whether the position being read is inside a live // conditional branch. Directives inside a disabled branch contribute // nothing, and its content lines are dropped, exactly as the toolchain's // input stack does. func (pp *preproc) enabled() bool { return len(pp.ifdefStack) == 0 || pp.ifdefStack[len(pp.ifdefStack)-1] } func (pp *preproc) errorf(pos token.Position, format string, args ...any) { pp.errs = append(pp.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)}) } // fileLines tokenizes and preprocesses one file into logical lines. // Directive lines are kept (the parser records them for the tooling); // #include lines are replaced by the included file's lines. includePos is // the position of the #include that pulled this file in, zero for the // top-level file, and only serves cycle diagnostics. func (pp *preproc) fileLines(path string, tokens []token.Token, includePos token.Position) [][]token.Token { abs, err := filepath.Abs(path) if err != nil { abs = filepath.Clean(path) } if slices.Contains(pp.stack, abs) { if includePos.IsValid() { pp.errorf(includePos, "#include %q: include cycle (%s is already being read)", path, filepath.Base(path)) } return nil } pp.stack = append(pp.stack, abs) var out [][]token.Token for _, line := range splitLines(tokens) { if len(line) == 0 { out = append(out, line) continue } if line[0].Kind == token.Hash { out = append(out, pp.directive(line, filepath.Dir(path))...) continue } if !pp.enabled() { continue } out = append(out, splitOnSemicolons(pp.expandTokens(line))...) } pp.stack = pp.stack[:len(pp.stack)-1] if len(pp.stack) == 0 && len(pp.ifdefStack) > 0 { // The stack is per-input, shared across includes, so only the // top-level file's end can decide the input was left unclosed. pp.errorf(token.Position{Line: 1, Column: 1}, "unclosed #ifdef or #ifndef") } return out } // directive processes one '#' line and returns the lines to keep in the // stream: every directive line is kept as-is for the parser (which records // it), except #include, which is replaced by the spliced content. // Conditionals are tracked on every line; every other directive is inert // inside a disabled branch. func (pp *preproc) directive(line []token.Token, dir string) [][]token.Token { if len(line) < 2 || line[1].Kind != token.Ident { return [][]token.Token{line} } switch line[1].Text { case "ifdef", "ifndef": pp.ifdef(line, line[1].Text == "ifndef") case "else": pp.elseBranch(line) case "endif": pp.endif(line) case "define": if pp.enabled() { pp.define(line) } case "undef": if pp.enabled() { pp.undef(line) } case "include": if pp.enabled() { return pp.include(line, dir) } default: // #line and unknown directives are recorded but not interpreted: // conservative support keeps the parser's view intact and files // using them fail on their content, not silently. } return [][]token.Token{line} } // ifdef handles "#ifdef NAME" and "#ifndef NAME", pushing the branch's truth // onto the conditional stack. A branch opened inside a disabled region is // itself disabled, however the name resolves. func (pp *preproc) ifdef(line []token.Token, inverted bool) { truth := false if len(line) >= 3 && line[2].Kind == token.Ident { _, defined := pp.macros[line[2].Text] truth = defined != inverted } else { pp.errorf(line[0].Pos, "expected identifier after #%s", line[1].Text) } if !pp.enabled() { truth = false } pp.ifdefStack = append(pp.ifdefStack, truth) } // elseBranch flips the innermost conditional's truth, but only when the // region enclosing it is itself live: the toolchain keeps outer overrides. func (pp *preproc) elseBranch(line []token.Token) { if len(pp.ifdefStack) == 0 { pp.errorf(line[0].Pos, "unmatched #else") return } if len(pp.ifdefStack) == 1 || pp.ifdefStack[len(pp.ifdefStack)-2] { pp.ifdefStack[len(pp.ifdefStack)-1] = !pp.ifdefStack[len(pp.ifdefStack)-1] } } // endif closes the innermost conditional. func (pp *preproc) endif(line []token.Token) { if len(pp.ifdefStack) == 0 { pp.errorf(line[0].Pos, "unmatched #endif") return } pp.ifdefStack = pp.ifdefStack[:len(pp.ifdefStack)-1] } // define parses "#define NAME[(formals)] body" into the macro table. The // body runs to the end of the logical line (the lexer has already spliced // backslash continuations) and stops at a comment, which never expands. func (pp *preproc) define(line []token.Token) { if len(line) < 3 || line[2].Kind != token.Ident { return } name := line[2] args := []string(nil) body := line[3:] // The definition is parameterised only when '(' follows the name // directly; the toolchain separates "#define A(x)" from // "#define A (x)" by adjacency, and so does the column check here. if len(body) > 0 && body[0].Kind == token.LParen && body[0].Pos.Column == name.Pos.Column+utf8.RuneCountInString(name.Text) { args = []string{} i := 1 for i < len(body) && body[i].Kind != token.RParen { if body[i].Kind == token.Ident { args = append(args, body[i].Text) } i++ } if i < len(body) { body = body[i+1:] } else { body = nil } } if i := slices.IndexFunc(body, func(t token.Token) bool { return t.Kind == token.Comment }); i >= 0 { body = body[:i] } if _, exists := pp.macros[name.Text]; exists { // The toolchain refuses redefinition, so a file the oracle accepts // never redefines; failing here keeps that contract visible. pp.errorf(name.Pos, "redefinition of macro %s", name.Text) } pp.macros[name.Text] = ¯oDef{name: name.Text, args: args, body: pp.bodyWithBreaks(body)} } // bodyWithBreaks records the statement boundaries the continuations carry. // The lexer splices backslash-continued lines into one logical line, but the // toolchain keeps the newline as a token in the stored body, which is how a // multi-instruction body without semicolons (the arm64 style) still splits // into statements on expansion. A line change inside the logical line is // exactly a continuation, so the boundary is restored from the positions. func (pp *preproc) bodyWithBreaks(body []token.Token) []token.Token { out := make([]token.Token, 0, len(body)) for i, t := range body { if i > 0 && t.Pos.Line != body[i-1].Pos.Line { out = append(out, token.Token{Kind: token.Newline, Text: "\n", Pos: t.Pos, End: t.Pos}) } out = append(out, t) } return out } // undef handles "#undef NAME", which the toolchain honours and requires to // name a defined macro. func (pp *preproc) undef(line []token.Token) { if len(line) < 3 || line[2].Kind != token.Ident { return } if _, ok := pp.macros[line[2].Text]; !ok { pp.errorf(line[2].Pos, "#undef for undefined macro %s", line[2].Text) return } delete(pp.macros, line[2].Text) } // include resolves and splices "#include \"file\"". A header that cannot be // read keeps the directive line in the stream, with a diagnostic. func (pp *preproc) include(line []token.Token, dir string) [][]token.Token { if len(line) < 3 || line[2].Kind != token.String { return [][]token.Token{line} } header := line[2] name, err := strconv.Unquote(header.Text) if err != nil { pp.errorf(header.Pos, "unquoting include file name: %v", err) return [][]token.Token{line} } if filepath.Base(name) == textflagHeader { // Flag macros are handled natively (see textflagHeader); the // directive stays so tools still see the include. return [][]token.Token{line} } resolved, ok := pp.resolve(name, dir) if !ok { searched := append([]string{dir}, pp.opts.IncludeDirs...) pp.errorf(header.Pos, "#include %q: file not found (searched %s)", name, strings.Join(searched, ", ")) return [][]token.Token{line} } src, err := os.ReadFile(resolved) if err != nil { pp.errorf(header.Pos, "#include %q: %v", name, err) return [][]token.Token{line} } return pp.fileLines(resolved, lexer.Tokenize(string(src)), header.Pos) } // resolve looks an include name up the way the toolchain does: as written // (relative to the working directory), then relative to the including // file's directory, then in each -I directory in order. func (pp *preproc) resolve(name, dir string) (string, bool) { candidates := []string{name} if !filepath.IsAbs(name) { candidates = append(candidates, filepath.Join(dir, name)) for _, d := range pp.opts.IncludeDirs { candidates = append(candidates, filepath.Join(d, name)) } } for _, c := range candidates { if st, err := os.Stat(c); err == nil && !st.IsDir() { return c, true } } return "", false } // expandTokens expands every macro invocation in a token sequence, // recursively, with a depth guard. A body is spliced into the sequence in // place and rescanned, the way the toolchain's input stack re-reads pushed // tokens: an object macro may name a parameterised one, and the argument // list of the expansion may then come from the tokens that follow. func (pp *preproc) expandTokens(in []token.Token) []token.Token { s := in i := 0 consecutive := 0 for i < len(s) { t := s[i] if t.Kind != token.Ident { i++ consecutive = 0 continue } def, suffix := pp.macroFor(t.Text) if def == nil { i++ consecutive = 0 continue } // The guard mirrors the toolchain's: 100 nested invocations in a // row without a plain token between them means recursion. consecutive++ if consecutive > maxExpansionDepth { pp.errorf(t.Pos, "recursive macro invocation (deeper than %d levels)", maxExpansionDepth) return nil } if def.args == nil { body := restamp(def.body, t.Pos) if suffix != "" { // The macro was reached only through a compound spelling // (ACC0.B16 over "#define ACC0 V8"), so the selector has // to travel with the expansion. body = appendSelector(body, suffix, t.Pos) } s = append(s[:i], append(body, s[i+1:]...)...) continue } // A parameterised macro invoked without its parentheses stands // unexpanded, naming itself, as in the toolchain. if i+1 >= len(s) || s[i+1].Kind != token.LParen { i++ consecutive = 0 continue } args, next := pp.collectArgs(s, i+1, t) if args == nil { return nil } // A zero-argument macro may be invoked as NAME(). if len(def.args) == 0 && len(args) == 1 && len(args[0]) == 0 { args = nil } if len(args) != len(def.args) { pp.errorf(t.Pos, "wrong arg count for macro %s: got %d, want %d", t.Text, len(args), len(def.args)) i = next consecutive = 0 continue } sub := make([]token.Token, 0, len(def.body)) for _, bt := range def.body { if bt.Kind == token.Ident { if k := slices.Index(def.args, bt.Text); k >= 0 { sub = append(sub, restamp(args[k], t.Pos)...) continue } // A parameter used with an element or lane selector: the // lexer folds A.S4 into one identifier, so the whole-token // match above cannot see the parameter. The toolchain // lexes the period separately and substitutes the name // alone; splitting at the FIRST period and pasting the // argument back in front of the selector is the equivalent // for this lexer. if k, sel := parameterSelector(bt.Text, def.args); k >= 0 { sub = append(sub, restamp(pasteSelector(args[k], sel), t.Pos)...) continue } } sub = append(sub, bt) } s = append(s[:i], append(sub, s[next:]...)...) } return s } // macroFor finds the macro a use names. The lexer folds NAME.selector into // one identifier token, so a macro written behind a selector suffix // (ACC0.B16 over "#define ACC0 V8") never matches a whole-token table // lookup; the toolchain splits on the period and reads the two halves, so // the prefix before the FIRST period is tried here as well and the caller // re-attaches the suffix to whatever the macro expands to. Only a whole // name counts: AB.S4 does not reach a macro named A, and a parameterised // macro is not hidden behind a selector, because its invocation would need // the parentheses to follow the bare name. func (pp *preproc) macroFor(text string) (*macroDef, string) { if def := pp.macros[text]; def != nil { return def, "" } if j := strings.IndexByte(text, '.'); j > 0 { if def := pp.macros[text[:j]]; def != nil && def.args == nil { return def, text[j:] } } return nil, "" } // appendSelector glues a selector suffix onto an object macro's expansion: // the selector binds to the identifier the expansion ends with, the way the // toolchain's operand parser reads V0 and .B16 back as one register // spelling. An expansion that does not end in an identifier carries the // selector as its own token, which the parser then reports where it cannot // parse it. func appendSelector(body []token.Token, suffix string, pos token.Position) []token.Token { if n := len(body); n > 0 && body[n-1].Kind == token.Ident { body[n-1].Text += suffix return body } return append(body, token.Token{Kind: token.Ident, Text: suffix, Pos: pos, End: pos}) } // parameterSelector reports the argument a compound body token names: the // parameter whose whole name occupies the text before the token's FIRST // period, with the selector that follows. k is negative when no parameter // matches, which leaves tokens like AB.S4 untouched even though a parameter // A is bound. func parameterSelector(text string, args []string) (int, string) { j := strings.IndexByte(text, '.') if j <= 0 { return -1, "" } if k := slices.Index(args, text[:j]); k >= 0 { return k, text[j:] } return -1, "" } // pasteSelector joins an argument with the selector a compound body token // carries, textually: the selector binds to the identifier the argument // ends with, so A.S4 over the argument V0.B16 spells V0.B16.S4, exactly the // operand the toolchain's split-then-substitute leaves behind. An argument // with no trailing identifier carries the selector as a separate token, // which the parser then reports where it cannot parse it. func pasteSelector(val []token.Token, suffix string) []token.Token { if len(val) == 0 { return []token.Token{{Kind: token.Ident, Text: suffix}} } out := slices.Clone(val) if n := len(out); out[n-1].Kind == token.Ident { out[n-1].Text += suffix return out } return append(out, token.Token{Kind: token.Ident, Text: suffix}) } // collectArgs reads the actual argument tokens of an invocation; the opening // parenthesis is at start. Commas separate arguments except inside nested // parentheses. A nil result means the list was unterminated, which is a // diagnostic. func (pp *preproc) collectArgs(in []token.Token, start int, name token.Token) ([][]token.Token, int) { var args [][]token.Token var cur []token.Token nesting := 0 for i := start + 1; i < len(in); i++ { t := in[i] switch t.Kind { case token.LParen: nesting++ cur = append(cur, t) case token.RParen: if nesting == 0 { return append(args, cur), i + 1 } nesting-- cur = append(cur, t) case token.Comma: if nesting == 0 { args = append(args, cur) cur = nil continue } cur = append(cur, t) case token.Comment: pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text) return nil, i default: cur = append(cur, t) } } pp.errorf(name.Pos, "unterminated arg list invoking macro %s", name.Text) return nil, len(in) } // restamp copies body tokens to the invocation's position, so diagnostics // and the line table point where the macro was used, as the toolchain's // input stack does. func restamp(body []token.Token, pos token.Position) []token.Token { out := make([]token.Token, len(body)) for i, t := range body { t.Pos, t.End = pos, pos out[i] = t } return out } // splitOnSemicolons breaks a token sequence at ';' statement separators and // at the Newline markers that record continuation boundaries inside macro // bodies, producing the logical lines the parser expects. The separators // carry no meaning beyond the break, so the pieces are exactly what the same // statements on separate lines would produce. func splitOnSemicolons(ts []token.Token) [][]token.Token { var out [][]token.Token start := 0 for i, t := range ts { if t.Kind == token.Semicolon || t.Kind == token.Newline { if i > start { out = append(out, ts[start:i]) } start = i + 1 } } if start < len(ts) { out = append(out, ts[start:]) } return out }