// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // Package parser turns a GAsm token stream into an abstract syntax tree. It // is line-oriented, matching how the Plan 9 assembler itself reads a file, and // tolerant: a malformed line is reported as an error but never aborts the // parse of the rest of the file. package parser import ( "fmt" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/lexer" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) // Error is a single parse diagnostic. type Error struct { Pos token.Position Msg string } func (e Error) Error() string { return e.Pos.String() + ": " + e.Msg } // Parse scans and parses src, returning the file and any diagnostics. The // returned file is usable even when errors is non-empty. func Parse(path, src string) (*ast.File, []error) { toks := lexer.Tokenize(src) lines := splitLines(toks) p := &state{path: path} p.parse(lines) return p.file, p.errs } // state carries the mutable context for one parse. type state struct { path string file *ast.File errs []error curText *ast.Text // the TEXT body labels/instructions attach to pending []string // comment lines awaiting a TEXT to become its Doc } func (p *state) errorf(pos token.Position, format string, args ...any) { p.errs = append(p.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)}) } // splitLines groups the token stream into lines, dropping the Newline tokens. func splitLines(toks []token.Token) [][]token.Token { var lines [][]token.Token var cur []token.Token for _, t := range toks { if t.Kind == token.EOF { break } if t.Kind == token.Newline { lines = append(lines, cur) cur = nil continue } cur = append(cur, t) } if len(cur) > 0 { lines = append(lines, cur) } return lines } func (p *state) parse(lines [][]token.Token) { p.file = &ast.File{Path: p.path, Macros: map[string]bool{}} for _, line := range lines { line = trimSpace(line) if len(line) == 0 { // Blank line: a comment block ends here only if it was not // directly preceding a declaration; keep pending doc intact // across a single blank line is not desired, so reset. p.pending = nil continue } p.parseLine(line) } } // trimSpace is a no-op placeholder kept for symmetry; the lexer already drops // horizontal whitespace, but this documents the intent. func trimSpace(line []token.Token) []token.Token { return line } func (p *state) parseLine(line []token.Token) { first := line[0] // A lone comment accumulates as documentation for a following TEXT. if len(line) == 1 && first.Kind == token.Comment { p.pending = append(p.pending, commentText(first.Text)) return } // Preprocessor line. if first.Kind == token.Hash { p.parsePreproc(line) p.pending = nil return } // Directives and instructions are identified by a leading identifier. if first.Kind == token.Ident { switch first.Text { case "TEXT": p.parseText(line) return case "GLOBL": p.file.Decls = append(p.file.Decls, p.parseGlobl(line)) p.curText = nil p.pending = nil return case "DATA": p.file.Decls = append(p.file.Decls, p.parseData(line)) p.curText = nil p.pending = nil return } // Label (ident immediately followed by a colon). if len(line) >= 2 && line[1].Kind == token.Colon { lbl := &ast.Label{Name: line[0], Colon: line[1]} p.addStmt(lbl) // A label may share its line with an instruction: "loop: MOVQ …". if rest := dropColon(line); len(rest) > 0 { p.parseInstr(rest) } p.pending = nil return } // Otherwise it is an instruction. p.parseInstr(line) p.pending = nil return } p.errorf(first.Pos, "unexpected token %s at start of line", first.Kind) p.pending = nil } // dropColon removes the leading "ident :" of a label, returning the remainder. func dropColon(line []token.Token) []token.Token { if len(line) >= 2 && line[1].Kind == token.Colon { return line[2:] } return nil } func (p *state) addStmt(s ast.Stmt) { if p.curText != nil { p.curText.Body = append(p.curText.Body, s) return } p.file.Orphans = append(p.file.Orphans, s) } func (p *state) parsePreproc(line []token.Token) { hash := line[0] if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" && line[2].Kind == token.String { p.file.Decls = append(p.file.Decls, &ast.Include{ Hash: hash, Name: line[1], Header: line[2], }) return } // Record macro names so the linter can recognise their invocations. if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "define" && line[2].Kind == token.Ident { p.file.Macros[line[2].Text] = true } p.file.Decls = append(p.file.Decls, &ast.Preproc{ Hash: hash, Raw: joinRaw(line[1:]), }) } func (p *state) parseText(line []token.Token) { text := &ast.Text{Keyword: line[0]} if len(p.pending) > 0 { text.Doc = strings.Join(p.pending, "\n") } p.pending = nil rest := line[1:] sym, n := parseSymbolPrefix(rest) if sym == nil { p.errorf(line[0].Pos, "TEXT missing a symbol name") } text.Name = sym rest = rest[n:] // Consume flags (identifiers, possibly '|' joined) up to the frame '$'. rest = skipComma(rest) for len(rest) > 0 && rest[0].Kind != token.Dollar { if rest[0].Kind == token.Ident { text.Flags = append(text.Flags, rest[0].Text) } // Commas, '|' (Illegal) and anything else between flags is skipped. rest = rest[1:] } // Frame: $number ; optional args: -number. if len(rest) > 0 && rest[0].Kind == token.Dollar { text.Frame = parseOperand(rest[:2]) // "$" "number" rest = rest[2:] if len(rest) >= 2 && rest[0].Kind == token.Minus && rest[1].Kind == token.Number { text.Args = &ast.Operand{ Kind: ast.OpImmediate, Imm: ast.Immediate{Val: parseInt(rest[1].Text), HasVal: true}, Raw: "-" + rest[1].Text, Pos: rest[0].Pos, } rest = rest[2:] } } p.file.Decls = append(p.file.Decls, text) p.curText = text } func (p *state) parseGlobl(line []token.Token) *ast.Globl { g := &ast.Globl{Keyword: line[0]} rest := skipComma(line[1:]) sym, n := parseSymbolPrefix(rest) g.Name = sym rest = skipComma(rest[n:]) for len(rest) > 0 && rest[0].Kind != token.Dollar { if rest[0].Kind == token.Ident { g.Flags = append(g.Flags, rest[0].Text) } rest = rest[1:] } if len(rest) > 0 && rest[0].Kind == token.Dollar { g.Size = parseOperand(rest) } return g } func (p *state) parseData(line []token.Token) *ast.Data { d := &ast.Data{Keyword: line[0]} rest := line[1:] // The name may carry a /width suffix: ·idx+0(SB)/4. Split it off the // symbol group that precedes the first top-level comma. nameGroup, valuePart := splitFirstComma(rest) nameGroup, width := splitTrailingWidth(nameGroup) sym, _ := parseSymbolPrefix(nameGroup) d.Name = sym d.Width = width if len(valuePart) > 0 { d.Value = parseOperand(stripComment(valuePart)) } return d } func (p *state) parseInstr(line []token.Token) { body, comment := splitTrailingComment(line) if len(body) == 0 { return } instr := &ast.Instr{Mnemonic: body[0], Comment: comment} for _, grp := range splitOperands(body[1:]) { if op := parseOperand(grp); op != nil { instr.Operands = append(instr.Operands, op) } } p.addStmt(instr) } // --- symbol parsing --------------------------------------------------------- var pseudoRegs = map[string]bool{"FP": true, "SP": true, "SB": true, "PC": true} // parseSymbolPrefix parses a leading symbol reference from g and returns it // together with the number of tokens consumed. It returns (nil, 0) when no // symbol is present. func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) { if len(g) == 0 || g[0].Kind != token.Ident { return nil, 0 } sym := &ast.Symbol{Pos: g[0].Pos} i := 0 setName(g[0].Text, sym) i++ if i+1 < len(g) && g[i].Kind == token.LAngle && g[i+1].Kind == token.RAngle { sym.Static = true i += 2 } if i < len(g) && g[i].Kind == token.Plus { i++ if i < len(g) && g[i].Kind == token.Number { sym.Offset, sym.HasOff = parseInt(g[i].Text), true i++ } } if i+2 < len(g) && g[i].Kind == token.LParen && g[i+1].Kind == token.Ident && pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen { sym.Pseudo = g[i+1].Text i += 3 } sym.Raw = joinRaw(g[:i]) return sym, i } // setName splits a raw identifier on the middle dot into package and name. func setName(raw string, sym *ast.Symbol) { const dot = "\u00B7" switch { case strings.HasPrefix(raw, dot): sym.Pkg = "" sym.Name = strings.TrimPrefix(raw, dot) case strings.Contains(raw, dot): parts := strings.SplitN(raw, dot, 2) sym.Pkg = parts[0] sym.Name = parts[1] default: sym.Name = raw } } // --- operand parsing -------------------------------------------------------- // parseOperand parses one operand group into an Operand. func parseOperand(g []token.Token) *ast.Operand { g = stripComment(g) if len(g) == 0 { return nil } op := &ast.Operand{Raw: joinRaw(g), Pos: g[0].Pos} if g[0].Kind == token.Dollar { op.Kind = ast.OpImmediate op.Imm = parseImmediate(g[1:]) return op } op.Kind = ast.OpAddr op.Addr = parseAddress(g) return op } // parseImmediate parses the tokens following a '$'. func parseImmediate(g []token.Token) ast.Immediate { var imm ast.Immediate if len(g) == 0 { return imm } // $sym(…) form. if findPseudoParen(g) >= 0 || (g[0].Kind == token.Ident) { if sym, n := parseSymbolPrefix(g); sym != nil && (sym.Pseudo != "" || sym.Static) { imm.Sym = sym _ = n return imm } } i := 0 if g[i].Kind == token.Minus { imm.Neg = true i++ } else if g[i].Kind == token.Plus { i++ } if i < len(g) && g[i].Kind == token.Number { text := g[i].Text if v, ok := tryInt(text); ok { imm.Val = v imm.HasVal = true } else if u, err := strconv.ParseUint(text, 0, 64); err == nil && !imm.Neg { // Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…) // overflow int64; keep the bit pattern. imm.Val = int64(u) imm.HasVal = true } else { imm.Float = text } i++ } else if i < len(g) && (g[i].Kind == token.String || g[i].Kind == token.Rune) { imm.Str = g[i].Text i++ } return imm } // parseAddress parses a non-immediate operand. func parseAddress(g []token.Token) ast.Address { var addr ast.Address if len(g) == 0 { return addr } // Symbol-with-pseudo form: name[<>][+off](PSEUDO). // When the prefix is not a valid symbol name (e.g. a bare number like // 0(SP) in RISC-V), sym is nil and we fall through to regular memory // operand parsing instead of returning an empty address. if idx := findPseudoParen(g); idx >= 0 { sym, _ := parseSymbolPrefix(g[:idx+3]) if sym != nil { addr.Sym = sym return addr } } i := 0 // Optional leading displacement before a '(' base group. A sign pushes // the parenthesis one token further out: -4(DX) has it at i+2. if isSignedNumber(g, i) { paren := i + 1 if g[i].Kind == token.Minus || g[i].Kind == token.Plus { paren = i + 2 } if paren < len(g) && g[paren].Kind == token.LParen { neg := false if g[i].Kind == token.Minus { neg = true i++ } else if g[i].Kind == token.Plus { i++ } if i < len(g) && g[i].Kind == token.Number { addr.Offset = parseInt(g[i].Text) addr.HasOff = true if neg { addr.Offset = -addr.Offset } i++ } } } // First parenthesised group: the base register. if i < len(g) && g[i].Kind == token.LParen { i++ if i < len(g) && g[i].Kind == token.Ident { addr.Base = g[i].Text i++ } if i < len(g) && g[i].Kind == token.RParen { i++ } } // Optional second group: (index*scale) or (index). if i < len(g) && g[i].Kind == token.LParen { i++ if i < len(g) && g[i].Kind == token.Ident { addr.Index = g[i].Text i++ } if i < len(g) && g[i].Kind == token.Star { i++ if i < len(g) && g[i].Kind == token.Number { addr.Scale = int(parseInt(g[i].Text)) i++ } } if i < len(g) && g[i].Kind == token.RParen { i++ } } // Bare name (register, label or symbol) possibly with an arm64 shift. if addr.Base == "" && addr.Sym == nil && g[0].Kind == token.Ident { sym := &ast.Symbol{Pos: g[0].Pos} setName(g[0].Text, sym) sym.Raw = g[0].Text addr.Sym = sym i = 1 } // Any remaining tokens form a verbatim shift/extension suffix (arm64). if i > 0 && i < len(g) { addr.Shift = joinRaw(g[i:]) } return addr } // findPseudoParen returns the index of the '(' that begins a (PSEUDO) group, // or -1 when none is present. func findPseudoParen(g []token.Token) int { for i := 0; i+2 < len(g); i++ { if g[i].Kind == token.LParen && g[i+1].Kind == token.Ident && pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen { return i } } return -1 } // --- token helpers ---------------------------------------------------------- // splitOperands splits a token slice on top-level commas (commas outside any // parenthesis group). func splitOperands(g []token.Token) [][]token.Token { var out [][]token.Token var cur []token.Token depth := 0 for _, t := range g { switch t.Kind { case token.LParen: depth++ cur = append(cur, t) case token.RParen: depth-- cur = append(cur, t) case token.Comma: if depth == 0 { if len(cur) > 0 { out = append(out, cur) } cur = nil } else { cur = append(cur, t) } case token.Comment: // A comment terminates the operand list. if len(cur) > 0 { out = append(out, cur) } return out default: cur = append(cur, t) } } if len(cur) > 0 { out = append(out, cur) } return out } // splitFirstComma splits g at the first top-level comma. func splitFirstComma(g []token.Token) (before, after []token.Token) { depth := 0 for i, t := range g { switch t.Kind { case token.LParen: depth++ case token.RParen: depth-- case token.Comma: if depth == 0 { return g[:i], g[i+1:] } } } return g, nil } // splitTrailingComment separates a trailing comment from the line body. func splitTrailingComment(g []token.Token) (body []token.Token, comment string) { for i, t := range g { if t.Kind == token.Comment { return g[:i], commentText(t.Text) } } return g, "" } // stripComment removes a trailing comment token from a group. func stripComment(g []token.Token) []token.Token { for i, t := range g { if t.Kind == token.Comment { return g[:i] } } return g } // splitTrailingWidth removes a "/width" suffix from a DATA name group. func splitTrailingWidth(g []token.Token) ([]token.Token, int) { for i := 0; i+1 < len(g); i++ { if g[i].Kind == token.Slash && g[i+1].Kind == token.Number { return g[:i], int(parseInt(g[i+1].Text)) } } return g, 0 } func skipComma(g []token.Token) []token.Token { if len(g) > 0 && g[0].Kind == token.Comma { return g[1:] } return g } func isSignedNumber(g []token.Token, i int) bool { if i >= len(g) { return false } if g[i].Kind == token.Number { return true } if (g[i].Kind == token.Minus || g[i].Kind == token.Plus) && i+1 < len(g) && g[i+1].Kind == token.Number { return true } return false } func joinRaw(g []token.Token) string { parts := make([]string, len(g)) for i, t := range g { parts[i] = t.Text } return strings.Join(parts, " ") } // commentText removes a leading // or /* marker from a comment token's text. func commentText(s string) string { if strings.HasPrefix(s, "//") { return strings.TrimSpace(strings.TrimPrefix(s, "//")) } if strings.HasPrefix(s, "/*") { s = strings.TrimPrefix(s, "/*") s = strings.TrimSuffix(s, "*/") return strings.TrimSpace(s) } return s } func parseInt(text string) int64 { v, _ := tryInt(text) return v } func tryInt(text string) (int64, bool) { v, err := strconv.ParseInt(text, 0, 64) if err != nil { return 0, false } return v, true }