// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause // Package parser turns a GAsm token stream into an abstract syntax tree. It // is line-oriented, matching how the Plan 9 assembler itself reads a file, and // tolerant: a malformed line is reported as an error but never aborts the // parse of the rest of the file. package parser import ( "fmt" "math" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" "sourcedock.dev/petrbalvin/gasm-devkit/lexer" "sourcedock.dev/petrbalvin/gasm-devkit/token" ) // Error is a single parse diagnostic. type Error struct { Pos token.Position Msg string } func (e Error) Error() string { return e.Pos.String() + ": " + e.Msg } // Parse scans and parses src, returning the file and any diagnostics. The // returned file is usable even when errors is non-empty. func Parse(path, src string) (*ast.File, []error) { tokens := lexer.Tokenize(src) p := &state{path: path} p.parse(statementLines(tokens)) return p.file, p.errs } // state carries the mutable context for one parse. type state struct { path string file *ast.File errs []error curText *ast.Text // the TEXT body labels/instructions attach to pending []string // comment lines awaiting a TEXT to become its Doc } func (p *state) errorf(pos token.Position, format string, args ...any) { p.errs = append(p.errs, Error{Pos: pos, Msg: fmt.Sprintf(format, args...)}) } // splitLines groups the token stream into lines, dropping the Newline tokens. func splitLines(tokens []token.Token) [][]token.Token { var lines [][]token.Token var cur []token.Token for _, t := range tokens { if t.Kind == token.EOF { break } if t.Kind == token.Newline { lines = append(lines, cur) cur = nil continue } cur = append(cur, t) } if len(cur) > 0 { lines = append(lines, cur) } return lines } // statementLines turns the token stream into the logical lines the parser // reads: physical lines split at the ';' statement separators, exactly the // way the expansion path treats the expanded bodies. The runtime writes // "ROLQ $3, DI; ROLQ $13, DI" and "REP; MOVSB" in plain files, and the // separator carries no meaning beyond the break. Comments are statement // text, not structure: the lexer delivers a whole comment as one token, so // a ';' inside a comment is never a separator; a comment after a statement // stays on that statement's line; and a comment that sits between // statements (the runtime's "NO_LOCAL_POINTERS; /* … */" style) stands as // its own logical line, like a whole-line comment. func statementLines(tokens []token.Token) [][]token.Token { var out [][]token.Token var cur []token.Token flush := func() { if len(cur) > 0 { out = append(out, cur) cur = nil } } for _, t := range tokens { switch t.Kind { case token.EOF: // The stream's terminator is not statement content. case token.Newline, token.Semicolon: flush() case token.Comment: if len(cur) > 0 { cur = append(cur, t) } else { out = append(out, []token.Token{t}) } flush() default: cur = append(cur, t) } } flush() return out } func (p *state) parse(lines [][]token.Token) { p.file = &ast.File{Path: p.path, Macros: map[string]bool{}} for _, line := range lines { if len(line) == 0 { // Blank line: a comment block ends here only if it was not // directly preceding a declaration; keep pending doc intact // across a single blank line is not desired, so reset. p.pending = nil continue } p.parseLine(line) } } func (p *state) parseLine(line []token.Token) { first := line[0] // A lone comment accumulates as documentation for a following TEXT. if len(line) == 1 && first.Kind == token.Comment { p.pending = append(p.pending, commentText(first.Text)) return } // Preprocessor line. if first.Kind == token.Hash { p.parsePreproc(line) p.pending = nil return } // Directives and instructions are identified by a leading identifier. if first.Kind == token.Ident { switch first.Text { case "TEXT": p.parseText(line) return case "GLOBL": p.file.Decls = append(p.file.Decls, p.parseGlobl(line)) p.curText = nil p.pending = nil return case "DATA": p.file.Decls = append(p.file.Decls, p.parseData(line)) p.curText = nil p.pending = nil return } // Label (ident immediately followed by a colon). if len(line) >= 2 && line[1].Kind == token.Colon { lbl := &ast.Label{Name: line[0], Colon: line[1]} p.addStmt(lbl) // A label may share its line with an instruction: "loop: MOVQ …". if rest := dropColon(line); len(rest) > 0 { p.parseInstr(rest) } p.pending = nil return } // Otherwise it is an instruction. p.parseInstr(line) p.pending = nil return } p.errorf(first.Pos, "unexpected token %s at start of line", first.Kind) p.pending = nil } // dropColon removes the leading "ident :" of a label, returning the remainder. func dropColon(line []token.Token) []token.Token { if len(line) >= 2 && line[1].Kind == token.Colon { return line[2:] } return nil } func (p *state) addStmt(s ast.Stmt) { if p.curText != nil { p.curText.Body = append(p.curText.Body, s) return } p.file.Orphans = append(p.file.Orphans, s) } func (p *state) parsePreproc(line []token.Token) { hash := line[0] if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "include" && line[2].Kind == token.String { p.file.Decls = append(p.file.Decls, &ast.Include{ Hash: hash, Name: line[1], Header: line[2], }) return } // Record macro names so the linter can recognise their invocations. if len(line) >= 3 && line[1].Kind == token.Ident && line[1].Text == "define" && line[2].Kind == token.Ident { p.file.Macros[line[2].Text] = true } p.file.Decls = append(p.file.Decls, &ast.Preproc{ Hash: hash, Raw: joinRaw(line[1:]), }) } func (p *state) parseText(line []token.Token) { text := &ast.Text{Keyword: line[0]} if len(p.pending) > 0 { text.Doc = strings.Join(p.pending, "\n") } p.pending = nil rest := line[1:] sym, n := parseSymbolPrefix(rest) if sym == nil { p.errorf(line[0].Pos, "TEXT missing a symbol name") // Keep the decl in the tree with a placeholder name: the linter and // LSP dereference Name on every parsed TEXT, so the file must stay // usable alongside its errors. sym = &ast.Symbol{Pos: line[0].Pos, Raw: "?", Name: "?"} } text.Name = sym rest = rest[n:] // Consume flags (identifiers, possibly '|' joined) up to the frame '$'. rest = skipComma(rest) for len(rest) > 0 && rest[0].Kind != token.Dollar { if rest[0].Kind == token.Ident { text.Flags = append(text.Flags, rest[0].Text) } // Commas, '|' and anything else between flags is skipped. rest = rest[1:] } // Frame: $[-]number ; optional args: -number. The Go runtime writes // zero frames with an explicit sign ("$-0-24"), so the number may carry // one. Whatever remains after the header is the body and is parsed by // the caller. if len(rest) > 0 && rest[0].Kind == token.Dollar { n := 1 neg := false if n < len(rest) && (rest[n].Kind == token.Minus || rest[n].Kind == token.Plus) { neg = rest[n].Kind == token.Minus n++ } if n < len(rest) && rest[n].Kind == token.Number { val := parseInt(rest[n].Text) if neg { val = -val } text.Frame = &ast.Operand{ Kind: ast.OpImmediate, Imm: ast.Immediate{Val: val, HasVal: true}, Raw: joinRaw(rest[:n+1]), Pos: rest[0].Pos, } // The argument area: a minus sign followed by a number. if n+2 < len(rest) && rest[n+1].Kind == token.Minus && rest[n+2].Kind == token.Number { text.Args = &ast.Operand{ Kind: ast.OpImmediate, Imm: ast.Immediate{Val: parseInt(rest[n+2].Text), HasVal: true}, Raw: "-" + rest[n+2].Text, Pos: rest[n+1].Pos, } } } else { p.errorf(rest[0].Pos, "TEXT frame size must be a number after $") } } p.file.Decls = append(p.file.Decls, text) p.curText = text } func (p *state) parseGlobl(line []token.Token) *ast.Globl { g := &ast.Globl{Keyword: line[0]} rest := skipComma(line[1:]) sym, n := parseSymbolPrefix(rest) g.Name = sym rest = skipComma(rest[n:]) // Flags are identifiers (RODATA, DUPOK) or legacy numeric constants // (2, 8, 9, 10) from runtime/textflag.h. for len(rest) > 0 && rest[0].Kind != token.Dollar { if rest[0].Kind == token.Ident || rest[0].Kind == token.Number { g.Flags = append(g.Flags, rest[0].Text) } rest = rest[1:] } if len(rest) > 0 && rest[0].Kind == token.Dollar { g.Size = parseOperand(rest, false) } return g } func (p *state) parseData(line []token.Token) *ast.Data { d := &ast.Data{Keyword: line[0]} rest := line[1:] // The name may carry a /width suffix: ·idx+0(SB)/4. Split it off the // symbol group that precedes the first top-level comma. nameGroup, valuePart := splitFirstComma(rest) nameGroup, width := splitTrailingWidth(nameGroup) sym, _ := parseSymbolPrefix(nameGroup) d.Name = sym d.Width = width if len(valuePart) > 0 { d.Value = parseOperand(stripComment(valuePart), false) } return d } func (p *state) parseInstr(line []token.Token) { body, comment := splitTrailingComment(line) if len(body) == 0 { return } instr := &ast.Instr{Mnemonic: body[0], Comment: comment} grps := splitOperands(body[1:]) for i, grp := range grps { // Only the final operand slot may carry a bare constant: the // toolchain reads the trailing 1 of CMPSD X1, X0, 1 as $1 // (math/floor_amd64.s), while an earlier bare number names an // absolute address, a form this parser keeps out of the tree. if op := parseOperand(grp, i == len(grps)-1); op != nil { instr.Operands = append(instr.Operands, op) } } p.addStmt(instr) } // --- symbol parsing --------------------------------------------------------- var pseudoRegs = map[string]bool{"FP": true, "SP": true, "SB": true, "PC": true} // parseSymbolPrefix parses a leading symbol reference from g and returns it // together with the number of tokens consumed. It returns (nil, 0) when no // symbol is present. Raw is the verbatim spelling, with the bracket group, // offset and pseudo-register glued to the name the way the assembler writes // the reference. func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) { if len(g) == 0 || g[0].Kind != token.Ident { return nil, 0 } sym := &ast.Symbol{Pos: g[0].Pos} i := 0 setName(g[0].Text, sym) i++ var raw strings.Builder raw.WriteString(g[0].Text) // An optional bracket group after the name: <> marks a file-static symbol // and selects the ABI of the reference. The ABI form is the // standard runtime spelling (TEXT ·foo(SB)) and must be // consumed here, or it leaks into the directive's flag list. if i < len(g) && g[i].Kind == token.LAngle { switch { case i+1 < len(g) && g[i+1].Kind == token.RAngle: sym.Static = true raw.WriteString("<>") i += 2 case i+2 < len(g) && g[i+1].Kind == token.Ident && g[i+2].Kind == token.RAngle: sym.ABI = g[i+1].Text raw.WriteString("<" + sym.ABI + ">") i += 3 } } if i < len(g) && (g[i].Kind == token.Plus || g[i].Kind == token.Minus) { neg := g[i].Kind == token.Minus raw.WriteString(g[i].Text) i++ if i < len(g) && g[i].Kind == token.Number { sym.Offset, sym.HasOff = parseInt(g[i].Text), true if neg { sym.Offset = -sym.Offset } raw.WriteString(g[i].Text) i++ } } if i+2 < len(g) && g[i].Kind == token.LParen && g[i+1].Kind == token.Ident && pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen { sym.Pseudo = g[i+1].Text raw.WriteString("(" + sym.Pseudo + ")") i += 3 } sym.Raw = raw.String() return sym, i } // setName splits a raw identifier on the middle dot into package and name. func setName(raw string, sym *ast.Symbol) { const dot = "\u00B7" switch { case strings.HasPrefix(raw, dot): sym.Pkg = "" sym.Name = strings.TrimPrefix(raw, dot) case strings.Contains(raw, dot): parts := strings.SplitN(raw, dot, 2) sym.Pkg = parts[0] sym.Name = parts[1] default: sym.Name = raw } } // --- operand parsing -------------------------------------------------------- // parseOperand parses one operand group into an Operand. allowBare marks // the final operand slot of an instruction, where the toolchain reads a // bare constant expression as an immediate. func parseOperand(g []token.Token, allowBare bool) *ast.Operand { g = stripComment(g) if len(g) == 0 { return nil } op := &ast.Operand{Raw: joinRaw(g), Pos: g[0].Pos} if g[0].Kind == token.Dollar { op.Kind = ast.OpImmediate op.Imm = parseImmediate(g[1:]) return op } op.Kind = ast.OpAddr op.Addr = parseAddress(g) // A trailing bare constant leaves every address field empty: the // grammar sees no register, memory reference or symbol, and the closed // constant expression is the whole group. Read it as the immediate it // names, exactly what the $ spelling would produce. if allowBare && isEmptyAddress(op.Addr) { if v, rest, ok := foldExpr(g); ok && len(rest) == 0 { op.Kind = ast.OpImmediate op.Imm = ast.Immediate{Val: v, HasVal: true} } } return op } // isEmptyAddress reports whether parseAddress populated nothing, its sign // that the group is no register, memory reference, symbol or register range. func isEmptyAddress(a ast.Address) bool { return a.Sym == nil && a.Base == "" && a.Index == "" && a.Range == nil && a.Shift == "" } // parseImmediate parses the tokens following a '$'. func parseImmediate(g []token.Token) ast.Immediate { var imm ast.Immediate if len(g) == 0 { return imm } // $sym(…) form. if findPseudoParen(g) >= 0 || (g[0].Kind == token.Ident) { if sym, n := parseSymbolPrefix(g); n > 0 && (sym.Pseudo != "" || sym.Static) { imm.Sym = sym return imm } } // A constant expression introduced by '(' or '~'. Textual macro // substitution leaves arithmetic such as $(32-shift) and $~63 behind, // and the toolchain evaluates it in place; only shapes the ordinary // paths below cannot read reach the folder, so every existing form // keeps its exact parse. if g[0].Kind == token.LParen || g[0].Kind == token.Tilde { if v, rest, ok := foldExpr(g); ok && len(rest) == 0 { imm.Val = v imm.HasVal = true return imm } } i := 0 if g[i].Kind == token.Minus { imm.Neg = true i++ } else if g[i].Kind == token.Plus { i++ } // A constant expression after the sign: $-(R - 8), $+(32-shift). The // toolchain folds the negated value in place (the cgo ABI macros write // ADJSP $-(REGS_HOST_TO_ABI0_STACK - 8)), so the sign applies to the // folded value exactly as it does to a bare literal. if i < len(g) && (g[i].Kind == token.LParen || g[i].Kind == token.Tilde) { if v, rest, ok := foldExpr(g[i:]); ok && len(rest) == 0 { imm.Val = v imm.HasVal = true return imm } } if i < len(g) && g[i].Kind == token.Number { text := g[i].Text if v, ok := tryInt(text); ok { imm.Val = v imm.HasVal = true } else if u, err := strconv.ParseUint(text, 0, 64); err == nil && (!imm.Neg || u == 1<<63) { // Unsigned 64-bit literals (DATA mask<>+8(SB)/8, $0x8000…) can // overflow int64; keep the bit pattern. A negated magnitude of // exactly 1<<63 is the int64 minimum: ParseInt rejects it, but // the value is representable, so Val carries it exactly. if imm.Neg { imm.Val = math.MinInt64 } else { imm.Val = int64(u) } imm.HasVal = true } else { imm.Float = text } } else if i < len(g) && (g[i].Kind == token.String || g[i].Kind == token.Rune) { imm.Str = g[i].Text } return imm } // parseAddress parses a non-immediate operand. func parseAddress(g []token.Token) ast.Address { var addr ast.Address if len(g) == 0 { return addr } // A bracketed register range, [Z0-Z3]: the amd64 4FMAPS/4VNNIW // multi-source operand. The bracket runes arrive as Illegal tokens // (the lexer has no bracket kind), so the shape matches on their text. if isBracket(g[0], "[") && len(g) == 5 && g[1].Kind == token.Ident && g[2].Kind == token.Minus && g[3].Kind == token.Ident && isBracket(g[4], "]") { addr.Range = &ast.RegRange{Lo: g[1].Text, Hi: g[3].Text, Pos: g[0].Pos} return addr } // Symbol-with-pseudo form: name[<>][+off](PSEUDO). // When the prefix is not a valid symbol name (e.g. a bare number like // 0(SP) in RISC-V), sym is nil, and we fall through to regular memory // operand parsing instead of returning an empty address. if idx := findPseudoParen(g); idx >= 0 { sym, _ := parseSymbolPrefix(g[:idx+3]) if sym != nil { addr.Sym = sym return addr } } i := 0 // A parenthesised constant expression as the displacement: substituted // macro bodies carry ((index*4)+0)(base) shapes. As with the signed // number path below, the value is committed only when a base group // follows. if i < len(g) && g[i].Kind == token.LParen { if v, rest, ok := foldExpr(g[i:]); ok && len(rest) > 0 && rest[0].Kind == token.LParen { addr.Offset = v addr.HasOff = true i = len(g) - len(rest) } } // The same expression under a leading sign: -(24+8)(X6) puts the sign // outside the fold. The base group must follow for the value to // commit, exactly as in the unsigned branch above. if i < len(g) && (g[i].Kind == token.Minus || g[i].Kind == token.Plus) && i+1 < len(g) && g[i+1].Kind == token.LParen { if v, rest, ok := foldExpr(g[i+1:]); ok && len(rest) > 0 && rest[0].Kind == token.LParen { if g[i].Kind == token.Minus { v = -v } addr.Offset = v addr.HasOff = true i = len(g) - len(rest) } } // Optional leading displacement before a '(' base group. A sign pushes // the parenthesis one token further out: -4(DX) has it at i+2. if isSignedNumber(g, i) { j := i neg := false if g[j].Kind == token.Minus { neg = true j++ } else if g[j].Kind == token.Plus { j++ } if j < len(g) && g[j].Kind == token.Number { v := parseInt(g[j].Text) j++ // A term may carry a *number factor: 0*8(base). for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number { v *= parseInt(g[j+1].Text) j += 2 } if neg { v = -v } // Further +/- terms, each with its optional factor: // 3*8+8(base), 8-4*2(base). for { termNeg := false if j < len(g) && g[j].Kind == token.Minus { termNeg = true } else if j < len(g) && g[j].Kind == token.Plus { } else { break } if j+1 < len(g) && g[j+1].Kind == token.Number { tv := parseInt(g[j+1].Text) j += 2 for j+1 < len(g) && g[j].Kind == token.Star && g[j+1].Kind == token.Number { tv *= parseInt(g[j+1].Text) j += 2 } if termNeg { tv = -tv } v += tv continue } break } // Commit only when the expression is followed by the base // group; a bare number stays untouched for the caller. if j < len(g) && g[j].Kind == token.LParen { addr.Offset = v addr.HasOff = true i = j } } } // A lone (index*scale) group is the VSIB index-only form: the // gather/scatter families address memory through a scaled vector index // with no base register, 8(X4*1). The two-group grammar below reads // (base)(index*scale), so a first group whose member carries a scale // factor can only be an index. if isIndexGroup(g[i:]) { addr.Index = g[i+1].Text addr.Scale = int(parseInt(g[i+3].Text)) i += 5 } // First parenthesised group: the base register. if i < len(g) && g[i].Kind == token.LParen { i++ if i < len(g) && g[i].Kind == token.Ident { addr.Base = g[i].Text i++ } if i < len(g) && g[i].Kind == token.RParen { i++ } } // Optional second group: (index*scale) or (index). if i < len(g) && g[i].Kind == token.LParen { i++ if i < len(g) && g[i].Kind == token.Ident { addr.Index = g[i].Text i++ } if i < len(g) && g[i].Kind == token.Star { i++ if i < len(g) && g[i].Kind == token.Number { addr.Scale = int(parseInt(g[i].Text)) i++ } } if i < len(g) && g[i].Kind == token.RParen { i++ } } // Bare name (register, label or symbol) possibly with an arm64 shift. if addr.Base == "" && addr.Sym == nil && g[0].Kind == token.Ident { sym := &ast.Symbol{Pos: g[0].Pos} setName(g[0].Text, sym) sym.Raw = g[0].Text addr.Sym = sym i = 1 } // Any remaining tokens form a verbatim shift/extension suffix (arm64). if i > 0 && i < len(g) { addr.Shift = joinRaw(g[i:]) } return addr } // findPseudoParen returns the index of the '(' that begins a (PSEUDO) group, // or -1 when none is present. func findPseudoParen(g []token.Token) int { for i := 0; i+2 < len(g); i++ { if g[i].Kind == token.LParen && g[i+1].Kind == token.Ident && pseudoRegs[g[i+1].Text] && g[i+2].Kind == token.RParen { return i } } return -1 } // isBracket reports whether t is a square bracket. The lexer has no bracket // kind, so '[' and ']' arrive as Illegal tokens. func isBracket(t token.Token, text string) bool { return t.Kind == token.Illegal && t.Text == text } // isIndexGroup reports whether g begins with a complete (index*scale) group: // one identifier followed by a scale factor, all inside a single parenthesis. func isIndexGroup(g []token.Token) bool { return len(g) >= 5 && g[0].Kind == token.LParen && g[1].Kind == token.Ident && g[2].Kind == token.Star && g[3].Kind == token.Number && g[4].Kind == token.RParen } // --- token helpers ---------------------------------------------------------- // splitOperands splits a token slice on top-level commas (commas outside any // parenthesis group). func splitOperands(g []token.Token) [][]token.Token { var out [][]token.Token var cur []token.Token depth := 0 for _, t := range g { switch t.Kind { case token.LParen: depth++ cur = append(cur, t) case token.RParen: depth-- cur = append(cur, t) case token.Comma: if depth == 0 { if len(cur) > 0 { out = append(out, cur) } cur = nil } else { cur = append(cur, t) } case token.Comment: // A comment terminates the operand list. if len(cur) > 0 { out = append(out, cur) } return out default: cur = append(cur, t) } } if len(cur) > 0 { out = append(out, cur) } return out } // splitFirstComma splits g at the first top-level comma. func splitFirstComma(g []token.Token) (before, after []token.Token) { depth := 0 for i, t := range g { switch t.Kind { case token.LParen: depth++ case token.RParen: depth-- case token.Comma: if depth == 0 { return g[:i], g[i+1:] } default: // Every other token kind is inert at the top level of the // group; the scan just keeps looking for the first comma. } } return g, nil } // splitTrailingComment separates a trailing comment from the line body. func splitTrailingComment(g []token.Token) (body []token.Token, comment string) { for i, t := range g { if t.Kind == token.Comment { return g[:i], commentText(t.Text) } } return g, "" } // stripComment removes a trailing comment token from a group. func stripComment(g []token.Token) []token.Token { for i, t := range g { if t.Kind == token.Comment { return g[:i] } } return g } // splitTrailingWidth removes a "/width" suffix from a DATA name group. func splitTrailingWidth(g []token.Token) ([]token.Token, int) { for i := 0; i+1 < len(g); i++ { if g[i].Kind == token.Slash && g[i+1].Kind == token.Number { return g[:i], int(parseInt(g[i+1].Text)) } } return g, 0 } func skipComma(g []token.Token) []token.Token { if len(g) > 0 && g[0].Kind == token.Comma { return g[1:] } return g } func isSignedNumber(g []token.Token, i int) bool { if i >= len(g) { return false } if g[i].Kind == token.Number { return true } if (g[i].Kind == token.Minus || g[i].Kind == token.Plus) && i+1 < len(g) && g[i+1].Kind == token.Number { return true } return false } func joinRaw(g []token.Token) string { parts := make([]string, len(g)) for i, t := range g { parts[i] = t.Text } return strings.Join(parts, " ") } // commentText removes a leading // or /* marker from a comment token's text. func commentText(s string) string { if body, ok := strings.CutPrefix(s, "//"); ok { return strings.TrimSpace(body) } if body, ok := strings.CutPrefix(s, "/*"); ok { body = strings.TrimSuffix(body, "*/") return strings.TrimSpace(body) } return s } func parseInt(text string) int64 { v, _ := tryInt(text) return v } func tryInt(text string) (int64, bool) { v, err := strconv.ParseInt(text, 0, 64) if err != nil { return 0, false } return v, true }