// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: MIT package mathml import "strings" // parser walks the token list. The variant field is the mathvariant the // current style command imposes on the atoms it covers. type parser struct { src []byte toks []token pos int display bool variant string macros map[string]macro expansions int } func newParser(src []byte, display bool) *parser { return &parser{src: src, toks: tokenise(src), display: display, macros: map[string]macro{}} } func (p *parser) cur() token { p.expandMacros() return p.toks[p.pos] } func (p *parser) at(kind tokenKind) bool { return p.cur().kind == kind } func (p *parser) atCommand(name string) bool { t := p.cur() return t.kind == tokCommand && t.text == `\`+name } // parseAll parses the whole source into nodes. A boundary with nothing to // bound degrades as the stray it is. func (p *parser) parseAll() []*node { nodes := p.sequence(true) for p.cur().kind != tokEOF { if isBoundary(p.cur()) { t := p.cur() p.pos++ nodes = append(nodes, errorNode(t.text)) continue } nodes = append(nodes, p.sequence(true)...) } return nodes } // isBoundary reports whether the token ends a sequence: a cell separator, // a row separator, an environment close or a closing brace. func isBoundary(t token) bool { if t.kind == tokRBrace || t.kind == tokAmpersand || isRowEnd(t) { return true } return t.kind == tokCommand && t.text == `\end` } func isRowEnd(t token) bool { return t.kind == tokCommand && t.text == `\\` } // sequence parses atoms until a boundary or the end. When allowOver is // set, the TeX infix constructs \over, \atop and \choose may appear and // take everything parsed so far as their numerator. func (p *parser) sequence(allowOver bool) []*node { var nodes []*node for { t := p.cur() if t.kind == tokEOF || isBoundary(t) { break } if t.kind == tokDegraded { nodes = append(nodes, errorNode(t.text)) p.pos++ continue } if t.kind == tokCommand && isOverCommand(t.text) { p.pos++ if !allowOver { nodes = append(nodes, errorNode(t.text)) continue } num := wrapRow(nodes) den := wrapRow(p.sequence(false)) nodes = []*node{p.overNode(t.text, num, den)} continue } if n := p.atom(); n != nil { nodes = append(nodes, n) } } return nodes } func isOverCommand(text string) bool { switch text { case `\over`, `\atop`, `\choose`, `\overwithdelims`, `\atopwithdelims`, `\abovewithdelims`: return true } return false } func (p *parser) overNode(cmd string, num, den *node) *node { switch cmd { case `\choose`: return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")") case `\atop`: return elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den) default: return el("mfrac", num, den) } } // atom parses one unit with its scripts, or a whole command construct. func (p *parser) atom() *node { t := p.cur() switch t.kind { case tokLBrace: return p.braceGroup() case tokChar: return p.scripts(p.charAtom(t)) case tokCommand: n := p.command() if n == nil { return nil } return p.scripts(n) case tokCaret, tokUnderscore: // A script without a base, the isotope notation of ^3He, takes // an empty base: the script renders over nothing, the way the // notation is printed, and degrading it would refuse notation // the mathematics carries. return p.scripts(el("mrow")) } p.pos++ return errorNode(t.text) } // braceGroup parses a braced group, degrading the whole span when the // closing brace never comes. A group of one node is that node: the braces // only grouped. func (p *parser) braceGroup() *node { open := p.cur() p.pos++ nodes := p.sequence(true) if p.at(tokRBrace) { p.pos++ if len(nodes) == 1 { return nodes[0] } if len(nodes) == 0 { return el("mrow") } row := el("mrow") row.children = nodes return row } p.pos = len(p.toks) - 1 return errorNode(string(p.src[open.start:])) } // charAtom maps one character token to its element. The reserved TeX // characters degrade rather than pass as content. func (p *parser) charAtom(t token) *node { c := t.text switch c { case "#", "$", "%": p.pos++ return errorNode(c) case "~": p.pos++ return spaceNode("0.25em") case "'": p.pos++ return text("mo", "′") } r := []rune(c)[0] switch { case isDigitByte(c[0]) && len(c) == 1: p.pos++ return p.numberRun(c) case isIdentifierRune(r): p.pos++ return p.identifier(c) default: p.pos++ return p.withVariant(el("mo"), c) } } // numberRun collects the digits and decimal points that follow. func (p *parser) numberRun(first string) *node { var b strings.Builder b.WriteString(first) for { t := p.cur() if t.kind != tokChar || len(t.text) != 1 || !isDigitByte(t.text[0]) && t.text != "." { break } b.WriteString(t.text) p.pos++ } return p.withVariant(el("mn"), b.String()) } // identifier renders one letter, upright when the variant says so. func (p *parser) identifier(c string) *node { return p.withVariant(el("mi"), c) } // withVariant fills a leaf node with text, applying the active variant. func (p *parser) withVariant(n *node, c string) *node { n.text = c if p.variant != "" && (n.kind == "mi" || n.kind == "mn" || n.kind == "mo") { if !(n.kind == "mi" && p.variant == "italic") { n.attrs = append(n.attrs, attribute{"mathvariant", p.variant}) } } return n } // scripts attaches the sub and superscript runs that follow a base. A // movable base puts its scripts under and over in display style; inline, // the movablelimits attribute lets the renderer decide. func (p *parser) scripts(base *node) *node { movable := baseMovable(base) if p.atCommand("limits") { p.pos++ movable = true } else if p.atCommand("nolimits") { p.pos++ movable = false } var sub, sup *node for { t := p.cur() if t.kind == tokUnderscore && sub == nil { p.pos++ sub = p.argument() continue } if t.kind == tokCaret && sup == nil { p.pos++ sup = p.argument() continue } break } under := movable && p.display switch { case sub == nil && sup == nil: return base case under && sub != nil && sup != nil: return el("munderover", base, sub, sup) case under && sub != nil: return el("munder", base, sub) case under && sup != nil: return el("mover", base, sup) case sub != nil && sup != nil: return el("msubsup", base, sub, sup) case sub != nil: return el("msub", base, sub) default: return el("msup", base, sup) } } // baseMovable reports whether the base carries movable limits. func baseMovable(base *node) bool { if base.kind != "mo" { return false } for _, a := range base.attrs { if a.key == "movablelimits" { return a.value == "true" } } return false } // argument reads one macro argument or script: a braced group or a single // token. A single digit stays one digit, the way TeX reads x^10. func (p *parser) argument() *node { t := p.cur() switch { case t.kind == tokLBrace: return p.braceGroup() case t.kind == tokChar && len(t.text) == 1 && isDigitByte(t.text[0]): p.pos++ return p.withVariant(el("mn"), t.text) case t.kind == tokChar || t.kind == tokCommand: n := p.atom() if n == nil { return el("mrow") } return n } return errorNode(t.text) } // rawBraced reads a braced group from the source as verbatim text, // keeping the spaces. It fails when the closing brace is missing. func (p *parser) rawBraced() (string, bool) { t := p.cur() if t.kind != tokLBrace { return "", false } depth := 0 for i := t.start; i < len(p.src); { switch p.src[i] { case '{': depth++ case '}': depth-- if depth == 0 { text := string(p.src[t.start+1 : i]) // consume the tokens the span covers for p.pos < len(p.toks) && p.toks[p.pos].start <= i { p.pos++ } return text, true } case '\\': i++ // an escaped character never opens or closes } i++ } return "", false }