Files
scriptorium/internal/mathml/parse.go
T

333 lines
7.9 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "strings"
// parser walks the token list. The variant field is the mathvariant the
// current style command imposes on the atoms it covers.
type parser struct {
src []byte
toks []token
pos int
display bool
variant string
macros map[string]macro
expansions int
}
func newParser(src []byte, display bool) *parser {
return &parser{src: src, toks: tokenise(src), display: display, macros: map[string]macro{}}
}
func (p *parser) cur() token {
p.expandMacros()
return p.toks[p.pos]
}
func (p *parser) at(kind tokenKind) bool { return p.cur().kind == kind }
func (p *parser) atCommand(name string) bool {
t := p.cur()
return t.kind == tokCommand && t.text == `\`+name
}
// parseAll parses the whole source into nodes. A boundary with nothing to
// bound degrades as the stray it is.
func (p *parser) parseAll() []*node {
nodes := p.sequence(true)
for p.cur().kind != tokEOF {
if isBoundary(p.cur()) {
t := p.cur()
p.pos++
nodes = append(nodes, errorNode(t.text))
continue
}
nodes = append(nodes, p.sequence(true)...)
}
return nodes
}
// isBoundary reports whether the token ends a sequence: a cell separator,
// a row separator, an environment close or a closing brace.
func isBoundary(t token) bool {
if t.kind == tokRBrace || t.kind == tokAmpersand || isRowEnd(t) {
return true
}
return t.kind == tokCommand && t.text == `\end`
}
func isRowEnd(t token) bool {
return t.kind == tokCommand && t.text == `\\`
}
// sequence parses atoms until a boundary or the end. When allowOver is
// set, the TeX infix constructs \over, \atop and \choose may appear and
// take everything parsed so far as their numerator.
func (p *parser) sequence(allowOver bool) []*node {
var nodes []*node
for {
t := p.cur()
if t.kind == tokEOF || isBoundary(t) {
break
}
if t.kind == tokDegraded {
nodes = append(nodes, errorNode(t.text))
p.pos++
continue
}
if t.kind == tokCommand && isOverCommand(t.text) {
p.pos++
if !allowOver {
nodes = append(nodes, errorNode(t.text))
continue
}
num := wrapRow(nodes)
den := wrapRow(p.sequence(false))
nodes = []*node{p.overNode(t.text, num, den)}
continue
}
if n := p.atom(); n != nil {
nodes = append(nodes, n)
}
}
return nodes
}
func isOverCommand(text string) bool {
switch text {
case `\over`, `\atop`, `\choose`, `\overwithdelims`, `\atopwithdelims`, `\abovewithdelims`:
return true
}
return false
}
func (p *parser) overNode(cmd string, num, den *node) *node {
switch cmd {
case `\choose`:
return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")")
case `\atop`:
return elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den)
default:
return el("mfrac", num, den)
}
}
// atom parses one unit with its scripts, or a whole command construct.
func (p *parser) atom() *node {
t := p.cur()
switch t.kind {
case tokLBrace:
return p.braceGroup()
case tokChar:
return p.scripts(p.charAtom(t))
case tokCommand:
n := p.command()
if n == nil {
return nil
}
return p.scripts(n)
case tokCaret, tokUnderscore:
// A script without a base, the isotope notation of ^3He, takes
// an empty base: the script renders over nothing, the way the
// notation is printed, and degrading it would refuse notation
// the mathematics carries.
return p.scripts(el("mrow"))
}
p.pos++
return errorNode(t.text)
}
// braceGroup parses a braced group, degrading the whole span when the
// closing brace never comes. A group of one node is that node: the braces
// only grouped.
func (p *parser) braceGroup() *node {
open := p.cur()
p.pos++
nodes := p.sequence(true)
if p.at(tokRBrace) {
p.pos++
if len(nodes) == 1 {
return nodes[0]
}
if len(nodes) == 0 {
return el("mrow")
}
row := el("mrow")
row.children = nodes
return row
}
p.pos = len(p.toks) - 1
return errorNode(string(p.src[open.start:]))
}
// charAtom maps one character token to its element. The reserved TeX
// characters degrade rather than pass as content.
func (p *parser) charAtom(t token) *node {
c := t.text
switch c {
case "#", "$", "%":
p.pos++
return errorNode(c)
case "~":
p.pos++
return spaceNode("0.25em")
case "'":
p.pos++
return text("mo", "′")
}
r := []rune(c)[0]
switch {
case isDigitByte(c[0]) && len(c) == 1:
p.pos++
return p.numberRun(c)
case isIdentifierRune(r):
p.pos++
return p.identifier(c)
default:
p.pos++
return p.withVariant(el("mo"), c)
}
}
// numberRun collects the digits and decimal points that follow.
func (p *parser) numberRun(first string) *node {
var b strings.Builder
b.WriteString(first)
for {
t := p.cur()
if t.kind != tokChar || len(t.text) != 1 || !isDigitByte(t.text[0]) && t.text != "." {
break
}
b.WriteString(t.text)
p.pos++
}
return p.withVariant(el("mn"), b.String())
}
// identifier renders one letter, upright when the variant says so.
func (p *parser) identifier(c string) *node {
return p.withVariant(el("mi"), c)
}
// withVariant fills a leaf node with text, applying the active variant.
func (p *parser) withVariant(n *node, c string) *node {
n.text = c
if p.variant != "" && (n.kind == "mi" || n.kind == "mn" || n.kind == "mo") {
if !(n.kind == "mi" && p.variant == "italic") {
n.attrs = append(n.attrs, attribute{"mathvariant", p.variant})
}
}
return n
}
// scripts attaches the sub and superscript runs that follow a base. A
// movable base puts its scripts under and over in display style; inline,
// the movablelimits attribute lets the renderer decide.
func (p *parser) scripts(base *node) *node {
movable := baseMovable(base)
if p.atCommand("limits") {
p.pos++
movable = true
} else if p.atCommand("nolimits") {
p.pos++
movable = false
}
var sub, sup *node
for {
t := p.cur()
if t.kind == tokUnderscore && sub == nil {
p.pos++
sub = p.argument()
continue
}
if t.kind == tokCaret && sup == nil {
p.pos++
sup = p.argument()
continue
}
break
}
under := movable && p.display
switch {
case sub == nil && sup == nil:
return base
case under && sub != nil && sup != nil:
return el("munderover", base, sub, sup)
case under && sub != nil:
return el("munder", base, sub)
case under && sup != nil:
return el("mover", base, sup)
case sub != nil && sup != nil:
return el("msubsup", base, sub, sup)
case sub != nil:
return el("msub", base, sub)
default:
return el("msup", base, sup)
}
}
// baseMovable reports whether the base carries movable limits.
func baseMovable(base *node) bool {
if base.kind != "mo" {
return false
}
for _, a := range base.attrs {
if a.key == "movablelimits" {
return a.value == "true"
}
}
return false
}
// argument reads one macro argument or script: a braced group or a single
// token. A single digit stays one digit, the way TeX reads x^10.
func (p *parser) argument() *node {
t := p.cur()
switch {
case t.kind == tokLBrace:
return p.braceGroup()
case t.kind == tokChar && len(t.text) == 1 && isDigitByte(t.text[0]):
p.pos++
return p.withVariant(el("mn"), t.text)
case t.kind == tokChar || t.kind == tokCommand:
n := p.atom()
if n == nil {
return el("mrow")
}
return n
}
return errorNode(t.text)
}
// rawBraced reads a braced group from the source as verbatim text,
// keeping the spaces. It fails when the closing brace is missing.
func (p *parser) rawBraced() (string, bool) {
t := p.cur()
if t.kind != tokLBrace {
return "", false
}
depth := 0
for i := t.start; i < len(p.src); {
switch p.src[i] {
case '{':
depth++
case '}':
depth--
if depth == 0 {
text := string(p.src[t.start+1 : i])
// consume the tokens the span covers
for p.pos < len(p.toks) && p.toks[p.pos].start <= i {
p.pos++
}
return text, true
}
case '\\':
i++ // an escaped character never opens or closes
}
i++
}
return "", false
}