Files
scriptorium/internal/mathml/parse.go
T

333 lines
7.9 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "strings"
// parser walks the token list. The variant field is the mathvariant the
// current style command imposes on the atoms it covers.
type parser struct {
src []byte
toks []token
pos int
display bool
variant string
macros map[string]macro
expansions int
}
func newParser(src []byte, display bool) *parser {
return &parser{src: src, toks: tokenise(src), display: display, macros: map[string]macro{}}
}
func (p *parser) cur() token {
p.expandMacros()
return p.toks[p.pos]
}
func (p *parser) at(kind tokenKind) bool { return p.cur().kind == kind }
func (p *parser) atCommand(name string) bool {
t := p.cur()
return t.kind == tokCommand && t.text == `\`+name
}
// parseAll parses the whole source into nodes. A boundary with nothing to
// bound degrades as the stray it is.
func (p *parser) parseAll() []*node {
nodes := p.sequence(true)
for p.cur().kind != tokEOF {
if isBoundary(p.cur()) {
t := p.cur()
p.pos++
nodes = append(nodes, errorNode(t.text))
continue
}
nodes = append(nodes, p.sequence(true)...)
}
return nodes
}
// isBoundary reports whether the token ends a sequence: a cell separator,
// a row separator, an environment close or a closing brace.
func isBoundary(t token) bool {
if t.kind == tokRBrace || t.kind == tokAmpersand || isRowEnd(t) {
return true
}
return t.kind == tokCommand && t.text == `\end`
}
func isRowEnd(t token) bool {
return t.kind == tokCommand && t.text == `\\`
}
// sequence parses atoms until a boundary or the end. When allowOver is
// set, the TeX infix constructs \over, \atop and \choose may appear and
// take everything parsed so far as their numerator.
func (p *parser) sequence(allowOver bool) []*node {
var nodes []*node
for {
t := p.cur()
if t.kind == tokEOF || isBoundary(t) {
break
}
if t.kind == tokDegraded {
nodes = append(nodes, errorNode(t.text))
p.pos++
continue
}
if t.kind == tokCommand && isOverCommand(t.text) {
p.pos++
if !allowOver {
nodes = append(nodes, errorNode(t.text))
continue
}
num := wrapRow(nodes)
den := wrapRow(p.sequence(false))
nodes = []*node{p.overNode(t.text, num, den)}
continue
}
if n := p.atom(); n != nil {
nodes = append(nodes, n)
}
}
return nodes
}
func isOverCommand(text string) bool {
switch text {
case `\over`, `\atop`, `\choose`, `\overwithdelims`, `\atopwithdelims`, `\abovewithdelims`:
return true
}
return false
}
func (p *parser) overNode(cmd string, num, den *node) *node {
switch cmd {
case `\choose`:
return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")")
case `\atop`:
return elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den)
default:
return el("mfrac", num, den)
}
}
// atom parses one unit with its scripts, or a whole command construct.
func (p *parser) atom() *node {
t := p.cur()
switch t.kind {
case tokLBrace:
return p.braceGroup()
case tokChar:
return p.scripts(p.charAtom(t))
case tokCommand:
n := p.command()
if n == nil {
return nil
}
return p.scripts(n)
case tokCaret, tokUnderscore:
// A script without a base, the isotope notation of ^3He, takes
// an empty base: the script renders over nothing, the way the
// notation is printed, and degrading it would refuse notation
// the mathematics carries.
return p.scripts(el("mrow"))
}
p.pos++
return errorNode(t.text)
}
// braceGroup parses a braced group, degrading the whole span when the
// closing brace never comes. A group of one node is that node: the braces
// only grouped.
func (p *parser) braceGroup() *node {
open := p.cur()
p.pos++
nodes := p.sequence(true)
if p.at(tokRBrace) {
p.pos++
if len(nodes) == 1 {
return nodes[0]
}
if len(nodes) == 0 {
return el("mrow")
}
row := el("mrow")
row.children = nodes
return row
}
p.pos = len(p.toks) - 1
return errorNode(string(p.src[open.start:]))
}
// charAtom maps one character token to its element. The reserved TeX
// characters degrade rather than pass as content.
func (p *parser) charAtom(t token) *node {
c := t.text
switch c {
case "#", "$", "%":
p.pos++
return errorNode(c)
case "~":
p.pos++
return spaceNode("0.25em")
case "'":
p.pos++
return text("mo", "′")
}
r := []rune(c)[0]
switch {
case isDigitByte(c[0]) && len(c) == 1:
p.pos++
return p.numberRun(c)
case isIdentifierRune(r):
p.pos++
return p.identifier(c)
default:
p.pos++
return p.withVariant(el("mo"), c)
}
}
// numberRun collects the digits and decimal points that follow.
func (p *parser) numberRun(first string) *node {
var b strings.Builder
b.WriteString(first)
for {
t := p.cur()
if t.kind != tokChar || len(t.text) != 1 || !isDigitByte(t.text[0]) && t.text != "." {
break
}
b.WriteString(t.text)
p.pos++
}
return p.withVariant(el("mn"), b.String())
}
// identifier renders one letter, upright when the variant says so.
func (p *parser) identifier(c string) *node {
return p.withVariant(el("mi"), c)
}
// withVariant fills a leaf node with text, applying the active variant.
func (p *parser) withVariant(n *node, c string) *node {
n.text = c
if p.variant != "" && (n.kind == "mi" || n.kind == "mn" || n.kind == "mo") {
if !(n.kind == "mi" && p.variant == "italic") {
n.attrs = append(n.attrs, attribute{"mathvariant", p.variant})
}
}
return n
}
// scripts attaches the sub and superscript runs that follow a base. A
// movable base puts its scripts under and over in display style; inline,
// the movablelimits attribute lets the renderer decide.
func (p *parser) scripts(base *node) *node {
movable := baseMovable(base)
if p.atCommand("limits") {
p.pos++
movable = true
} else if p.atCommand("nolimits") {
p.pos++
movable = false
}
var sub, sup *node
for {
t := p.cur()
if t.kind == tokUnderscore && sub == nil {
p.pos++
sub = p.argument()
continue
}
if t.kind == tokCaret && sup == nil {
p.pos++
sup = p.argument()
continue
}
break
}
under := movable && p.display
switch {
case sub == nil && sup == nil:
return base
case under && sub != nil && sup != nil:
return el("munderover", base, sub, sup)
case under && sub != nil:
return el("munder", base, sub)
case under && sup != nil:
return el("mover", base, sup)
case sub != nil && sup != nil:
return el("msubsup", base, sub, sup)
case sub != nil:
return el("msub", base, sub)
default:
return el("msup", base, sup)
}
}
// baseMovable reports whether the base carries movable limits.
func baseMovable(base *node) bool {
if base.kind != "mo" {
return false
}
for _, a := range base.attrs {
if a.key == "movablelimits" {
return a.value == "true"
}
}
return false
}
// argument reads one macro argument or script: a braced group or a single
// token. A single digit stays one digit, the way TeX reads x^10.
func (p *parser) argument() *node {
t := p.cur()
switch {
case t.kind == tokLBrace:
return p.braceGroup()
case t.kind == tokChar && len(t.text) == 1 && isDigitByte(t.text[0]):
p.pos++
return p.withVariant(el("mn"), t.text)
case t.kind == tokChar || t.kind == tokCommand:
n := p.atom()
if n == nil {
return el("mrow")
}
return n
}
return errorNode(t.text)
}
// rawBraced reads a braced group from the source as verbatim text,
// keeping the spaces. It fails when the closing brace is missing.
func (p *parser) rawBraced() (string, bool) {
t := p.cur()
if t.kind != tokLBrace {
return "", false
}
depth := 0
for i := t.start; i < len(p.src); {
switch p.src[i] {
case '{':
depth++
case '}':
depth--
if depth == 0 {
text := string(p.src[t.start+1 : i])
// consume the tokens the span covers
for p.pos < len(p.toks) && p.toks[p.pos].start <= i {
p.pos++
}
return text, true
}
case '\\':
i++ // an escaped character never opens or closes
}
i++
}
return "", false
}