feat: render Markdown, mathematics and Mermaid diagrams server-side

This commit is contained in:
2026-09-27 19:15:50 +02:00
commit 35bc42623b
72 changed files with 8914 additions and 0 deletions
+570
View File
@@ -0,0 +1,570 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import (
"strings"
"unicode"
)
func isDigitByte(c byte) bool { return c >= '0' && c <= '9' }
// isIdentifierRune reports whether the rune names a variable: a letter in
// any script.
func isIdentifierRune(r rune) bool {
return unicode.IsLetter(r)
}
func spaceNode(width string) *node {
return &node{kind: "mspace", attrs: []attribute{{"width", width}}}
}
// fenced wraps content in stretchy fence delimiters.
func fenced(open string, content *node, close string) *node {
row := el("mrow")
row.children = []*node{fenceNode(open), content, fenceNode(close)}
return row
}
func fenceNode(char string) *node {
return text("mo", char, attribute{"fence", "true"}, attribute{"stretchy", "true"})
}
// wrapRow packs nodes into one row; an empty list becomes an empty row.
func wrapRow(nodes []*node) *node {
if len(nodes) == 1 {
return nodes[0]
}
row := el("mrow")
row.children = nodes
return row
}
// command parses one backslash command and whatever it takes with it. The
// cursor sits on the command token.
func (p *parser) command() *node {
t := p.cur()
name := t.text[1:]
p.pos++
// The single-character escapes: braces, the reserved characters and
// the double bar.
switch name {
case "{", "}", "%", "$", "#", "&", "_", "|", "<", ">", ",", ":", ";", "!", " ":
if w, ok := spaces[name]; ok {
return spaceNode(w)
}
if name == "|" {
return text("mo", "‖")
}
return p.withVariant(el("mo"), name)
}
switch name {
case "frac", "dfrac", "tfrac", "cfrac":
num := p.argument()
den := p.argument()
return el("mfrac", num, den)
case "binom", "dbinom", "tbinom":
num := p.argument()
den := p.argument()
return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")")
case "genfrac":
return p.genfrac()
case "sqrt":
return p.sqrt()
case "substack":
raw, ok := p.rawBraced()
if !ok {
return errorNode(t.text)
}
return p.substack(raw)
case "text", "textrm", "textnormal", "mbox", "textmd", "textsc", "textsl":
return p.textArgument()
case "textbf", "textup":
return p.textArgument(attribute{"fontweight", "bold"})
case "textit", "emph":
return p.textArgument(attribute{"fontstyle", "italic"})
case "textsf":
return p.textArgument(attribute{"mathvariant", "sans-serif"})
case "texttt":
return p.textArgument(attribute{"mathvariant", "monospace"})
case "operatorname", "operatornamewithlimits":
raw, ok := p.rawBraced()
if !ok {
return errorNode(t.text)
}
return identifierMulti(raw, name == "operatornamewithlimits")
case "operatorname*":
raw, ok := p.rawBraced()
if !ok {
return errorNode(t.text)
}
return identifierMulti(raw, true)
case "mathchoice":
// four arguments, of which the parser renders the second, the
// text style one; the display, script and scriptscript variants
// of MathML Core are the renderer's business
_, _ = p.argument(), p.argument()
if n := p.argument(); n != nil {
_, _ = p.argument(), p.argument()
return n
}
return errorNode(t.text)
case "phantom", "hphantom", "vphantom":
arg := p.argument()
inner := el("mphantom")
inner.children = arg.children
inner.text = arg.text
return inner
case "overset", "stackrel":
over := p.argument()
base := p.argument()
return el("mover", base, over)
case "underset":
under := p.argument()
base := p.argument()
return el("munder", base, under)
case "pmod":
arg := p.argument()
row := el("mrow")
row.children = []*node{
identifierMulti("mod", false), spaceNode("0.2778em"), arg,
}
return fenced("(", row, ")")
case "left":
return p.leftRight()
case "big", "Big", "bigg", "Bigg",
"bigl", "Bigl", "biggl", "Biggl",
"bigr", "Bigr", "biggr", "Biggr",
"bigm", "Bigm", "biggm", "Biggm":
return p.bigDelimiter()
case "middle":
return text("mo", "", attribute{"fence", "true"}, attribute{"stretchy", "true"})
case "pod":
return fenced("(", p.argument(), ")")
case "xrightarrow", "xleftarrow", "xleftrightarrow", "xhookrightarrow",
"xhookleftarrow", "xRightarrow", "xLeftarrow", "xLeftrightarrow",
"xrightleftharpoons", "xmapsto", "xtwoheadleftarrow", "xtwoheadrightarrow",
"xleftharpoonup", "xrightharpoonup", "xleftharpoondown", "xrightharpoondown",
"xleftrightharpoons", "xtofrom", "xlongequal":
return p.xArrow(name)
case "displaystyle", "textstyle", "scriptstyle", "scriptscriptstyle":
return p.styleDeclaration(name)
case "newcommand", "renewcommand", "providecommand", "def", "gdef", "DeclareMathOperator", "DeclareMathOperator*":
return p.macroDefinition(name)
case "begin":
return p.environment()
case "end":
return errorNode(t.text)
case "limits", "nolimits":
return errorNode(t.text)
case "not":
return p.not()
case "hspace", "mspace":
return p.spaceArgument()
case "color", "textcolor":
return p.color(name)
}
if s, ok := symbols[name]; ok {
return p.symbolNode(t.text, s)
}
if functions[name] {
return identifierMulti(name, movableFunctions[name])
}
if v, ok := styles[name]; ok {
return p.styleArgument(v)
}
if accent, ok := accents[name]; ok {
return p.accent(accent)
}
if w, ok := spaces[name]; ok {
return spaceNode(w)
}
return errorNode(t.text)
}
// symbolNode renders a table symbol, upright under a variant switch.
func (p *parser) symbolNode(source string, s symbol) *node {
if s.mi {
return p.identifier(s.char)
}
n := p.withVariant(el("mo"), s.char)
if s.movable {
n.attrs = append(n.attrs, attribute{"movablelimits", "true"})
}
return n
}
// identifierMulti renders a name as an upright identifier, which a
// multi-character mi is by default. The movable flag marks the limit
// operators.
func identifierMulti(name string, movable bool) *node {
if movable {
return text("mi", name, attribute{"movablelimits", "true"})
}
return text("mi", name)
}
// textArgument reads a text command's argument verbatim.
func (p *parser) textArgument(attrs ...attribute) *node {
raw, ok := p.rawBraced()
if !ok {
return errorNode(`\` + "text")
}
return &node{kind: "mtext", text: raw, attrs: attrs}
}
// styleArgument parses the argument under a forced variant.
func (p *parser) styleArgument(variant string) *node {
save := p.variant
if variant == "italic" {
p.variant = ""
} else {
p.variant = variant
}
arg := p.argument()
p.variant = save
return arg
}
// accent puts its mark over or under the argument.
func (p *parser) accent(a struct {
char string
under bool
}) *node {
base := p.argument()
mark := text("mo", a.char, attribute{"stretchy", "false"})
if a.under {
return el("munder", base, mark)
}
if a.char == "\u203e" || a.char == "\u23de" || a.char == "_" || a.char == "\u23e1" || a.char == "\u23df" {
mark = text("mo", a.char, attribute{"stretchy", "true"})
}
return el("mover", base, mark)
}
// sqrt parses a radical with its optional index.
func (p *parser) sqrt() *node {
if p.at(tokChar) && p.cur().text == "[" {
p.pos++
var index []*node
for !p.at(tokEOF) && !(p.at(tokChar) && p.cur().text == "]") {
if n := p.atom(); n != nil {
index = append(index, n)
}
}
if p.at(tokChar) && p.cur().text == "]" {
p.pos++
return el("mroot", p.argument(), wrapRow(index))
}
return errorNode(`\sqrt`)
}
return el("msqrt", p.argument())
}
// genfrac parses the six arguments of \genfrac: the delimiters, the rule
// thickness, the style and the numerator and denominator.
func (p *parser) genfrac() *node {
open := p.delimiterArg()
close := p.delimiterArg()
thick, hasThick := p.bracketArg()
style, _ := p.bracketArg()
num := p.argument()
den := p.argument()
frac := el("mfrac")
if hasThick {
frac.attrs = append(frac.attrs, attribute{"linethickness", thick})
}
frac.children = []*node{num, den}
if style >= "2" {
wrapped := elA("mstyle", []attribute{{"scriptlevel", "1"}})
wrapped.children = []*node{frac}
frac = wrapped
}
if open == "" && close == "" {
return frac
}
if open == "" {
open = "."
}
if close == "" {
close = "."
}
row := el("mrow")
row.children = []*node{p.fenceOf(open), frac, p.fenceOf(close)}
return row
}
// bracketArg reads an optional bracketed argument, reporting whether one
// was there.
func (p *parser) bracketArg() (string, bool) {
if !p.at(tokChar) || p.cur().text != "[" {
return "", false
}
p.pos++
var b strings.Builder
for {
t := p.cur()
if t.kind == tokEOF {
return "", true
}
if t.kind == tokChar && t.text == "]" {
p.pos++
return b.String(), true
}
b.WriteString(t.text)
p.pos++
}
}
// delimiterArg reads a \genfrac delimiter argument: a character, a
// command or an empty group.
func (p *parser) delimiterArg() string {
t := p.cur()
switch {
case t.kind == tokLBrace:
p.pos++
if p.at(tokRBrace) {
p.pos++
return ""
}
d, _ := p.delimiter()
return d
case t.kind == tokChar || t.kind == tokCommand:
d, _ := p.delimiter()
return d
}
return ""
}
// leftRight parses a stretchy delimited row.
func (p *parser) leftRight() *node {
start := p.pos
open, ok := p.delimiter()
if !ok {
return errorNode(`\left`)
}
var content []*node
for {
t := p.cur()
if t.kind == tokEOF {
p.pos = len(p.toks) - 1
return errorNode(string(p.src[start-1:]))
}
if t.kind == tokCommand && t.text == `\right` {
p.pos++
break
}
if n := p.atom(); n != nil {
content = append(content, n)
}
}
close, _ := p.delimiter()
row := el("mrow")
row.children = append([]*node{p.fenceOf(open)}, content...)
row.children = append(row.children, p.fenceOf(close))
return row
}
// delimiter reads one delimiter: a character or a table command, with the
// dot meaning invisible. The second result reports whether a delimiter
// was there at all.
func (p *parser) delimiter() (string, bool) {
t := p.cur()
switch t.kind {
case tokChar:
p.pos++
if t.text == "." {
return "", true
}
return t.text, true
case tokCommand:
name := t.text[1:]
if name == "|" {
p.pos++
return "‖", true
}
if s, ok := symbols[name]; ok {
p.pos++
return s.char, true
}
if _, ok := spaces[name]; ok {
p.pos++
return "", true
}
}
return "", false
}
// fenceOf makes the fence for a delimiter name; an empty name is the
// invisible fence.
func (p *parser) fenceOf(d string) *node {
if d == "" {
return text("mo", "", attribute{"fence", "true"})
}
if d == "." {
return text("mo", "", attribute{"fence", "true"})
}
return fenceNode(d)
}
// bigDelimiter renders \big and its relatives around one delimiter.
func (p *parser) bigDelimiter() *node {
t := p.cur()
d, ok := p.delimiter()
if !ok {
return errorNode(t.text)
}
return text("mo", d, attribute{"stretchy", "true"})
}
// xArrowNames gives the shaft of every extensible arrow command.
var xArrowNames = map[string]string{
"xrightarrow": "→",
"xleftarrow": "←",
"xleftrightarrow": "↔",
"xhookrightarrow": "↪",
"xhookleftarrow": "↩",
"xRightarrow": "⇒",
"xLeftarrow": "⇐",
"xLeftrightarrow": "⇔",
"xrightleftharpoons": "⇌",
"xmapsto": "↦",
"xtwoheadleftarrow": "↞",
"xtwoheadrightarrow": "↠",
"xleftharpoonup": "↼",
"xrightharpoonup": "⇀",
"xleftharpoondown": "↽",
"xrightharpoondown": "⇁",
"xleftrightharpoons": "⇋",
"xtofrom": "⇄",
"xlongequal": "=",
}
// xArrow parses an extensible arrow: an optional underscript in brackets,
// then the overscript in braces.
func (p *parser) xArrow(name string) *node {
char := xArrowNames[name]
arrow := text("mo", char, attribute{"stretchy", "true"})
var under *node
if c, ok := p.bracketArg(); ok && c != "" {
under = &node{kind: "mrow", text: c}
}
over := p.argument()
if under == nil {
return el("mover", arrow, over)
}
return el("munderover", arrow, under, over)
}
// styleDeclaration wraps the rest of the current group in an mstyle.
func (p *parser) styleDeclaration(name string) *node {
rest := p.sequence(false)
switch name {
case "displaystyle":
n := elA("mstyle", []attribute{{"displaystyle", "true"}})
n.children = rest
return n
case "textstyle":
n := elA("mstyle", []attribute{{"displaystyle", "false"}})
n.children = rest
return n
case "scriptstyle":
n := elA("mstyle", []attribute{{"scriptlevel", "1"}})
n.children = rest
return n
default:
n := elA("mstyle", []attribute{{"scriptlevel", "2"}})
n.children = rest
return n
}
}
// not combines the negation slash with the relation that follows.
func (p *parser) not() *node {
t := p.cur()
if t.kind == tokCommand {
name := t.text[1:]
if s, ok := symbols[name]; ok {
p.pos++
return text("mo", s.char+"̸")
}
}
return text("mo", "¬")
}
// spaceArgument reads the argument of \hspace and \mspace.
func (p *parser) spaceArgument() *node {
t := p.cur()
if t.kind == tokLBrace {
p.pos++
var b strings.Builder
for {
c := p.cur()
if c.kind == tokEOF || c.kind == tokRBrace {
break
}
b.WriteString(c.text)
p.pos++
}
if p.at(tokRBrace) {
p.pos++
}
if validWidth(b.String()) {
return spaceNode(b.String())
}
return errorNode(t.text)
}
return errorNode(t.text)
}
// validWidth accepts a number with a CSS length unit.
func validWidth(s string) bool {
i := 0
for i < len(s) && (isDigitByte(s[i]) || s[i] == '.' || s[i] == '-') {
i++
}
if i == 0 {
return false
}
switch s[i:] {
case "em", "ex", "px", "pt", "cm", "mm", "in", "mu", "%":
return true
}
return false
}
// color wraps its argument or the rest of the group in a coloured style.
func (p *parser) color(name string) *node {
c, ok := p.rawBraced()
if !ok || !validColour(c) {
return errorNode(`\` + name)
}
n := elA("mstyle", []attribute{{"mathcolor", c}})
if name == "textcolor" {
n.children = []*node{p.argument()}
return n
}
n.children = p.sequence(false)
return n
}
// validColour accepts the colour names and the hex forms.
func validColour(s string) bool {
switch s {
case "red", "green", "blue", "cyan", "magenta", "yellow", "black",
"white", "gray", "grey", "orange", "purple", "brown", "pink",
"olive", "violet", "teal", "navy", "darkgray", "lightgray":
return true
}
if len(s) == 7 && s[0] == '#' || len(s) == 4 && s[0] == '#' {
for i := 1; i < len(s); i++ {
c := s[i]
if !isDigitByte(c) && !(c >= 'a' && c <= 'f') && !(c >= 'A' && c <= 'F') {
return false
}
}
return true
}
return false
}
+252
View File
@@ -0,0 +1,252 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "strings"
// envSpec describes a table environment: its delimiters, its column
// alignment, whether it renders in script size, whether it takes a column
// specification and whether it takes a number argument first.
type envSpec struct {
fences [2]string
align string
scripted bool
colSpec bool
number bool
}
var environments = map[string]envSpec{
"matrix": {}, "pmatrix": {fences: [2]string{"(", ")"}},
"bmatrix": {fences: [2]string{"[", "]"}},
"Bmatrix": {fences: [2]string{"{", "}"}},
"vmatrix": {fences: [2]string{"|", "|"}},
"Vmatrix": {fences: [2]string{"‖", "‖"}},
"matrix*": {},
"pmatrix*": {fences: [2]string{"(", ")"}},
"bmatrix*": {fences: [2]string{"[", "]"}},
"Bmatrix*": {fences: [2]string{"{", "}"}},
"vmatrix*": {fences: [2]string{"|", "|"}},
"Vmatrix*": {fences: [2]string{"‖", "‖"}},
"smallmatrix": {scripted: true},
"cases": {fences: [2]string{"{", ""}, align: "left"},
"rcases": {fences: [2]string{"", "}"}, align: "left"},
"dcases": {fences: [2]string{"{", ""}, align: "left"},
"drcases": {fences: [2]string{"", "}"}, align: "left"},
"aligned": {align: "right left"},
"align": {align: "right left"},
"align*": {align: "right left"},
"alignedat": {align: "right left", number: true},
"alignat": {align: "right left", number: true},
"alignat*": {align: "right left", number: true},
"split": {align: "right left"},
"gather": {}, "gather*": {},
"equation": {}, "equation*": {},
"array": {colSpec: true},
"darray": {colSpec: true},
"subarray": {colSpec: true, scripted: true},
}
// environment parses a whole \begin{name}...\end{name} construct. The
// cursor sits just after the \begin token.
func (p *parser) environment() *node {
begin := p.toks[p.pos-1]
name, ok := p.envName()
if !ok {
return errorNode(begin.text)
}
spec, supported := environments[name]
if !supported {
return p.degradeEnvironment(begin)
}
if spec.colSpec {
raw, ok := p.rawBraced()
if !ok {
return errorNode(begin.text)
}
_, supported = parseColSpec(raw)
if !supported {
return p.degradeEnvironment(begin)
}
} else if spec.number {
if _, ok := p.rawBraced(); !ok {
return errorNode(begin.text)
}
}
rows := p.tableRows()
if !p.atCommand("end") {
p.pos = len(p.toks) - 1
return errorNode(string(p.src[begin.start:]))
}
endStart := p.toks[p.pos].start
p.pos++
endName, ok := p.envName()
if !ok || endName != name {
p.pos = len(p.toks) - 1
return errorNode(string(p.src[begin.start:]))
}
_ = endStart
table := buildTable(rows, spec)
if spec.scripted {
inner := elA("mstyle", []attribute{{"scriptlevel", "1"}})
inner.children = []*node{table}
table = inner
}
if spec.fences == [2]string{"", ""} {
return table
}
row := el("mrow")
row.children = []*node{p.fenceOf(spec.fences[0]), table, p.fenceOf(spec.fences[1])}
return row
}
// envName reads the environment name in braces.
func (p *parser) envName() (string, bool) {
raw, ok := p.rawBraced()
if !ok || raw == "" {
return "", false
}
return raw, true
}
// degradeEnvironment consumes a whole unsupported environment, up to and
// including its matching \end, and degrades it as its verbatim source.
func (p *parser) degradeEnvironment(begin token) *node {
depth := 1
end := len(p.src)
i := p.pos
for ; i < len(p.toks); i++ {
t := p.toks[i]
if t.kind != tokCommand {
continue
}
switch t.text {
case `\begin`:
depth++
case `\end`:
depth--
if depth == 0 {
end = t.end
// include the name argument of \end
if i+2 < len(p.toks) && p.toks[i+1].kind == tokLBrace && p.toks[i+2].kind == tokRBrace {
end = p.toks[i+2].end
i += 2
}
i++
p.pos = i
return errorNode(string(p.src[begin.start:end]))
}
}
}
p.pos = len(p.toks) - 1
return errorNode(string(p.src[begin.start:]))
}
// tableRows parses the rows of a table, each row a slice of cells, until
// the \end or the end of input.
func (p *parser) tableRows() [][]*node {
var rows [][]*node
for {
var cells []*node
for {
nodes := p.sequence(true)
cell := el("mtd")
cell.children = nodes
cells = append(cells, cell)
if p.at(tokAmpersand) {
p.pos++
continue
}
break
}
rows = append(rows, cells)
if isRowEnd(p.cur()) {
p.pos++
p.rowSpacing()
if p.at(tokEOF) || p.atCommand("end") {
break
}
continue
}
break
}
return rows
}
// rowSpacing skips the optional bracket after a row separator.
func (p *parser) rowSpacing() {
if p.at(tokChar) && p.cur().text == "[" {
for {
t := p.cur()
p.pos++
if t.kind == tokEOF || t.kind == tokChar && t.text == "]" {
return
}
}
}
}
// parseColSpec reads an array column specification: alignment letters and
// vertical rules, nothing else.
func parseColSpec(raw string) (int, bool) {
count := 0
for _, c := range raw {
switch c {
case 'l', 'c', 'r':
count++
case '|', ' ', '\t':
default:
return 0, false
}
}
if count == 0 {
return 0, false
}
return count, true
}
// buildTable assembles the mtable with its alignment attributes. The
// "right left" alignment alternates over the widest row.
func buildTable(rows [][]*node, spec envSpec) *node {
table := el("mtable")
cols := 0
for _, row := range rows {
if len(row) > cols {
cols = len(row)
}
}
switch {
case spec.align == "left" && cols > 0:
table.attrs = append(table.attrs, attribute{"columnalign", "left"})
case spec.align == "right left" && cols > 0:
var b strings.Builder
for i := range cols {
if i > 0 {
b.WriteString(" ")
}
if i%2 == 0 {
b.WriteString("right")
} else {
b.WriteString("left")
}
}
table.attrs = append(table.attrs, attribute{"columnalign", b.String()})
}
for _, row := range rows {
tr := el("mtr")
tr.children = row
table.children = append(table.children, tr)
}
return table
}
// substack renders the rows of a \substack argument in script size.
func (p *parser) substack(raw string) *node {
q := newParser([]byte(raw), false)
rows := q.tableRows()
table := buildTable(rows, envSpec{})
inner := elA("mstyle", []attribute{{"scriptlevel", "1"}})
inner.children = []*node{table}
return inner
}
+221
View File
@@ -0,0 +1,221 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "strings"
// macro is a user-defined command: the number of arguments its body takes
// and the body itself as tokens.
type macro struct {
args int
body []token
}
// The depth limits one call site from nesting forever; the budget bounds
// the whole parse, so a self-splicing macro can never outrun the parser.
const (
macroExpansionDepth = 64
macroExpansionBudget = 10000
)
// expandMacros replaces the macro call at the cursor with its expanded
// body, ready for the parser to read. A call beyond the limits degrades
// to its own source.
func (p *parser) expandMacros() {
for range macroExpansionDepth {
t := p.toks[p.pos]
if t.kind != tokCommand {
return
}
if p.expansions >= macroExpansionBudget {
p.degradeAt(p.pos, t)
return
}
m, ok := p.macros[t.text[1:]]
if !ok {
return
}
j := p.pos + 1
args := make([][]token, 0, m.args)
for range m.args {
arg, next, ok := p.argTokens(j)
if !ok {
break
}
args = append(args, arg)
j = next
}
if len(args) != m.args {
p.degradeAt(p.pos, t)
return
}
var body []token
for i := 0; i < len(m.body); i++ {
bt := m.body[i]
if bt.kind == tokChar && bt.text == "#" && i+1 < len(m.body) &&
m.body[i+1].kind == tokChar && len(m.body[i+1].text) == 1 &&
isDigitByte(m.body[i+1].text[0]) {
k := int(m.body[i+1].text[0] - '0')
if k >= 1 && k <= len(args) {
body = append(body, args[k-1]...)
}
i++
continue
}
body = append(body, bt)
}
// The spliced tokens carry the call site as their position, so a
// construct that fails inside a macro degrades at the call.
for i := range body {
body[i].start = t.start
body[i].end = t.end
}
spliced := make([]token, 0, len(p.toks)-(j-p.pos)+len(body))
spliced = append(spliced, p.toks[:p.pos]...)
spliced = append(spliced, body...)
spliced = append(spliced, p.toks[j:]...)
p.toks = spliced
p.expansions++
}
t := p.toks[p.pos]
p.degradeAt(p.pos, t)
}
// degradeAt replaces one token with a degraded token.
func (p *parser) degradeAt(i int, t token) {
p.toks[i] = token{kind: tokDegraded, text: t.text, start: t.start, end: t.end}
}
// argTokens reads one macro argument from position j: a braced group with
// its braces, or a single token.
func (p *parser) argTokens(j int) ([]token, int, bool) {
if j >= len(p.toks) || p.toks[j].kind == tokEOF {
return nil, j, false
}
if p.toks[j].kind != tokLBrace {
return []token{p.toks[j]}, j + 1, true
}
depth := 0
for k := j; k < len(p.toks); k++ {
switch p.toks[k].kind {
case tokLBrace:
depth++
case tokRBrace:
depth--
if depth == 0 {
group := make([]token, k+1-j)
copy(group, p.toks[j:k+1])
return group, k + 1, true
}
case tokEOF:
return nil, j, false
}
}
return nil, j, false
}
// macroDefinition registers a \newcommand or \def style definition and
// produces no output. The cursor sits just after the definition command.
func (p *parser) macroDefinition(kind string) *node {
source := `\` + kind
if kind == "DeclareMathOperator" || kind == "DeclareMathOperator*" {
name := p.defName()
if name == "" {
return errorNode(source)
}
body, ok := p.rawBraced()
if !ok {
return errorNode(source)
}
wrap := `\operatorname{` + body + `}`
if strings.HasSuffix(kind, "*") {
wrap = `\operatorname*{` + body + `}`
}
p.macros[name] = macro{body: tokenise([]byte(wrap))}
return nil
}
if kind == "def" || kind == "gdef" {
return p.tecDefinition(source)
}
name := p.defName()
if name == "" {
return errorNode(source)
}
args := 0
if count, ok := p.bracketArg(); ok && count != "" {
n := 0
for i := 0; i < len(count); i++ {
if !isDigitByte(count[i]) {
return errorNode(source)
}
n = n*10 + int(count[i]-'0')
}
if n > 9 {
return errorNode(source)
}
args = n
}
body, ok := p.rawBraced()
if !ok {
return errorNode(source)
}
p.macros[name] = macro{args: args, body: tokenise([]byte(body))}
return nil
}
// tecDefinition registers a \def, whose parameter text names undelimited
// arguments with #1 up to #9.
func (p *parser) tecDefinition(source string) *node {
name := p.defName()
if name == "" {
return errorNode(source)
}
args := 0
for {
t := p.cur()
if t.kind == tokChar && t.text == "#" {
p.pos++
d := p.cur()
if d.kind != tokChar || len(d.text) != 1 || !isDigitByte(d.text[0]) {
return errorNode(source)
}
if int(d.text[0]-'0') != args+1 {
return errorNode(source)
}
args++
p.pos++
continue
}
break
}
body, ok := p.rawBraced()
if !ok {
return errorNode(source)
}
p.macros[name] = macro{args: args, body: tokenise([]byte(body))}
return nil
}
// defName reads the name a definition declares: a braced command or a
// bare command.
func (p *parser) defName() string {
if p.at(tokLBrace) {
p.pos++
if p.at(tokCommand) {
name := p.cur().text[1:]
p.pos++
if p.at(tokRBrace) {
p.pos++
return name
}
}
return ""
}
if p.at(tokCommand) {
name := p.cur().text[1:]
p.pos++
return name
}
return ""
}
+97
View File
@@ -0,0 +1,97 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
// Package mathml converts TeX mathematics to MathML Core with an engine of
// its own, built on the standard library alone. The grammar covers the
// standard command surface that maps to MathML: the complete symbol tables,
// fractions, scripts, operators with movable limits, stretchy delimiters,
// the amsmath environments and bounded macros.
//
// A construct outside the mappable surface is neither dropped nor
// mistranslated: it stays in the output as its verbatim source inside an
// merror element, so the author sees exactly what was not understood.
package mathml
import "strings"
// Render converts the TeX source to a MathML Core math element. The block
// form sets display="block". The same source always produces
// byte-identical output.
func Render(source []byte, display bool) []byte {
p := newParser(source, display)
body := p.parseAll()
var b strings.Builder
b.WriteString(`<math xmlns="http://www.w3.org/1998/Math/MathML"`)
if display {
b.WriteString(` display="block"`)
}
b.WriteString(">")
writeNodes(&b, body)
b.WriteString("</math>")
return []byte(b.String())
}
// node is one element of the MathML tree. Attributes keep the order the
// parser gave them, which keeps the output deterministic.
type node struct {
kind string
text string
attrs []attribute
children []*node
}
type attribute struct {
key string
value string
}
func el(kind string, children ...*node) *node {
return &node{kind: kind, children: children}
}
// elA builds an element that carries attributes.
func elA(kind string, attrs []attribute, children ...*node) *node {
return &node{kind: kind, attrs: attrs, children: children}
}
func text(kind, text string, attrs ...attribute) *node {
return &node{kind: kind, text: text, attrs: attrs}
}
// errorNode degrades a span of source to its verbatim text in a marked
// element.
func errorNode(source string) *node {
return &node{kind: "merror", children: []*node{{kind: "mtext", text: source}}}
}
// writeNodes renders nodes without indentation; the output is one line,
// which keeps it a single inline unit in the surrounding HTML.
func writeNodes(b *strings.Builder, nodes []*node) {
for _, n := range nodes {
writeNode(b, n)
}
}
func writeNode(b *strings.Builder, n *node) {
b.WriteString("<")
b.WriteString(n.kind)
for _, a := range n.attrs {
b.WriteString(" ")
b.WriteString(a.key)
b.WriteString(`="`)
escapeXML(b, a.value)
b.WriteString(`"`)
}
b.WriteString(">")
escapeXML(b, n.text)
writeNodes(b, n.children)
b.WriteString("</")
b.WriteString(n.kind)
b.WriteString(">")
}
var xmlEscaper = strings.NewReplacer("&", "&amp;", "<", "&lt;", ">", "&gt;", `"`, "&quot;")
func escapeXML(b *strings.Builder, s string) {
xmlEscaper.WriteString(b, s)
}
+189
View File
@@ -0,0 +1,189 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import (
"bytes"
"encoding/xml"
"testing"
)
// corpus is the project's own hand-written corpus: every case states a TeX
// input and the exact MathML Core the engine produces for it.
var corpus = []struct {
name string
input string
want string
}{
{"single letter", "x", `<mi>x</mi>`},
{"digits", "42", `<mn>42</mn>`},
{"decimal", "1.5", `<mn>1.5</mn>`},
{"sum", "a+b", `<mi>a</mi><mo>+</mo><mi>b</mi>`},
{"greek", `\alpha + \beta`, `<mi>α</mi><mo>+</mo><mi>β</mi>`},
{"uppercase greek", `\Gamma`, `<mi>Γ</mi>`},
{"relation", `a \leq b`, `<mi>a</mi><mo>≤</mo><mi>b</mi>`},
{"operator with limits", `\sum_{i=1}^{n} i`,
`<msubsup><mo movablelimits="true">∑</mo><mrow><mi>i</mi><mo>=</mo><mn>1</mn></mrow><mi>n</mi></msubsup><mi>i</mi>`},
{"integral keeps side scripts", `\int_0^1 x`,
`<msubsup><mo>∫</mo><mn>0</mn><mn>1</mn></msubsup><mi>x</mi>`},
{"fraction", `\frac{a}{b}`, `<mfrac><mi>a</mi><mi>b</mi></mfrac>`},
{"fraction shorthand", `\frac12`, `<mfrac><mn>1</mn><mn>2</mn></mfrac>`},
{"binom", `\binom{n}{k}`,
`<mrow><mo fence="true" stretchy="true">(</mo><mfrac linethickness="0em"><mi>n</mi><mi>k</mi></mfrac><mo fence="true" stretchy="true">)</mo></mrow>`},
{"over infix", `a \over b`, `<mfrac><mi>a</mi><mi>b</mi></mfrac>`},
{"choose infix", `a \choose b`,
`<mrow><mo fence="true" stretchy="true">(</mo><mfrac linethickness="0em"><mi>a</mi><mi>b</mi></mfrac><mo fence="true" stretchy="true">)</mo></mrow>`},
{"sqrt", `\sqrt{x}`, `<msqrt><mi>x</mi></msqrt>`},
{"root with index", `\sqrt[3]{x}`, `<mroot><mi>x</mi><mn>3</mn></mroot>`},
{"sub and sup", `x_1^2`, `<msubsup><mi>x</mi><mn>1</mn><mn>2</mn></msubsup>`},
{"single digit script", `x^10`, `<msup><mi>x</mi><mn>1</mn></msup><mn>0</mn>`},
{"group", `{xy}`, `<mrow><mi>x</mi><mi>y</mi></mrow>`},
{"left right", `\left( x \right)`,
`<mrow><mo fence="true" stretchy="true">(</mo><mi>x</mi><mo fence="true" stretchy="true">)</mo></mrow>`},
{"left right invisible", `\left. x \right)`,
`<mrow><mo fence="true"></mo><mi>x</mi><mo fence="true" stretchy="true">)</mo></mrow>`},
{"function", `\sin x`, `<mi>sin</mi><mi>x</mi>`},
{"limit under", `\lim_{x \to 0}`,
`<msub><mi movablelimits="true">lim</mi><mrow><mi>x</mi><mo>→</mo><mn>0</mn></mrow></msub>`},
{"accent", `\hat{x}`, `<mover><mi>x</mi><mo stretchy="false">^</mo></mover>`},
{"vec accent", `\vec{v}`, `<mover><mi>v</mi><mo stretchy="false">→</mo></mover>`},
{"overline", `\overline{x}`, `<mover><mi>x</mi><mo stretchy="true">‾</mo></mover>`},
{"text", `\text{if}`, `<mtext>if</mtext>`},
{"text keeps spaces", `\text{hello world}`, `<mtext>hello world</mtext>`},
{"mathrm", `\mathrm{d}x`, `<mi mathvariant="normal">d</mi><mi>x</mi>`},
{"mathbb", `\mathbb{R}`, `<mi mathvariant="double-struck">R</mi>`},
{"operatorname", `\operatorname{sgn}`, `<mi>sgn</mi>`},
{"spacing", `a \, b`, `<mi>a</mi><mspace width="0.1667em"></mspace><mi>b</mi>`},
{"quad", `a \quad b`, `<mi>a</mi><mspace width="1em"></mspace><mi>b</mi>`},
{"prime", `x'`, `<mi>x</mi><mo>′</mo>`},
{"escaped brace", `\{x\}`, `<mo>{</mo><mi>x</mi><mo>}</mo>`},
{"overset", `\overset{a}{b}`, `<mover><mi>b</mi><mi>a</mi></mover>`},
{"matrix", `\begin{matrix} a & b \\ c & d \end{matrix}`,
`<mtable><mtr><mtd><mi>a</mi></mtd><mtd><mi>b</mi></mtd></mtr><mtr><mtd><mi>c</mi></mtd><mtd><mi>d</mi></mtd></mtr></mtable>`},
{"pmatrix", `\begin{pmatrix} a \\ b \end{pmatrix}`,
`<mrow><mo fence="true" stretchy="true">(</mo><mtable><mtr><mtd><mi>a</mi></mtd></mtr><mtr><mtd><mi>b</mi></mtd></mtr></mtable><mo fence="true" stretchy="true">)</mo></mrow>`},
{"cases", `\begin{cases} a & b \\ c & d \end{cases}`,
`<mrow><mo fence="true" stretchy="true">{</mo><mtable columnalign="left"><mtr><mtd><mi>a</mi></mtd><mtd><mi>b</mi></mtd></mtr><mtr><mtd><mi>c</mi></mtd><mtd><mi>d</mi></mtd></mtr></mtable><mo fence="true"></mo></mrow>`},
{"aligned", `\begin{aligned} a &= b \\ c &= d \end{aligned}`,
`<mtable columnalign="right left"><mtr><mtd><mi>a</mi></mtd><mtd><mo>=</mo><mi>b</mi></mtd></mtr><mtr><mtd><mi>c</mi></mtd><mtd><mo>=</mo><mi>d</mi></mtd></mtr></mtable>`},
{"array spec", `\begin{array}{c|l} a & b \end{array}`,
`<mtable><mtr><mtd><mi>a</mi></mtd><mtd><mi>b</mi></mtd></mtr></mtable>`},
{"macro", `\newcommand{\R}{\mathbb{R}} \R`,
`<mi mathvariant="double-struck">R</mi>`},
{"macro with argument", `\newcommand{\ip}[2]{\langle #1, #2 \rangle} \ip{a}{b}`,
`<mo>⟨</mo><mi>a</mi><mo>,</mo><mi>b</mi><mo>⟩</mo>`},
{"def", `\def\dx{\mathrm{d}x} \dx`,
`<mi mathvariant="normal">d</mi><mi>x</mi>`},
{"escaping in text", `\text{a < b & c}`,
`<mtext>a &lt; b &amp; c</mtext>`},
{"unicode letter", `λ`, `<mi>λ</mi>`},
{"pmod", `x \pmod n`,
`<mi>x</mi><mrow><mo fence="true" stretchy="true">(</mo><mrow><mi>mod</mi><mspace width="0.2778em"></mspace><mi>n</mi></mrow><mo fence="true" stretchy="true">)</mo></mrow>`},
{"colon relation precomposed", `f \coloneqq g`,
`<mi>f</mi><mo>≔</mo><mi>g</mi>`},
{"colon relation linear", `f \coloneq g`,
`<mi>f</mi><mo>:−</mo><mi>g</mi>`},
{"colon relation double", `f \Coloneqq g`,
`<mi>f</mi><mo>∷=</mo><mi>g</mi>`},
{"eqqcolon", `a \eqqcolon b`,
`<mi>a</mi><mo>≕</mo><mi>b</mi>`},
}
func TestCorpus(t *testing.T) {
for _, tc := range corpus {
t.Run(tc.name, func(t *testing.T) {
got := string(Render([]byte(tc.input), false))
want := `<math xmlns="http://www.w3.org/1998/Math/MathML">` + tc.want + `</math>`
if got != want {
t.Errorf("input %q\ngot: %s\nwant: %s", tc.input, got, want)
}
})
}
}
// degrade holds the inputs whose constructs lie outside the mappable
// surface: nothing may disappear, everything stays as verbatim source in
// a marked element.
var degrade = []struct {
name string
input string
}{
{"unknown command", `\tikz{x}`},
{"unsupported environment", `\begin{tikzpicture} \draw (0,0); \end{tikzpicture}`},
{"unclosed group", `{x`},
{"unclosed environment", `\begin{matrix} a \end{pmatrix}`},
{"reserved character", `a # b`},
{"stray ampersand", `a & b`},
{"stray row end", `a \\ b`},
{"recursive macro", `\newcommand{\x}{\x}\x`},
{"boxed", `\boxed{x}`},
{"sideset", `\sideset{_a^b}{_c^d}\sum`},
{"tag", `\tag{1} x`},
}
func TestDegradation(t *testing.T) {
for _, tc := range degrade {
t.Run(tc.name, func(t *testing.T) {
out := Render([]byte(tc.input), false)
if !bytes.Contains(out, []byte("<merror>")) {
t.Errorf("input %q produced no merror:\n%s", tc.input, out)
}
if !wellFormed(out) {
t.Errorf("input %q produced malformed XML:\n%s", tc.input, out)
}
})
}
}
func TestVerbatimSourceKept(t *testing.T) {
out := string(Render([]byte(`a + \unknowncmd b`), false))
want := `<merror><mtext>\unknowncmd</mtext></merror>`
if !bytes.Contains([]byte(out), []byte(want)) {
t.Errorf("degraded construct lost its source:\n%s", out)
}
}
func TestDeterministicOutput(t *testing.T) {
source := []byte(`\frac{1}{2}\sqrt[3]{x}\begin{pmatrix} a & b \\ c & d \end{pmatrix}\sum_{i=1}^{n} i`)
first := Render(source, true)
for range 5 {
if next := Render(source, true); !bytes.Equal(first, next) {
t.Fatal("Render of the same source differs between calls")
}
}
}
// wellFormed checks that the output parses as XML.
func wellFormed(out []byte) bool {
dec := xml.NewDecoder(bytes.NewReader(out))
for {
_, err := dec.Token()
if err != nil {
return err.Error() == "EOF"
}
}
}
func TestAllCorpusWellFormed(t *testing.T) {
for _, tc := range corpus {
if out := Render([]byte(tc.input), false); !wellFormed(out) {
t.Errorf("input %q produced malformed XML:\n%s", tc.input, out)
}
}
}
func TestDisplayBlock(t *testing.T) {
if got := string(Render([]byte("x"), true)); got != `<math xmlns="http://www.w3.org/1998/Math/MathML" display="block"><mi>x</mi></math>` {
t.Errorf("display form = %s", got)
}
}
func TestMacroDepthBounded(t *testing.T) {
// A macro that doubles itself grows past any depth limit; the parser
// must degrade rather than hang or exhaust memory.
out := Render([]byte(`\newcommand{\a}{\a\a}`+"\n"+`\a`), false)
if !bytes.Contains(out, []byte("<merror>")) {
t.Errorf("unbounded macro produced no merror:\n%s", out)
}
}
+326
View File
@@ -0,0 +1,326 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "strings"
// parser walks the token list. The variant field is the mathvariant the
// current style command imposes on the atoms it covers.
type parser struct {
src []byte
toks []token
pos int
display bool
variant string
macros map[string]macro
expansions int
}
func newParser(src []byte, display bool) *parser {
return &parser{src: src, toks: tokenise(src), display: display, macros: map[string]macro{}}
}
func (p *parser) cur() token {
p.expandMacros()
return p.toks[p.pos]
}
func (p *parser) at(kind tokenKind) bool { return p.cur().kind == kind }
func (p *parser) atCommand(name string) bool {
t := p.cur()
return t.kind == tokCommand && t.text == `\`+name
}
// parseAll parses the whole source into nodes. A boundary with nothing to
// bound degrades as the stray it is.
func (p *parser) parseAll() []*node {
nodes := p.sequence(true)
for p.cur().kind != tokEOF {
if isBoundary(p.cur()) {
t := p.cur()
p.pos++
nodes = append(nodes, errorNode(t.text))
continue
}
nodes = append(nodes, p.sequence(true)...)
}
return nodes
}
// isBoundary reports whether the token ends a sequence: a cell separator,
// a row separator, an environment close or a closing brace.
func isBoundary(t token) bool {
if t.kind == tokRBrace || t.kind == tokAmpersand || isRowEnd(t) {
return true
}
return t.kind == tokCommand && t.text == `\end`
}
func isRowEnd(t token) bool {
return t.kind == tokCommand && t.text == `\\`
}
// sequence parses atoms until a boundary or the end. When allowOver is
// set, the TeX infix constructs \over, \atop and \choose may appear and
// take everything parsed so far as their numerator.
func (p *parser) sequence(allowOver bool) []*node {
var nodes []*node
for {
t := p.cur()
if t.kind == tokEOF || isBoundary(t) {
break
}
if t.kind == tokDegraded {
nodes = append(nodes, errorNode(t.text))
p.pos++
continue
}
if t.kind == tokCommand && isOverCommand(t.text) {
p.pos++
if !allowOver {
nodes = append(nodes, errorNode(t.text))
continue
}
num := wrapRow(nodes)
den := wrapRow(p.sequence(false))
nodes = []*node{p.overNode(t.text, num, den)}
continue
}
if n := p.atom(); n != nil {
nodes = append(nodes, n)
}
}
return nodes
}
func isOverCommand(text string) bool {
switch text {
case `\over`, `\atop`, `\choose`, `\overwithdelims`, `\atopwithdelims`, `\abovewithdelims`:
return true
}
return false
}
func (p *parser) overNode(cmd string, num, den *node) *node {
switch cmd {
case `\choose`:
return fenced("(", elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den), ")")
case `\atop`:
return elA("mfrac", []attribute{{"linethickness", "0em"}}, num, den)
default:
return el("mfrac", num, den)
}
}
// atom parses one unit with its scripts, or a whole command construct.
func (p *parser) atom() *node {
t := p.cur()
switch t.kind {
case tokLBrace:
return p.braceGroup()
case tokChar:
return p.scripts(p.charAtom(t))
case tokCommand:
n := p.command()
if n == nil {
return nil
}
return p.scripts(n)
}
p.pos++
return errorNode(t.text)
}
// braceGroup parses a braced group, degrading the whole span when the
// closing brace never comes. A group of one node is that node: the braces
// only grouped.
func (p *parser) braceGroup() *node {
open := p.cur()
p.pos++
nodes := p.sequence(true)
if p.at(tokRBrace) {
p.pos++
if len(nodes) == 1 {
return nodes[0]
}
if len(nodes) == 0 {
return el("mrow")
}
row := el("mrow")
row.children = nodes
return row
}
p.pos = len(p.toks) - 1
return errorNode(string(p.src[open.start:]))
}
// charAtom maps one character token to its element. The reserved TeX
// characters degrade rather than pass as content.
func (p *parser) charAtom(t token) *node {
c := t.text
switch c {
case "#", "$", "%":
p.pos++
return errorNode(c)
case "~":
p.pos++
return spaceNode("0.25em")
case "'":
p.pos++
return text("mo", "′")
}
r := []rune(c)[0]
switch {
case isDigitByte(c[0]) && len(c) == 1:
p.pos++
return p.numberRun(c)
case isIdentifierRune(r):
p.pos++
return p.identifier(c)
default:
p.pos++
return p.withVariant(el("mo"), c)
}
}
// numberRun collects the digits and decimal points that follow.
func (p *parser) numberRun(first string) *node {
var b strings.Builder
b.WriteString(first)
for {
t := p.cur()
if t.kind != tokChar || len(t.text) != 1 || !isDigitByte(t.text[0]) && t.text != "." {
break
}
b.WriteString(t.text)
p.pos++
}
return p.withVariant(el("mn"), b.String())
}
// identifier renders one letter, upright when the variant says so.
func (p *parser) identifier(c string) *node {
return p.withVariant(el("mi"), c)
}
// withVariant fills a leaf node with text, applying the active variant.
func (p *parser) withVariant(n *node, c string) *node {
n.text = c
if p.variant != "" && (n.kind == "mi" || n.kind == "mn" || n.kind == "mo") {
if !(n.kind == "mi" && p.variant == "italic") {
n.attrs = append(n.attrs, attribute{"mathvariant", p.variant})
}
}
return n
}
// scripts attaches the sub and superscript runs that follow a base. A
// movable base puts its scripts under and over in display style; inline,
// the movablelimits attribute lets the renderer decide.
func (p *parser) scripts(base *node) *node {
movable := baseMovable(base)
if p.atCommand("limits") {
p.pos++
movable = true
} else if p.atCommand("nolimits") {
p.pos++
movable = false
}
var sub, sup *node
for {
t := p.cur()
if t.kind == tokUnderscore && sub == nil {
p.pos++
sub = p.argument()
continue
}
if t.kind == tokCaret && sup == nil {
p.pos++
sup = p.argument()
continue
}
break
}
under := movable && p.display
switch {
case sub == nil && sup == nil:
return base
case under && sub != nil && sup != nil:
return el("munderover", base, sub, sup)
case under && sub != nil:
return el("munder", base, sub)
case under && sup != nil:
return el("mover", base, sup)
case sub != nil && sup != nil:
return el("msubsup", base, sub, sup)
case sub != nil:
return el("msub", base, sub)
default:
return el("msup", base, sup)
}
}
// baseMovable reports whether the base carries movable limits.
func baseMovable(base *node) bool {
if base.kind != "mo" {
return false
}
for _, a := range base.attrs {
if a.key == "movablelimits" {
return a.value == "true"
}
}
return false
}
// argument reads one macro argument or script: a braced group or a single
// token. A single digit stays one digit, the way TeX reads x^10.
func (p *parser) argument() *node {
t := p.cur()
switch {
case t.kind == tokLBrace:
return p.braceGroup()
case t.kind == tokChar && len(t.text) == 1 && isDigitByte(t.text[0]):
p.pos++
return p.withVariant(el("mn"), t.text)
case t.kind == tokChar || t.kind == tokCommand:
n := p.atom()
if n == nil {
return el("mrow")
}
return n
}
return errorNode(t.text)
}
// rawBraced reads a braced group from the source as verbatim text,
// keeping the spaces. It fails when the closing brace is missing.
func (p *parser) rawBraced() (string, bool) {
t := p.cur()
if t.kind != tokLBrace {
return "", false
}
depth := 0
for i := t.start; i < len(p.src); {
switch p.src[i] {
case '{':
depth++
case '}':
depth--
if depth == 0 {
text := string(p.src[t.start+1 : i])
// consume the tokens the span covers
for p.pos < len(p.toks) && p.toks[p.pos].start <= i {
p.pos++
}
return text, true
}
case '\\':
i++ // an escaped character never opens or closes
}
i++
}
return "", false
}
+349
View File
@@ -0,0 +1,349 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
// symbol names the MathML element a bare command becomes and whether its
// scripts sit under and over it.
type symbol struct {
char string
mi bool
movable bool
}
func op(char string) symbol { return symbol{char: char} }
func mov(char string) symbol { return symbol{char: char, movable: true} }
func id(char string) symbol { return symbol{char: char, mi: true} }
// symbols holds the commands that stand for one character: Greek letters,
// letterlike symbols, binary operators, relations, arrows, delimiters, big
// operators and miscellaneous symbols.
var symbols = map[string]symbol{
// Greek letters, lowercase.
"alpha": id("α"), "beta": id("β"), "gamma": id("γ"), "delta": id("δ"),
"epsilon": id("ϵ"), "varepsilon": id("ε"), "zeta": id("ζ"), "eta": id("η"),
"theta": id("θ"), "vartheta": id("ϑ"), "iota": id("ι"), "kappa": id("κ"),
"lambda": id("λ"), "mu": id("μ"), "nu": id("ν"), "xi": id("ξ"),
"omicron": id("ο"), "pi": id("π"), "varpi": id("ϖ"), "rho": id("ρ"),
"varrho": id("ϱ"), "sigma": id("σ"), "varsigma": id("ς"), "tau": id("τ"),
"upsilon": id("υ"), "phi": id("ϕ"), "varphi": id("φ"), "chi": id("χ"),
"psi": id("ψ"), "omega": id("ω"), "digamma": id("ϝ"),
// Greek letters, uppercase; the ones that coincide with Latin capitals
// still stand as identifiers, and the var forms are the italic capitals.
"Gamma": id("Γ"), "Delta": id("Δ"), "Theta": id("Θ"), "Lambda": id("Λ"),
"Xi": id("Ξ"), "Pi": id("Π"), "Sigma": id("Σ"), "Upsilon": id("Υ"),
"Phi": id("Φ"), "Psi": id("Ψ"), "Omega": id("Ω"),
"Alpha": id("Α"), "Beta": id("Β"), "Epsilon": id("Ε"), "Zeta": id("Ζ"),
"Eta": id("Η"), "Iota": id("Ι"), "Kappa": id("Κ"), "Mu": id("Μ"),
"Nu": id("Ν"), "Omicron": id("Ο"), "Rho": id("Ρ"), "Tau": id("Τ"),
"Chi": id("Χ"),
"varGamma": id("𝛤"), "varDelta": id("𝛥"), "varTheta": id("𝛩"),
"varLambda": id("𝛬"), "varXi": id("𝛯"), "varPi": id("𝛱"),
"varSigma": id("𝛴"), "varUpsilon": id("𝛶"), "varPhi": id("𝛷"),
"varPsi": id("𝛹"), "varOmega": id("𝛺"),
"varkappa": id("ϰ"), "thetasym": id("ϑ"),
// Letterlike symbols.
"hbar": id("ℏ"), "ell": id("ℓ"), "imath": id("ı"), "jmath": id("ȷ"),
"wp": id("℘"), "Re": id("ℜ"), "Im": id("ℑ"), "aleph": id("ℵ"),
"beth": id("ℶ"), "gimel": id("ℷ"), "daleth": id("ℸ"),
"partial": id("∂"), "nabla": op("∇"), "mho": id("℧"),
"complement": op("∁"), "eth": id("ð"), "Finv": id("⅁"), "Game": id("⅂"),
"circledS": id("Ⓢ"), "Bbbk": id("𝕜"), "weierp": id("℘"),
// Big operators, with movable limits.
"sum": mov("∑"), "prod": mov("∏"), "coprod": mov("∐"),
"int": op("∫"), "iint": op("∬"), "iiint": op("∭"), "iiiint": op("⨌"),
"oint": op("∮"), "oiint": op("∯"), "oiiint": op("∰"),
"bigcap": mov("⋂"), "bigcup": mov("⋃"), "bigsqcup": mov("⨆"),
"bigvee": mov("⋁"), "bigwedge": mov("⋀"), "bigodot": mov("⨀"),
"bigotimes": mov("⨂"), "bigoplus": mov("⨁"), "biguplus": mov("⨄"),
"varointclockwise": op("∱"), "ointclockwise": op("∲"),
"ointctrclockwise": op("∳"),
// Binary operators.
"pm": op("±"), "mp": op("∓"), "times": op("×"), "div": op("÷"),
"ast": op("∗"), "star": op("⋆"), "circ": op("∘"), "bullet": op("∙"),
"cdot": op("⋅"), "cap": op("∩"), "cup": op("∪"), "uplus": op("⊎"),
"sqcap": op("⊓"), "sqcup": op("⊔"), "vee": op("∨"), "lor": op("∨"),
"wedge": op("∧"), "land": op("∧"), "setminus": op("∖"),
"wr": op("≀"), "diamond": op("⋄"), "bigtriangleup": op("△"),
"bigtriangledown": op("▽"), "triangleleft": op("◃"), "triangleright": op("▹"),
"lhd": op("⊲"), "rhd": op("⊳"), "unlhd": op("⊴"), "unrhd": op("⊵"),
"oplus": op("⊕"), "ominus": op("⊖"), "otimes": op("⊗"), "oslash": op("⊘"),
"odot": op("⊙"), "bigcirc": op("○"), "dagger": op("†"), "ddagger": op("‡"),
"amalg": op("⨿"), "dotplus": op("∔"), "smallsetminus": op("∖"),
"Cap": op("⋒"), "Cup": op("⋓"), "barwedge": op("⊼"), "veebar": op("⊻"),
"doublebarwedge": op("⩞"), "boxminus": op("⊟"), "boxplus": op("⊞"),
"boxtimes": op("⊠"), "boxdot": op("⊡"), "divideontimes": op("⋇"),
"intercal": op("⊺"), "circledcirc": op("⊚"), "circledast": op("⊛"),
"circleddash": op("⊝"), "curlywedge": op("⋏"), "curlyvee": op("⋎"),
"leftthreetimes": op("⋋"), "rightthreetimes": op("⋌"),
"looparrowleft": op("↫"), "looparrowright": op("↬"),
"curvearrowleft": op("↶"), "curvearrowright": op("↷"),
"circlearrowleft": op("↺"), "circlearrowright": op("↻"),
// Relations.
"leq": op("≤"), "le": op("≤"), "geq": op("≥"), "ge": op("≥"),
"neq": op("≠"), "ne": op("≠"), "sim": op("∼"), "simeq": op("≃"),
"approx": op("≈"), "cong": op("≅"), "equiv": op("≡"),
"prec": op("≺"), "preceq": op("⪯"), "precapprox": op("⪵"),
"precsim": op("≾"), "succ": op("≻"), "succeq": op("⪰"),
"succapprox": op("⪶"), "succsim": op("≿"),
"ll": op("≪"), "lll": op("⋘"), "gg": op("≫"), "ggg": op("⋙"),
"asymp": op("≍"), "doteq": op("≐"), "propto": op("∝"),
"mid": op("∣"), "nmid": op("∤"), "parallel": op("∥"), "shortparallel": op("∥"),
"perp": op("⊥"), "Subset": op("⋐"), "Supset": op("⋑"),
"sqsubset": op("⊏"), "sqsupset": op("⊐"),
"subset": op("⊂"), "supset": op("⊃"), "subseteq": op("⊆"), "supseteq": op("⊇"),
"subseteqq": op("⫅"), "supseteqq": op("⫆"),
"sqsubseteq": op("⊑"), "sqsupseteq": op("⊒"),
"in": op("∈"), "ni": op("∋"), "owns": op("∋"), "notin": op("∉"),
"vdash": op("⊢"), "dashv": op("⊣"), "Vdash": op("⊩"), "Vvdash": op("⊪"),
"models": op("⊨"), "smile": op("⌣"), "frown": op("⌢"),
"lesssim": op("≲"), "gtrsim": op("≳"), "lessapprox": op("⪅"),
"gtrapprox": op("⪆"), "lessgtr": op("≶"), "gtrless": op("≷"),
"lesseqgtr": op("⋚"), "gtreqless": op("⋛"), "leqq": op("≦"),
"geqq": op("≧"), "lneq": op("⪇"), "gneq": op("⪈"), "lvertneqq": op("≨"),
"gvertneqq": op("≩"), "lnsim": op("⪦"), "gnsim": op("⪧"),
"eqslantless": op("⪕"), "eqslantgtr": op("⪖"), "backsim": op("∽"),
"backsimeq": op("⋍"), "lesseqqgtr": op("⪋"), "gtreqqless": op("⪌"),
"nestedlessgreater": op("≺"), "nless": op("≮"), "ngtr": op("≯"),
"nleq": op("≰"), "nleqslant": op("≰"), "ngeq": op("≱"), "ngeqslant": op("≱"),
"nprec": op("⊀"), "nsucc": op("⊁"), "precnsim": op("⋨"), "succnsim": op("⋩"),
"nsubseteq": op("⊈"), "nsupseteq": op("⊉"), "subsetneq": op("⊊"),
"supsetneq": op("⊋"), "subsetneqq": op("⫋"), "supsetneqq": op("⫌"),
"vartriangleleft": op("⊲"), "vartriangleright": op("⊳"),
"trianglelefteq": op("⊴"), "trianglerighteq": op("⊵"),
"triangleq": op("≜"), "bumpeq": op("≏"), "Bumpeq": op("≎"),
"eqcirc": op("≖"), "circeq": op("≗"), "doteqdot": op("≑"),
"risingdotseq": op("≓"), "fallingdotseq": op("≒"),
"pitchfork": op("⋔"), "smallfrown": op("⌢"), "smallsmile": op("⌣"),
"therefore": op("∴"), "because": op("∵"),
"eqsim": op("≟"),
"bowtie": op("⋈"), "Join": op("⋈"), "backepsilon": op("϶"),
"thicksim": op("∼"), "thickapprox": op("≈"),
"preccurlyeq": op("≼"), "succcurlyeq": op("≽"),
"varpropto": op("∝"), "ratio": op("∶"), "vcentcolon": op(":"),
"curlyeqprec": op("⋞"), "curlyeqsucc": op("⋟"),
"between": op("≬"),
// The colon relations, mapped the way KaTeX's MathML branch maps
// them: the precomposed character where Unicode has one, the linear
// two-character operator where it has none.
"dblcolon": op("∷"),
"coloneqq": op("≔"),
"coloneq": op(":−"),
"Coloneqq": op("∷="),
"Coloneq": op("∷−"),
"eqqcolon": op("≕"),
"eqcolon": op("∹"),
"Eqqcolon": op("=∷"),
"Eqcolon": op("−∷"),
"colonapprox": op(":≈"),
"Colonapprox": op("∷≈"),
"colonsim": op(":∼"),
"Colonsim": op("∷∼"),
"approxcolon": op("≈:"),
"approxcoloncolon": op("≈∷"),
"simcolon": op("∼:"),
"simcoloncolon": op("∼∷"),
"origof": op("⊶"),
"imageof": op("⊷"),
// The last stragglers the full KaTeX symbol table carries.
"Doteq": op("≑"),
"Diamond": op("◆"),
"approxeq": op("≊"),
"doublecap": op("⋒"),
"doublecup": op("⋓"),
"geqslant": op("⩾"),
"leqslant": op("⩽"),
"gggtr": op("⋙"),
"llless": op("⋘"),
"gneqq": op("≩"),
"lneqq": op("≨"),
"gtrdot": op("⋗"),
"lessdot": op("⋖"),
"intop": op("∫"),
"smallint": op("∫"),
"ltimes": op("⋉"),
"rtimes": op("⋊"),
"nparallel": op("∦"),
"nsim": op("≁"),
"nvdash": op("⊬"),
"vDash": op("⊨"),
"shortmid": op("∣"),
"varvdots": op("⋮"),
// Negated relations.
"ncong": op("≇"), "npreceq": op("⋠"),
"nsucceq": op("⋡"),
"precnapprox": op("⪹"), "succnapprox": op("⪺"),
"precneqq": op("⪵"), "succneqq": op("⪶"),
"gnapprox": op("⪊"), "lnapprox": op("⪉"),
"nshortmid": op("∤"), "nshortparallel": op("∦"),
"nvDash": op("⊭"), "nVDash": op("⊯"), "nVdash": op("⊮"),
"ntriangleleft": op("⋪"), "ntriangleright": op("⋫"),
"ntrianglelefteq": op("⋬"), "ntrianglerighteq": op("⋭"),
"nleftrightarrow": op("↮"), "nLeftarrow": op("⇍"),
"nLeftrightarrow": op("⇎"), "nRightarrow": op("⇏"),
"nleftarrow": op("↰"), "nrightarrow": op("↱"),
// Arrows.
"leftarrow": op("←"), "gets": op("←"), "Leftarrow": op("⇐"),
"rightarrow": op("→"), "to": op("→"), "Rightarrow": op("⇒"),
"leftrightarrow": op("↔"), "Leftrightarrow": op("⇔"), "iff": op("⟺"),
"longleftarrow": op("⟵"), "Longleftarrow": op("⟸"),
"longrightarrow": op("⟶"), "Longrightarrow": op("⟹"),
"longleftrightarrow": op("⟷"), "Longleftrightarrow": op("⟺"),
"implies": op("⟹"), "impliedby": op("⟸"),
"mapsto": op("↦"), "longmapsto": op("⟼"),
"hookleftarrow": op("↩"), "hookrightarrow": op("↪"),
"leftharpoonup": op("↼"), "leftharpoondown": op("↽"),
"rightharpoonup": op("⇀"), "rightharpoondown": op("⇁"),
"rightleftharpoons": op("⇌"), "leadsto": op("↝"),
"nearrow": op("↗"), "searrow": op("↘"), "swarrow": op("↙"), "nwarrow": op("↖"),
"uparrow": op("↑"), "downarrow": op("↓"), "updownarrow": op("↕"),
"Uparrow": op("⇑"), "Downarrow": op("⇓"), "Updownarrow": op("⇕"),
"downdownarrows": op("⇊"), "upuparrows": op("⇈"),
"rightrightarrows": op("⇉"), "leftleftarrows": op("⇇"),
"rightleftarrows": op("⇄"), "leftrightarrows": op("⇆"),
"twoheadrightarrow": op("↠"), "twoheadleftarrow": op("↞"),
"leftarrowtail": op("↢"), "rightarrowtail": op("↣"),
"Lleftarrow": op("⤅"), "Rrightarrow": op("⤇"),
"upharpoonleft": op("↿"), "upharpoonright": op("↾"),
"downharpoonleft": op("⇃"), "downharpoonright": op("⇂"),
"restriction": op("↾"), "multimap": op("⊸"),
"harr": op("↔"), "hArr": op("⇔"), "Harr": op("⇔"),
"larr": op("←"), "lArr": op("⇐"), "Larr": op("⇐"),
"rarr": op("→"), "rArr": op("⇒"), "Rarr": op("⇒"),
"lrarr": op("↔"), "lrArr": op("⇔"), "Lrarr": op("⇔"),
"darr": op("↓"), "dArr": op("⇓"), "Darr": op("⇓"),
"uarr": op("↑"), "uArr": op("⇑"), "Uarr": op("⇑"),
"dashleftarrow": op("⇠"), "dashrightarrow": op("⇢"),
"rightsquigarrow": op("↝"), "leftrightsquigarrow": op("↭"),
"leftrightharpoons": op("⇋"), "mapsfrom": op("↤"),
"Lsh": op("↰"), "Rsh": op("↱"),
// Delimiters.
"langle": op("⟨"), "rangle": op("⟩"), "lfloor": op("⌊"), "rfloor": op("⌋"),
"lceil": op("⌈"), "rceil": op("⌉"), "vert": op("|"), "lvert": op("|"),
"rvert": op("|"), "Vert": op("‖"), "lVert": op("‖"), "rVert": op("‖"),
"lbrace": op("{"), "rbrace": op("}"), "lbrack": op("["), "rbrack": op("]"),
"lgroup": op("⟮"), "rgroup": op("⟯"),
"lmoustache": op("⌠"), "rmoustache": op("⌡"), "backslash": op("\\"),
"lparen": op("("), "rparen": op(")"), "lang": op("⟨"), "rang": op("⟩"),
"ulcorner": op("⌜"), "urcorner": op("⌝"), "llcorner": op("⌞"),
"lrcorner": op("⌟"), "llbracket": op("⟦"), "rrbracket": op("⟧"),
"lBrace": op("{"), "rBrace": op("}"),
// Miscellaneous symbols.
"infty": op("∞"), "forall": op("∀"), "exists": op("∃"), "nexists": op("∄"),
"emptyset": op("∅"), "varnothing": op("∅"), "top": op("⊤"), "bot": op("⊥"),
"vdots": op("⋮"), "cdots": op("⋯"), "ddots": op("⋱"), "iddots": op("⋰"),
"ldots": op("…"), "dots": op("…"), "dotsc": op("…"), "dotsb": op("⋯"),
"dotsm": op("⋯"), "dotsi": op("⋯"), "dotso": op("…"),
"prime": op("′"), "backprime": op("‵"), "degree": op("°"),
"angle": op("∠"), "measuredangle": op("∡"), "sphericalangle": op("∢"),
"triangle": op("△"), "square": op("□"), "blacksquare": op("■"),
"bigstar": op("★"), "blacktriangle": op("▲"), "blacktriangledown": op("▼"),
"blacktriangleleft": op("◀"), "blacktriangleright": op("▶"),
"diamondsuit": op("♦"), "heartsuit": op("♥"), "clubsuit": op("♣"),
"spadesuit": op("♠"), "flat": op("♭"), "natural": op("♮"), "sharp": op("♯"),
"checkmark": op("✓"), "maltese": op("✠"), "bull": op("∙"),
"ldotp": op("."), "cdotp": op("⋅"), "colon": op(":"),
"S": op("§"), "P": op("¶"), "copyright": op("©"), "circledR": op("®"),
"diagup": op("╱"), "diagdown": op("╲"),
"lozenge": op("◊"), "blacklozenge": op("◆"), "surd": op("√"),
"Box": op("□"), "triangledown": op("▽"), "vartriangle": op("△"),
"pounds": op("£"), "mathsterling": op("£"), "yen": op("¥"),
"dag": op("†"), "ddag": op("‡"), "Dagger": op("‡"),
"minuso": op("⦵"), "centerdot": op("·"), "plusmn": op("±"),
"And": op("&"), "lq": op("‘"), "rq": op("’"),
"sdot": op("⋅"), "mathellipsis": op("…"),
"neg": op("¬"), "lnot": op("¬"), "empty": op("∅"),
"isin": op("∈"), "exist": op("∃"),
"lt": op("<"), "gt": op(">"),
// Letter-like aliases KaTeX carries.
"alef": id("ℵ"), "alefsym": id("ℵ"), "hslash": id("ℏ"),
"image": id("ℑ"), "real": id("ℜ"), "reals": id("ℝ"),
"cnums": id("ℂ"), "Complex": id("ℂ"), "natnums": id("ℕ"),
"RR": id("ℝ"), "NN": id("ℕ"), "ZZ": id("ℤ"), "Q": id("ℚ"),
"infin": op("∞"),
}
// functions are the names typeset upright as identifiers.
var functions = map[string]bool{
"arccos": true, "arcsin": true, "arctan": true, "arg": true,
"cos": true, "cosh": true, "cot": true, "coth": true, "csc": true,
"deg": true, "det": true, "dim": true, "exp": true, "gcd": true,
"hom": true, "ker": true, "lg": true, "ln": true, "log": true,
"Pr": true, "sec": true, "sin": true, "sinh": true, "tan": true,
"tanh": true, "arcsinh": true, "arccosh": true, "arctanh": true,
"argmax": true, "argmin": true,
"mod": true, "bmod": true,
"min": true, "max": true, "sup": true, "inf": true,
"lim": true, "limsup": true, "liminf": true,
"injlim": true, "projlim": true, "varinjlim": true, "varprojlim": true,
"varliminf": true, "varlimsup": true, "plim": true,
"arctg": true, "arcctg": true, "ch": true, "cosec": true, "cotg": true,
"ctg": true, "cth": true, "sh": true, "tg": true, "th": true,
}
// movableFunctions take their scripts under and over: the limit operators.
var movableFunctions = map[string]bool{
"lim": true, "limsup": true, "liminf": true, "max": true, "min": true,
"sup": true, "inf": true, "gcd": true, "det": true, "Pr": true,
"injlim": true, "projlim": true, "varinjlim": true, "varprojlim": true,
"varliminf": true, "varlimsup": true, "plim": true,
"argmax": true, "argmin": true,
}
// accents put a mark over or under their argument.
var accents = map[string]struct {
char string
under bool
}{
"hat": {"\u005e", false}, "widehat": {"\u005e", false},
"tilde": {"~", false}, "widetilde": {"~", false},
"utilde": {"~", true},
"bar": {"\u00af", false}, "overline": {"\u203e", false},
"vec": {"\u2192", false}, "dot": {"\u02d9", false}, "ddot": {"\u00a8", false},
"dddot": {"\u20db", false}, "ddddot": {"\u20dc", false},
"mathring": {"\u02da", false}, "breve": {"\u02d8", false},
"check": {"\u02c7", false}, "widecheck": {"\u02c7", false},
"acute": {"\u00b4", false}, "grave": {"\u0060", false},
"overbrace": {"\u23de", false}, "underbrace": {"\u23df", true},
"overbracket": {"\u23b4", false}, "underbracket": {"\u23b5", true},
"overleftarrow": {"\u2190", false}, "overrightarrow": {"\u2192", false},
"Overrightarrow": {"\u21d2", false},
"underleftarrow": {"\u2190", true}, "underrightarrow": {"\u2192", true},
"overleftrightarrow": {"\u2194", false}, "underleftrightarrow": {"\u2194", true},
"overleftharpoon": {"\u21bc", false}, "overrightharpoon": {"\u21c0", false},
"overgroup": {"\u23e0", false}, "undergroup": {"\u23e1", true},
"underline": {"_", true}, "underbar": {"\u02cd", true},
"overlinesegment": {"\u23af", false}, "underlinesegment": {"\u23af", true},
}
// styles map the style commands to mathvariant values; the empty value
// marks a switch that renders its argument unchanged.
var styles = map[string]string{
"mathrm": "normal", "mathnormal": "italic", "mathit": "italic",
"mathbf": "bold", "mathbfit": "bold-italic", "mathbb": "double-struck",
"mathcal": "script", "mathscr": "script", "mathfrak": "fraktur",
"mathsf": "sans-serif", "mathsfit": "sans-serif-italic",
"mathsfbf": "sans-serif-bold", "mathtt": "monospace",
"boldsymbol": "bold-italic", "bm": "bold-italic",
// The old TeX switches, applied to what follows in the group.
"rm": "normal", "bf": "bold", "it": "italic", "sf": "sans-serif",
"tt": "monospace", "cal": "script", "scr": "script", "frak": "fraktur",
}
// spaces maps spacing commands to mspace widths; the empty width carries
// nothing.
var spaces = map[string]string{
",": "0.1667em", "thinspace": "0.1667em",
":": "0.2222em", "medspace": "0.2222em",
";": "0.2778em", "thickspace": "0.2778em",
"!": "-0.1667em", "negthinspace": "-0.1667em",
"negmedspace": "-0.2222em", "negthickspace": "-0.2778em",
" ": "0.25em", "quad": "1em", "qquad": "2em", "enspace": "0.5em",
}
// delimiterChars are the single characters accepted after \left, \right
// and the big size commands.
var delimiterChars = map[string]bool{
"(": true, ")": true, "[": true, "]": true, "|": true, "/": true,
"<": true, ">": true,
}
+85
View File
@@ -0,0 +1,85 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "unicode/utf8"
type tokenKind uint8
const (
tokEOF tokenKind = iota
tokCommand
tokChar
tokLBrace
tokRBrace
tokCaret
tokUnderscore
tokAmpersand
// tokDegraded stands for a span the parser gave up on: the parser
// renders it as its verbatim source.
tokDegraded
)
type token struct {
kind tokenKind
text string
start int
end int
}
// tokenise splits source into TeX tokens. Whitespace between tokens is
// dropped: math mode ignores it, and the text commands read their argument
// from the raw source instead.
func tokenise(src []byte) []token {
var toks []token
i := 0
for i < len(src) {
c := src[i]
start := i
switch {
case c == '\\' && i+1 < len(src):
i++
if isLetter(src[i]) {
for i < len(src) && isLetter(src[i]) {
i++
}
text := string(src[start:i])
// A control word eats the spaces behind it, without them
// becoming part of its name.
for i < len(src) && (src[i] == ' ' || src[i] == '\t' || src[i] == '\n') {
i++
}
toks = append(toks, token{kind: tokCommand, text: text, start: start, end: i})
} else {
i++
toks = append(toks, token{kind: tokCommand, text: string(src[start:i]), start: start, end: i})
}
case c == '{':
i++
toks = append(toks, token{kind: tokLBrace, text: "{", start: start, end: i})
case c == '}':
i++
toks = append(toks, token{kind: tokRBrace, text: "}", start: start, end: i})
case c == '^':
i++
toks = append(toks, token{kind: tokCaret, text: "^", start: start, end: i})
case c == '_':
i++
toks = append(toks, token{kind: tokUnderscore, text: "_", start: start, end: i})
case c == '&':
i++
toks = append(toks, token{kind: tokAmpersand, text: "&", start: start, end: i})
case c == ' ' || c == '\t' || c == '\n' || c == '\r':
i++
default:
_, size := utf8.DecodeRune(src[i:])
i += size
toks = append(toks, token{kind: tokChar, text: string(src[start:i]), start: start, end: i})
}
}
toks = append(toks, token{kind: tokEOF, start: len(src), end: len(src)})
return toks
}
func isLetter(c byte) bool { return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' }