Files

86 lines
2.2 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package mathml
import "unicode/utf8"
type tokenKind uint8
const (
tokEOF tokenKind = iota
tokCommand
tokChar
tokLBrace
tokRBrace
tokCaret
tokUnderscore
tokAmpersand
// tokDegraded stands for a span the parser gave up on: the parser
// renders it as its verbatim source.
tokDegraded
)
type token struct {
kind tokenKind
text string
start int
end int
}
// tokenise splits source into TeX tokens. Whitespace between tokens is
// dropped: math mode ignores it, and the text commands read their argument
// from the raw source instead.
func tokenise(src []byte) []token {
var toks []token
i := 0
for i < len(src) {
c := src[i]
start := i
switch {
case c == '\\' && i+1 < len(src):
i++
if isLetter(src[i]) {
for i < len(src) && isLetter(src[i]) {
i++
}
text := string(src[start:i])
// A control word eats the spaces behind it, without them
// becoming part of its name.
for i < len(src) && (src[i] == ' ' || src[i] == '\t' || src[i] == '\n') {
i++
}
toks = append(toks, token{kind: tokCommand, text: text, start: start, end: i})
} else {
i++
toks = append(toks, token{kind: tokCommand, text: string(src[start:i]), start: start, end: i})
}
case c == '{':
i++
toks = append(toks, token{kind: tokLBrace, text: "{", start: start, end: i})
case c == '}':
i++
toks = append(toks, token{kind: tokRBrace, text: "}", start: start, end: i})
case c == '^':
i++
toks = append(toks, token{kind: tokCaret, text: "^", start: start, end: i})
case c == '_':
i++
toks = append(toks, token{kind: tokUnderscore, text: "_", start: start, end: i})
case c == '&':
i++
toks = append(toks, token{kind: tokAmpersand, text: "&", start: start, end: i})
case c == ' ' || c == '\t' || c == '\n' || c == '\r':
i++
default:
_, size := utf8.DecodeRune(src[i:])
i += size
toks = append(toks, token{kind: tokChar, text: string(src[start:i]), start: start, end: i})
}
}
toks = append(toks, token{kind: tokEOF, start: len(src), end: len(src)})
return toks
}
func isLetter(c byte) bool { return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' }