// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package mathml import "unicode/utf8" type tokenKind uint8 const ( tokEOF tokenKind = iota tokCommand tokChar tokLBrace tokRBrace tokCaret tokUnderscore tokAmpersand // tokDegraded stands for a span the parser gave up on: the parser // renders it as its verbatim source. tokDegraded ) type token struct { kind tokenKind text string start int end int } // tokenise splits source into TeX tokens. Whitespace between tokens is // dropped: math mode ignores it, and the text commands read their argument // from the raw source instead. func tokenise(src []byte) []token { var toks []token i := 0 for i < len(src) { c := src[i] start := i switch { case c == '\\' && i+1 < len(src): i++ if isLetter(src[i]) { for i < len(src) && isLetter(src[i]) { i++ } text := string(src[start:i]) // A control word eats the spaces behind it, without them // becoming part of its name. for i < len(src) && (src[i] == ' ' || src[i] == '\t' || src[i] == '\n') { i++ } toks = append(toks, token{kind: tokCommand, text: text, start: start, end: i}) } else { i++ toks = append(toks, token{kind: tokCommand, text: string(src[start:i]), start: start, end: i}) } case c == '{': i++ toks = append(toks, token{kind: tokLBrace, text: "{", start: start, end: i}) case c == '}': i++ toks = append(toks, token{kind: tokRBrace, text: "}", start: start, end: i}) case c == '^': i++ toks = append(toks, token{kind: tokCaret, text: "^", start: start, end: i}) case c == '_': i++ toks = append(toks, token{kind: tokUnderscore, text: "_", start: start, end: i}) case c == '&': i++ toks = append(toks, token{kind: tokAmpersand, text: "&", start: start, end: i}) case c == ' ' || c == '\t' || c == '\n' || c == '\r': i++ default: _, size := utf8.DecodeRune(src[i:]) i += size toks = append(toks, token{kind: tokChar, text: string(src[start:i]), start: start, end: i}) } } toks = append(toks, token{kind: tokEOF, start: len(src), end: len(src)}) return toks } func isLetter(c byte) bool { return c >= 'a' && c <= 'z' || c >= 'A' && c <= 'Z' }