Files
interpres/parser.go
T
petrbalvin 10d49fbe60
Test / test (push) Canceled after 2m28s
feat: carry the byte offset and column in SyntaxError
Assisted-by: GLM 5.3 Flash
2026-09-21 23:55:58 +02:00

1488 lines
38 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"bytes"
"context"
"fmt"
"strconv"
"strings"
"unicode/utf8"
)
// ctxCheckInterval is the number of top-level parser iterations between
// context-cancellation checks. A small interval keeps the response snappy on
// cancellation; a too-small one wastes cycles on a non-cancelled run.
const ctxCheckInterval = 64
// parser is a recursive-descent TOML parser producing a map[string]any tree.
//
// The scanner works on bytes, not runes: every character that drives the
// grammar (quotes, separators, newlines, bare-key characters) is ASCII, the
// scan validates a multi-byte sequence where it meets one, and multi-byte
// runes matter only as content, where they are decoded on the spot. Holding
// the source as []rune instead would cost a conversion pass plus four bytes
// per rune of extra memory before parsing even starts.
type parser struct {
src []byte
pos int
line int
ctx context.Context
// maxDepth and depth bound the nesting the recursive descent may follow:
// arrays and inline tables nest through parseValue, and without a limit a
// hostile document would exhaust the stack.
maxDepth int
depth int
// useNumber leaves the numbers a Number carries the literal, instead of
// the evaluated int64 or float64 the tree holds by default.
useNumber bool
root map[string]any
current map[string]any
headers map[string]bool
frozen map[string]bool
dotted map[string]bool
arrays map[string]bool
currentPath []string
// keys interns key strings: a document that repeats a key across
// array-of-tables elements stores one string per distinct key instead of
// one per occurrence. The table is parser-local and dies with the parse;
// the tree keeps sharing the strings it was handed.
keys map[string]string
// keyBuf backs the transient single-segment result of parseKeyPath. A
// caller that keeps the path copies it out first, which is what
// retainPath does for the current section.
keyBuf [1]string
// absScratch backs the absolute path of a top-level key, which lives only
// for the statement being parsed.
absScratch [1]string
// wantDoc asks for the node tree the Document is built from; doc is that
// tree, and it stays nil when only the value tree is wanted. currentNode
// is the node of p.current; pending collects the comment lines since the
// last statement and trailing the comment on the statement's own line;
// lastEntry and lastTable name the statement those comments belong to;
// lastInline and lastArrayElems carry the nodes of the value parseValue has
// just produced.
wantDoc bool
doc *Table
currentNode *Table
pending []string
trailing string
lastEntry *Entry
lastTable *Table
lastInline *Table
lastArrayElems []*Table
// footer holds the comment lines that follow the last statement, which
// belong to the document rather than to any table or key.
footer []string
}
// maxNestingDepth bounds how deeply arrays and inline tables may nest when no
// limit is set. It matches the default encoding/json uses for the same reason,
// and sits far above any document a person writes.
const maxNestingDepth = 10000
// enterNesting counts one level of array or inline-table nesting and reports a
// document that nests deeper than the limit allows.
func (p *parser) enterNesting() error {
p.depth++
if p.depth > p.maxDepth {
return p.errf("nesting exceeds the limit of %d", p.maxDepth)
}
return nil
}
func (p *parser) leaveNesting() { p.depth-- }
func (p *parser) parse() (map[string]any, error) {
p.root = map[string]any{}
p.current = p.root
// The definition maps start unallocated: a document with no headers, no
// dotted keys and no inline tables never pays for them, and a nil map
// reads as empty. Each is created on its first write.
p.currentPath = nil
if p.wantDoc {
p.doc = newTable(p.root)
p.currentNode = p.doc
}
for i := 0; ; i++ {
if i%ctxCheckInterval == 0 {
if err := p.checkCtx(); err != nil {
return nil, err
}
}
if err := p.skipBlank(); err != nil {
return nil, err
}
if p.eof() {
break
}
p.lastEntry, p.lastTable = nil, nil
c := p.peek()
switch {
case c == '[':
if err := p.parseTableHeader(); err != nil {
return nil, err
}
default:
if err := p.parseKeyValue(); err != nil {
return nil, err
}
}
if err := p.expectLineEnd(); err != nil {
return nil, err
}
p.attachComments()
}
p.attachFooter()
return p.root, nil
}
// attachComments hands the collected comments to the statement just parsed:
// the lines above it to its entry or table, the comment on its own line as the
// trailing one.
func (p *parser) attachComments() {
if p.doc != nil {
switch {
case p.lastEntry != nil:
p.lastEntry.comments = p.pending
p.lastEntry.trailing = p.trailing
case p.lastTable != nil:
p.lastTable.comments = p.pending
p.lastTable.trailing = p.trailing
}
}
p.pending, p.trailing = nil, ""
}
// attachFooter hands the comment lines that follow the last statement to the
// document, which is where a comment block at the end of a file belongs.
func (p *parser) attachFooter() {
if p.doc != nil && len(p.pending) > 0 {
p.footer = p.pending
}
p.pending = nil
}
// checkCtx returns ctx.Err() when the context has been cancelled, nil
// otherwise. The call is a no-op when ctx is nil or the zero Background
// context, both of which never cancel.
func (p *parser) checkCtx() error {
if p.ctx == nil {
return nil
}
return p.ctx.Err()
}
// --- definition maps --------------------------------------------------------
// The definition maps record what a document has already defined, so a later
// statement cannot redefine it. Each is created on first write: reads on a
// nil map answer false, which is exactly the state of a map never written.
func (p *parser) markHeader(pk string) {
if p.headers == nil {
p.headers = make(map[string]bool, 4)
}
p.headers[pk] = true
}
func (p *parser) markFrozen(pk string) {
if p.frozen == nil {
p.frozen = make(map[string]bool, 4)
}
p.frozen[pk] = true
}
func (p *parser) markDotted(pk string) {
if p.dotted == nil {
p.dotted = make(map[string]bool, 4)
}
p.dotted[pk] = true
}
func (p *parser) markArray(pk string) {
if p.arrays == nil {
p.arrays = make(map[string]bool, 2)
}
p.arrays[pk] = true
}
// internKey returns the shared string for key bytes. The lookup works on the
// bytes directly, which the compiler lets run without allocating, so a
// repeated key costs no allocation at all and the tree stores one string per
// distinct key.
func (p *parser) internKey(b []byte) string {
if p.keys == nil {
p.keys = make(map[string]string, 16)
}
if s, ok := p.keys[string(b)]; ok {
return s
}
s := string(b)
p.keys[s] = s
return s
}
// --- table headers ---------------------------------------------------------
func (p *parser) parseTableHeader() error {
array := false
p.pos++ // consume '['
if !p.eof() && p.peek() == '[' {
array = true
p.pos++
}
first, rest, err := p.parseKeyPath()
if err != nil {
return err
}
p.skipInline()
if p.eof() || p.peek() != ']' {
return p.errf("expected ']' to close table header")
}
p.pos++
if array {
if p.eof() || p.peek() != ']' {
return p.errf("expected ']]' to close array-of-tables header")
}
p.pos++
}
// The key the rest of the header handling reads. parseKeyPath hands back
// a transient buffer for the single-segment case, the shape every
// repeated array-of-tables header has; anything longer is copied once.
key := p.keyBuf[:1]
key[0] = first
if len(rest) > 0 {
key = append([]string{first}, rest...)
}
if array {
tbl, elem, err := p.appendArrayTable(key)
if err != nil {
return err
}
// A new array-of-tables element starts a fresh scope: sub-table headers
// and inline-table freezes from the previous element no longer apply.
p.resetScopeUnder(key)
p.markArray(pathKey(key))
p.current = tbl
p.currentPath = p.retainPath(key)
p.currentNode = elem
p.lastTable = elem
return nil
}
pk := pathKey(key)
if p.headers[pk] || p.dotted[pk] || p.arrays[pk] {
return p.errf("table %q is defined more than once", strings.Join(key, "."))
}
p.markHeader(pk)
tbl, node, err := p.tableAt(key)
if err != nil {
return err
}
p.current = tbl
p.currentPath = p.retainPath(key)
p.currentNode = node
p.lastTable = node
return nil
}
// retainPath copies key into the parser-owned storage currentPath holds, so
// the transient key buffer is free to serve the next statement.
func (p *parser) retainPath(key []string) []string {
if cap(p.currentPath) < len(key) {
p.currentPath = make([]string, len(key))
} else {
p.currentPath = p.currentPath[:len(key)]
}
copy(p.currentPath, key)
return p.currentPath
}
// tableAt walks (creating intermediate tables) to the table named by key,
// relative to the document root, rejecting any step into a frozen inline table.
func (p *parser) tableAt(key []string) (map[string]any, *Table, error) {
cur := p.root
node := p.doc
// The intermediate-path bookkeeping allocates only when the key actually
// has intermediate segments; a single-segment key checks its own name.
var path []string
if len(key) > 1 {
path = make([]string, 0, len(key))
}
for _, k := range key {
if len(key) == 1 {
if p.frozen[k] {
return nil, nil, p.errf("cannot extend inline table %q", k)
}
} else {
path = append(path, k)
if p.frozen[pathKey(path)] {
return nil, nil, p.errf("cannot extend inline table %q", strings.Join(path, "."))
}
}
existing, ok := cur[k]
if !ok {
next := map[string]any{}
cur[k] = next
cur = next
if node != nil {
node = node.addTable(k, next)
}
continue
}
switch v := existing.(type) {
case map[string]any:
cur = v
if node != nil {
node = node.addTable(k, v)
}
case []map[string]any:
if len(v) == 0 {
return nil, nil, p.errf("key %q is an empty array of tables", k)
}
cur = v[len(v)-1]
if node != nil {
node = node.lastElement(k)
}
default:
return nil, nil, p.errf("key %q is not a table", k)
}
}
return cur, node, nil
}
func (p *parser) appendArrayTable(key []string) (map[string]any, *Table, error) {
parent := p.root
node := p.doc
// As in tableAt, the path slice exists only for a multi-segment key; the
// loop below runs for those alone.
var path []string
if len(key) > 1 {
path = make([]string, 0, len(key))
}
for _, k := range key[:len(key)-1] {
path = append(path, k)
if p.frozen[pathKey(path)] {
return nil, nil, p.errf("cannot extend inline table %q", strings.Join(path, "."))
}
existing, ok := parent[k]
if !ok {
next := map[string]any{}
parent[k] = next
parent = next
if node != nil {
node = node.addTable(k, next)
}
continue
}
switch v := existing.(type) {
case map[string]any:
parent = v
if node != nil {
node = node.addTable(k, v)
}
case []map[string]any:
parent = v[len(v)-1]
if node != nil {
node = node.lastElement(k)
}
default:
return nil, nil, p.errf("key %q is not a table", k)
}
}
leaf := key[len(key)-1]
tbl := map[string]any{}
switch existing := parent[leaf].(type) {
case nil:
parent[leaf] = []map[string]any{tbl}
case []map[string]any:
parent[leaf] = append(existing, tbl)
default:
return nil, nil, p.errf("key %q is not an array of tables", leaf)
}
var elem *Table
if node != nil {
elem = node.addElement(leaf, tbl)
}
return tbl, elem, nil
}
// --- key/value -------------------------------------------------------------
func (p *parser) parseKeyValue() error {
first, rest, err := p.parseKeyPath()
if err != nil {
return err
}
p.skipInline()
if p.eof() || p.peek() != '=' {
return p.errf("expected '=' after key")
}
p.pos++
p.skipInline()
val, err := p.parseValue()
if err != nil {
return err
}
dest := p.current
// The absolute path of the key drives the dotted-key bookkeeping and the
// inline-table freeze. A single top-level key needs it only for the
// freeze, where a one-element path sits in the parser's scratch.
var abs []string
if len(rest) > 0 || len(p.currentPath) > 0 {
abs = make([]string, 0, len(p.currentPath)+len(rest)+1)
abs = append(abs, p.currentPath...)
abs = append(abs, first)
} else {
abs = append(p.absScratch[:0], first)
}
// dests collects the map each dotted key descended into, which the node
// tree needs to build the matching tables around the value.
var dests []map[string]any
leaf := first
if len(rest) > 0 {
if err := p.descendKey(&dest, first, abs, &dests); err != nil {
return err
}
for _, k := range rest[:len(rest)-1] {
abs = append(abs, k)
if err := p.descendKey(&dest, k, abs, &dests); err != nil {
return err
}
}
leaf = rest[len(rest)-1]
abs = append(abs, leaf)
}
if _, exists := dest[leaf]; exists {
return p.errf("duplicate key %q", leaf)
}
dest[leaf] = val
if p.doc != nil {
node := p.currentNode
if len(rest) > 0 {
node = node.addTable(first, dests[0])
for i, k := range rest[:len(rest)-1] {
node = node.addTable(k, dests[i+1])
}
}
_, inline := val.(map[string]any)
entry := node.addValue(leaf, val, inline)
if inline {
entry.child = p.takeInline(val)
}
if nodes := p.takeArrayElems(val); nodes != nil {
entry.elements = nodes
}
p.lastEntry = entry
}
p.freezeInline(abs, val)
return nil
}
// descendKey walks dest into the sub-table named key on the dotted path abs,
// recording the path in the definition maps; dests collects the maps
// descended into.
func (p *parser) descendKey(dest *map[string]any, key string, abs []string, dests *[]map[string]any) error {
ak := pathKey(abs)
if p.frozen[ak] {
return p.errf("cannot extend inline table %q", strings.Join(abs, "."))
}
if p.headers[ak] {
return p.errf("cannot extend table %q with a dotted key", strings.Join(abs, "."))
}
p.markDotted(ak)
existing, ok := (*dest)[key]
if !ok {
next := map[string]any{}
(*dest)[key] = next
*dest = next
*dests = append(*dests, next)
return nil
}
m, ok := existing.(map[string]any)
if !ok {
return p.errf("key %q is not a table", key)
}
*dest = m
*dests = append(*dests, m)
return nil
}
// takeInline returns the node of the inline table just parsed, when v is that
// table's value, and clears it so a later value cannot pick it up.
func (p *parser) takeInline(v any) *Table {
node := p.lastInline
p.lastInline = nil
if _, ok := v.(map[string]any); !ok {
return nil
}
return node
}
// takeArrayElems returns the element nodes of the array just parsed, when v is
// that array's value, and clears them.
func (p *parser) takeArrayElems(v any) []*Table {
nodes := p.lastArrayElems
p.lastArrayElems = nil
if _, ok := v.([]any); !ok {
return nil
}
return nodes
}
// freezeInline marks the path of an inline table (and any nested inline tables)
// as immutable, so a later header or dotted key cannot extend it. The
// recursion appends into the caller's path slice; the frozen map keeps the
// joined strings, never the slice, so the backing is free to be reused.
func (p *parser) freezeInline(path []string, val any) {
m, ok := val.(map[string]any)
if !ok {
return
}
p.markFrozen(pathKey(path))
for k, v := range m {
p.freezeInline(append(path, k), v)
}
}
// resetScopeUnder forgets the definition records nested under key, which
// belong to the previous element of an array of tables: headers, frozen
// inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element.
func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00"
p.resetMapUnder(p.headers, prefix)
p.resetMapUnder(p.frozen, prefix)
p.resetMapUnder(p.dotted, prefix)
p.resetMapUnder(p.arrays, prefix)
}
// resetMapUnder deletes the entries m holds under prefix. An empty or
// unallocated map holds none, so the common case walks nothing.
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
if len(m) == 0 {
return
}
for k := range m {
if strings.HasPrefix(k, prefix) {
delete(m, k)
}
}
}
// parseKeyPath parses a dotted key. The first component comes back directly
// and the rest as a usually nil slice, because a single-component key is the
// common shape and a fresh slice per statement is what the allocation profile
// showed. The single-key slice a caller sees is parser-owned and transient.
func (p *parser) parseKeyPath() (string, []string, error) {
p.skipInline()
first, err := p.parseKeyComponent()
if err != nil {
return "", nil, err
}
p.skipInline()
if p.eof() || p.peek() != '.' {
return first, nil, nil
}
p.pos++
var rest []string
for {
p.skipInline()
part, err := p.parseKeyComponent()
if err != nil {
return "", nil, err
}
rest = append(rest, part)
p.skipInline()
if p.eof() || p.peek() != '.' {
return first, rest, nil
}
p.pos++
}
}
func (p *parser) parseKeyComponent() (string, error) {
if p.eof() {
return "", p.errf("expected a key")
}
switch p.peek() {
case '"':
if p.lookahead(`"""`) {
return "", p.errf("multiline strings are not allowed in keys")
}
return p.parseBasicString()
case '\'':
if p.lookahead(`'''`) {
return "", p.errf("multiline strings are not allowed in keys")
}
return p.parseLiteralString()
default:
start := p.pos
for !p.eof() {
c := p.peek()
if (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
(c >= '0' && c <= '9') || c == '_' || c == '-' {
p.pos++
continue
}
break
}
// The stopping byte decides the message: a multi-byte sequence that
// does not decode names that, before any grammar message can.
if !p.eof() && p.peek() >= utf8.RuneSelf {
if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos)
}
}
if p.pos == start {
r, _ := utf8.DecodeRune(p.src[p.pos:])
return "", p.errf("invalid key character %q", string(r))
}
return p.internKey(p.src[start:p.pos]), nil
}
}
// --- values ----------------------------------------------------------------
func (p *parser) parseValue() (any, error) {
if p.eof() {
return nil, p.errf("expected a value")
}
// A container value leaves its node behind for the caller to pick up; a
// value that follows must not find the previous one.
p.lastInline, p.lastArrayElems = nil, nil
switch c := p.peek(); {
case c == '"':
return p.parseBasicString()
case c == '\'':
return p.parseLiteralString()
case c == '[':
return p.parseArray()
case c == '{':
return p.parseInlineTable()
case c == 't' || c == 'f':
return p.parseBool()
default:
return p.parseAtom()
}
}
func (p *parser) parseBool() (any, error) {
if p.match("true") {
return true, nil
}
if p.match("false") {
return false, nil
}
return nil, p.errf("invalid value")
}
// parseAtom handles numbers, inf/nan, and date-times.
func (p *parser) parseAtom() (any, error) {
start := p.pos
p.scanBareToken()
tok := string(p.src[start:p.pos])
if tok == "" {
return nil, p.errf("expected a value")
}
if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
}
// A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
if next, ok := p.peekAt(1); ok && next >= '0' && next <= '9' {
p.pos++ // consume the separating space
timeStart := p.pos
p.scanBareToken()
tok = tok + " " + string(p.src[timeStart:p.pos])
}
}
if v, ok := parseDateTime(tok); ok {
return v, nil
}
v, err := decodeNumber(tok)
if err != nil {
return nil, p.errf("%s", err)
}
// The token's shape is validated either way; UseNumber only keeps the
// literal instead of the evaluated value.
if p.useNumber {
return Number(tok), nil
}
return v, nil
}
// scanBareToken advances past a bare value token (number, bool, or date-time),
// stopping at whitespace, a separator, or a comment.
func (p *parser) scanBareToken() {
for !p.eof() {
c := p.peek()
if c == ' ' || c == '\t' || c == '\n' || c == '\r' ||
c == ',' || c == ']' || c == '}' || c == '#' {
return
}
p.pos++
}
}
// hasHighByte reports whether s holds any byte outside ASCII, the cheap gate
// in front of a full UTF-8 check.
func hasHighByte(s string) bool {
for i := range len(s) {
if s[i] >= utf8.RuneSelf {
return true
}
}
return false
}
// --- strings ---------------------------------------------------------------
func (p *parser) parseBasicString() (string, error) {
if p.lookahead(`"""`) {
return p.parseMultilineString('"', true)
}
p.pos++ // opening quote
start := p.pos
// A run of plain characters up to the closing quote needs no builder, only
// one copy at the end; escapes, controls and multi-byte runes fall through
// to the builder loop, which validates them on the spot.
for p.pos < len(p.src) {
c := p.src[p.pos]
if c == '"' {
s := string(p.src[start:p.pos])
p.pos++
return s, nil
}
if c == '\\' || c == '\n' || c == '\r' || c >= utf8.RuneSelf ||
(c < 0x20 && c != '\t') || c == 0x7f {
break
}
p.pos++
}
if p.eof() {
return "", p.errf("unterminated string")
}
var b strings.Builder
b.Grow(p.pos - start)
b.Write(p.src[start:p.pos])
return p.parseBasicStringRest(&b)
}
// parseBasicStringRest continues a basic string whose fast scan has met a byte
// it does not handle: an escape, a control character, a multi-byte rune, or a
// bare newline, which the loop rejects.
func (p *parser) parseBasicStringRest(b *strings.Builder) (string, error) {
for {
if p.eof() {
return "", p.errf("unterminated string")
}
c := p.peek()
switch c {
case '"':
p.pos++
return b.String(), nil
case '\n':
return "", p.errf("unterminated string")
case '\r':
return "", p.errf("bare carriage return is not allowed in a string")
case '\\':
p.pos++
r, err := p.readEscape()
if err != nil {
return "", err
}
b.WriteRune(r)
default:
if err := p.writeContentRune(b); err != nil {
return "", err
}
}
}
}
func (p *parser) parseLiteralString() (string, error) {
if p.lookahead(`'''`) {
return p.parseMultilineString('\'', false)
}
p.pos++ // opening quote
start := p.pos
// The same fast scan as the basic string, without the escape case.
for p.pos < len(p.src) {
c := p.src[p.pos]
if c == '\'' {
s := string(p.src[start:p.pos])
p.pos++
return s, nil
}
if c == '\n' || c == '\r' || c >= utf8.RuneSelf ||
(c < 0x20 && c != '\t') || c == 0x7f {
break
}
p.pos++
}
if p.eof() {
return "", p.errf("unterminated literal string")
}
var b strings.Builder
b.Grow(p.pos - start)
b.Write(p.src[start:p.pos])
for {
if p.eof() {
return "", p.errf("unterminated literal string")
}
c := p.peek()
switch c {
case '\'':
p.pos++
return b.String(), nil
case '\n':
return "", p.errf("unterminated literal string")
case '\r':
return "", p.errf("bare carriage return is not allowed in a string")
default:
if err := p.writeContentRune(&b); err != nil {
return "", err
}
}
}
}
// writeContentRune appends the rune at the cursor to b and advances past it.
// An ASCII byte, which includes every control character the grammar forbids,
// is checked and written directly; a multi-byte rune is decoded, and a
// sequence that does not decode is the UTF-8 error reported where it sits.
func (p *parser) writeContentRune(b *strings.Builder) error {
c := p.peek()
if c < utf8.RuneSelf {
if isControlRune(rune(c)) {
return p.errf("control character U+%04X is not allowed in a string", c)
}
p.pos++
b.WriteByte(c)
return nil
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 in string at byte offset %d", p.pos)
}
p.pos += size
b.WriteRune(r)
return nil
}
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed.
if !p.eof() && p.peek() == '\r' {
p.pos++
}
if !p.eof() && p.peek() == '\n' {
p.line++
p.pos++
}
var b strings.Builder
for {
if p.eof() {
return "", p.errf("unterminated multiline string")
}
if p.peek() == quote {
// Count the run of delimiter characters. The last three close the
// string; up to two extra ones belong to the content.
n := 0
for p.pos+n < len(p.src) && p.src[p.pos+n] == quote {
n++
}
if n >= 3 {
if n > 5 {
return "", p.errf("too many '%c' before the closing delimiter", quote)
}
for range n - 3 {
b.WriteByte(quote)
}
p.skipN(n)
return b.String(), nil
}
for range n {
b.WriteByte(quote)
p.pos++
}
continue
}
c := p.peek()
switch {
case c == '\n':
p.line++
p.pos++
b.WriteByte(c)
case c == '\r':
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
b.WriteByte(c)
p.pos++
continue
}
return "", p.errf("bare carriage return is not allowed in a string")
case escapes && c == '\\':
p.pos++
// Line-ending backslash trims the following whitespace/newlines.
if p.trimLineEndingBackslash() {
continue
}
r, err := p.readEscape()
if err != nil {
return "", err
}
b.WriteRune(r)
default:
if err := p.writeContentRune(&b); err != nil {
return "", err
}
}
}
}
// trimLineEndingBackslash consumes whitespace through the next newline (and the
// blank lines that follow) when a backslash is the last token on a line.
// It reports whether it did so.
func (p *parser) trimLineEndingBackslash() bool {
save, saveLine := p.pos, p.line
for !p.eof() {
c := p.peek()
if c == ' ' || c == '\t' || c == '\r' {
p.pos++
continue
}
if c == '\n' {
break
}
// Not a line-ending backslash; restore.
p.pos, p.line = save, saveLine
return false
}
if p.eof() {
p.pos, p.line = save, saveLine
return false
}
// Consume the newline and all following whitespace.
for !p.eof() {
c := p.peek()
if c == '\n' {
p.line++
p.pos++
continue
}
if c == ' ' || c == '\t' || c == '\r' {
p.pos++
continue
}
break
}
return true
}
func (p *parser) readEscape() (rune, error) {
if p.eof() {
return 0, p.errf("unterminated escape sequence")
}
c := p.next()
switch c {
case 'b':
return '\b', nil
case 't':
return '\t', nil
case 'n':
return '\n', nil
case 'f':
return '\f', nil
case 'r':
return '\r', nil
case 'e':
// TOML 1.1: the escape character.
return '\x1b', nil
case '"':
return '"', nil
case '\\':
return '\\', nil
case 'x':
// TOML 1.1: two hex digits, code points 0x00 through 0xFF.
return p.readUnicode(2)
case 'u':
return p.readUnicode(4)
case 'U':
return p.readUnicode(8)
default:
// The byte just consumed starts a rune: the backslash before it is a
// boundary, and the input is valid UTF-8.
r, _ := utf8.DecodeRune(p.src[p.pos-1:])
return 0, p.errf("invalid escape sequence \\%c", r)
}
}
func (p *parser) readUnicode(n int) (rune, error) {
if p.pos+n > len(p.src) {
return 0, p.errf("invalid unicode escape")
}
hex := string(p.src[p.pos : p.pos+n])
p.pos += n
v, err := strconv.ParseInt(hex, 16, 64)
if err != nil {
return 0, p.errf("invalid unicode escape \\%s", hex)
}
if v > 0x10FFFF || (v >= 0xD800 && v <= 0xDFFF) {
return 0, p.errf("escape \\%s is not a valid Unicode scalar value", hex)
}
return rune(v), nil
}
// --- arrays and inline tables ---------------------------------------------
func (p *parser) parseArray() (val any, err error) {
if err := p.enterNesting(); err != nil {
return nil, err
}
defer p.leaveNesting()
p.pos++ // '['
// A small presize covers the arrays documents actually hold, and trades a
// little capacity on tiny arrays for the growth chain an append-from-nil
// costs per array.
arr := make([]any, 0, 4)
// elems carries the node of each element that is an inline table, so the
// caller can keep its key order; the entries are nil for other values.
var elems []*Table
if p.doc != nil {
defer func() {
if err == nil {
p.lastArrayElems = elems
}
}()
}
for {
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if p.eof() {
return nil, p.errf("unterminated array")
}
if p.peek() == ']' {
p.pos++
return arr, nil
}
v, err := p.parseValue()
if err != nil {
return nil, err
}
if p.doc != nil {
elems = append(elems, p.takeInline(v))
}
arr = append(arr, v)
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if p.eof() {
return nil, p.errf("unterminated array")
}
switch p.peek() {
case ',':
p.pos++
case ']':
p.pos++
return arr, nil
default:
return nil, p.errf("expected ',' or ']' in array")
}
}
}
func (p *parser) parseInlineTable() (val any, err error) {
if err := p.enterNesting(); err != nil {
return nil, err
}
defer p.leaveNesting()
p.pos++ // '{'
tbl := map[string]any{}
// assigned tracks the dotted paths written into this table. It is created
// on the first key, so an empty inline table allocates nothing for it.
var assigned map[string]bool
// The inline table is a node of its own, so the keys keep their order; the
// caller picks the node up when the table parses.
var node *Table
if p.doc != nil {
node = newTable(tbl)
node.inline = true
defer func() {
if err == nil {
p.lastInline = node
}
}()
}
// TOML 1.1 lets an inline table span lines: interior whitespace includes
// newlines and comments, and a trailing comma is allowed before the
// closing brace.
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if !p.eof() && p.peek() == '}' {
p.pos++
return tbl, nil
}
for {
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
first, rest, err := p.parseKeyPath()
if err != nil {
return nil, err
}
p.skipInline()
if p.eof() || p.peek() != '=' {
return nil, p.errf("expected '=' in inline table")
}
p.pos++
p.skipInline()
val, err := p.parseValue()
if err != nil {
return nil, err
}
dest := tbl
var path []string
var dests []map[string]any
leaf := first
if len(rest) > 0 {
path = append(p.absScratch[:0], first)
d, err := p.descendInline(&dest, first, path, assigned)
if err != nil {
return nil, err
}
dests = append(dests, d)
for _, k := range rest[:len(rest)-1] {
path = append(path, k)
d, err := p.descendInline(&dest, k, path, assigned)
if err != nil {
return nil, err
}
dests = append(dests, d)
}
leaf = rest[len(rest)-1]
path = append(path, leaf)
}
if _, exists := dest[leaf]; exists {
return nil, p.errf("duplicate key %q in inline table", leaf)
}
dest[leaf] = val
if assigned == nil {
assigned = make(map[string]bool, 4)
}
if len(rest) == 0 {
assigned[first] = true
} else {
assigned[pathKey(path)] = true
}
if node != nil {
child := node
if len(rest) > 0 {
child = child.addTable(first, dests[0])
for i, k := range rest[:len(rest)-1] {
child = child.addTable(k, dests[i+1])
}
}
_, inline := val.(map[string]any)
entry := child.addValue(leaf, val, inline)
if inline {
entry.child = p.takeInline(val)
}
if nodes := p.takeArrayElems(val); nodes != nil {
entry.elements = nodes
}
}
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if p.eof() {
return nil, p.errf("unterminated inline table")
}
switch p.peek() {
case ',':
p.pos++
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if !p.eof() && p.peek() == '}' {
p.pos++
return tbl, nil
}
case '}':
p.pos++
return tbl, nil
default:
return nil, p.errf("expected ',' or '}' in inline table")
}
}
}
// descendInline walks dest into the sub-table named key inside an inline
// table, rejecting a dotted segment the table has already defined.
func (p *parser) descendInline(dest *map[string]any, key string, path []string, assigned map[string]bool) (map[string]any, error) {
pk := pathKey(path)
if assigned[pk] {
return nil, p.errf("key %q is already defined", strings.Join(path, "."))
}
existing, ok := (*dest)[key]
if !ok {
m := map[string]any{}
(*dest)[key] = m
*dest = m
return m, nil
}
m, isMap := existing.(map[string]any)
if !isMap {
return nil, p.errf("key %q is already defined", key)
}
*dest = m
return m, nil
}
// --- scanning helpers ------------------------------------------------------
func (p *parser) eof() bool { return p.pos >= len(p.src) }
func (p *parser) peek() byte { return p.src[p.pos] }
// peekAt returns the byte at offset n from the current position and whether the
// offset is within the source. Use it instead of indexing p.src directly when
// the offset may sit past the end.
func (p *parser) peekAt(n int) (byte, bool) {
i := p.pos + n
if i < 0 || i >= len(p.src) {
return 0, false
}
return p.src[i], true
}
func (p *parser) next() byte {
c := p.src[p.pos]
p.pos++
return c
}
func (p *parser) skipN(n int) {
for i := 0; i < n && !p.eof(); i++ {
p.next()
}
}
func (p *parser) match(word string) bool {
if p.lookahead(word) {
p.skipN(len(word))
return true
}
return false
}
// lookahead reports whether s follows the cursor. Every lookahead argument in
// the grammar is ASCII, so comparing bytes is exact.
func (p *parser) lookahead(s string) bool {
return p.pos+len(s) <= len(p.src) && string(p.src[p.pos:p.pos+len(s)]) == s
}
// skipInline consumes spaces and tabs only.
func (p *parser) skipInline() {
for !p.eof() {
if c := p.peek(); c == ' ' || c == '\t' {
p.pos++
continue
}
break
}
}
// skipNestedSpace consumes whitespace, newlines, and comments inside a value
// container (an array, or an inline table under TOML 1.1).
func (p *parser) skipNestedSpace() error {
for !p.eof() {
switch p.peek() {
case ' ', '\t':
p.pos++
case '\r':
if err := p.expectCRLF(); err != nil {
return err
}
case '\n':
p.line++
p.pos++
case '#':
// A comment between values inside an array or an inline table is
// skipped; the Document does not carry those yet.
if _, err := p.skipComment(); err != nil {
return err
}
default:
return nil
}
}
return nil
}
// skipBlank consumes whitespace, blank lines, and comments between statements.
func (p *parser) skipBlank() error {
for !p.eof() {
switch p.peek() {
case ' ', '\t':
p.pos++
case '\r':
if err := p.expectCRLF(); err != nil {
return err
}
case '\n':
p.line++
p.pos++
case '#':
line, err := p.skipComment()
if err != nil {
return err
}
p.pending = append(p.pending, line)
default:
return nil
}
}
return nil
}
func (p *parser) skipComment() (string, error) {
p.pos++ // consume '#'
start := p.pos
for !p.eof() {
c := p.peek()
switch {
case c == '\n':
return commentText(string(p.src[start:p.pos])), nil
case c == '\r':
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
return commentText(string(p.src[start:p.pos])), nil
}
return "", p.errf("bare carriage return is not allowed")
case c == '\t':
p.pos++
case c < 0x20 || c == 0x7f:
return "", p.errf("control character U+%04X is not allowed in a comment", c)
case c < utf8.RuneSelf:
p.pos++
default:
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos)
}
p.pos += size
}
}
return commentText(string(p.src[start:p.pos])), nil
}
// commentText drops the one space that usually follows the '#', so a line
// stored in a Document reads as the comment itself.
func commentText(s string) string {
return strings.TrimPrefix(s, " ")
}
// expectCRLF consumes a carriage return that must be immediately followed by a
// line feed; a bare CR is invalid.
func (p *parser) expectCRLF() error {
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
p.pos++ // consume CR; the LF is handled by the caller
return nil
}
return p.errf("bare carriage return is not allowed")
}
// expectLineEnd consumes trailing inline whitespace and an optional comment,
// then requires a newline or end of input.
func (p *parser) expectLineEnd() error {
p.skipInline()
if p.eof() {
return nil
}
if p.peek() == '#' {
line, err := p.skipComment()
if err != nil {
return err
}
if p.doc != nil {
p.trailing = line
}
}
if p.eof() {
return nil
}
if p.peek() == '\r' {
if err := p.expectCRLF(); err != nil {
return err
}
}
if p.eof() {
return nil
}
if p.peek() == '\n' {
p.line++
p.pos++
return nil
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 after value at byte offset %d", p.pos)
}
return p.errf("unexpected %q after value", string(r))
}
// errf builds the SyntaxError with the position the scan stopped at: the line,
// the byte offset in the input, and the 1-based column on that line. The
// offset is the cursor, which on an escape or a delimiter run sits just after
// the bytes that caused the complaint; SourceLine renders the caret there.
func (p *parser) errf(format string, args ...any) error {
col := p.pos + 1
if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 {
col = p.pos - start
}
return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)}
}
// pathKey joins key components with a NUL separator so a dotted path can be
// used as a map key for tracking defined tables. A single component comes
// back as it is, with no join and no copy.
func pathKey(parts []string) string {
return strings.Join(parts, "\x00")
}
// isControlRune reports whether r is a control character disallowed in a string
// literal. Tab, line feed, and carriage return are permitted (handled
// elsewhere); everything else below U+0020, plus U+007F, is rejected.
func isControlRune(r rune) bool {
if r == '\t' || r == '\n' || r == '\r' {
return false
}
return r < 0x20 || r == 0x7f
}