2026-08-19 09:47:00 +02:00
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
|
|
|
|
|
|
package interpres
|
|
|
|
|
|
|
|
|
|
import (
|
2026-09-21 23:55:58 +02:00
|
|
|
"bytes"
|
2026-08-19 09:47:00 +02:00
|
|
|
"context"
|
|
|
|
|
"fmt"
|
|
|
|
|
"strconv"
|
|
|
|
|
"strings"
|
2026-09-17 23:11:50 +02:00
|
|
|
"unicode/utf8"
|
2026-08-19 09:47:00 +02:00
|
|
|
)
|
|
|
|
|
|
|
|
|
|
// ctxCheckInterval is the number of top-level parser iterations between
|
|
|
|
|
// context-cancellation checks. A small interval keeps the response snappy on
|
|
|
|
|
// cancellation; a too-small one wastes cycles on a non-cancelled run.
|
|
|
|
|
const ctxCheckInterval = 64
|
|
|
|
|
|
|
|
|
|
// parser is a recursive-descent TOML parser producing a map[string]any tree.
|
2026-09-17 23:11:50 +02:00
|
|
|
//
|
2026-09-20 22:15:10 +02:00
|
|
|
// The scanner works on bytes, not runes: every character that drives the
|
|
|
|
|
// grammar (quotes, separators, newlines, bare-key characters) is ASCII, the
|
|
|
|
|
// scan validates a multi-byte sequence where it meets one, and multi-byte
|
|
|
|
|
// runes matter only as content, where they are decoded on the spot. Holding
|
2026-09-17 23:11:50 +02:00
|
|
|
// the source as []rune instead would cost a conversion pass plus four bytes
|
|
|
|
|
// per rune of extra memory before parsing even starts.
|
2026-08-19 09:47:00 +02:00
|
|
|
type parser struct {
|
2026-09-17 23:11:50 +02:00
|
|
|
src []byte
|
2026-08-19 09:47:00 +02:00
|
|
|
pos int
|
|
|
|
|
line int
|
|
|
|
|
ctx context.Context
|
|
|
|
|
|
2026-09-19 19:36:24 +02:00
|
|
|
// maxDepth and depth bound the nesting the recursive descent may follow:
|
|
|
|
|
// arrays and inline tables nest through parseValue, and without a limit a
|
|
|
|
|
// hostile document would exhaust the stack.
|
|
|
|
|
maxDepth int
|
|
|
|
|
depth int
|
|
|
|
|
|
2026-09-21 23:49:39 +02:00
|
|
|
// useNumber leaves the numbers a Number carries the literal, instead of
|
|
|
|
|
// the evaluated int64 or float64 the tree holds by default.
|
|
|
|
|
useNumber bool
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
root map[string]any
|
|
|
|
|
current map[string]any
|
|
|
|
|
headers map[string]bool
|
|
|
|
|
frozen map[string]bool
|
|
|
|
|
dotted map[string]bool
|
|
|
|
|
arrays map[string]bool
|
|
|
|
|
|
2026-09-22 21:15:00 +02:00
|
|
|
// scopeMarks records the definition-map entries added under an array of
|
|
|
|
|
// tables, keyed by that array's path, so a new element's reset drops
|
|
|
|
|
// exactly what the previous element added. Without it the reset scans
|
|
|
|
|
// every map for the prefix, which a document with many elements and many
|
|
|
|
|
// definitions outside them turns quadratic.
|
|
|
|
|
scopeMarks map[string][]string
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
currentPath []string
|
2026-09-20 10:40:57 +02:00
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// keys interns key strings: a document that repeats a key across
|
|
|
|
|
// array-of-tables elements stores one string per distinct key instead of
|
|
|
|
|
// one per occurrence. The table is parser-local and dies with the parse;
|
|
|
|
|
// the tree keeps sharing the strings it was handed.
|
|
|
|
|
keys map[string]string
|
|
|
|
|
|
|
|
|
|
// keyBuf backs the transient single-segment result of parseKeyPath. A
|
|
|
|
|
// caller that keeps the path copies it out first, which is what
|
|
|
|
|
// retainPath does for the current section.
|
|
|
|
|
keyBuf [1]string
|
|
|
|
|
|
|
|
|
|
// absScratch backs the absolute path of a top-level key, which lives only
|
|
|
|
|
// for the statement being parsed.
|
|
|
|
|
absScratch [1]string
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
// wantDoc asks for the node tree the Document is built from; doc is that
|
|
|
|
|
// tree, and it stays nil when only the value tree is wanted. currentNode
|
|
|
|
|
// is the node of p.current; pending collects the comment lines since the
|
|
|
|
|
// last statement and trailing the comment on the statement's own line;
|
|
|
|
|
// lastEntry and lastTable name the statement those comments belong to;
|
|
|
|
|
// lastInline and lastArrayElems carry the nodes of the value parseValue has
|
|
|
|
|
// just produced.
|
|
|
|
|
wantDoc bool
|
|
|
|
|
doc *Table
|
|
|
|
|
currentNode *Table
|
|
|
|
|
pending []string
|
|
|
|
|
trailing string
|
|
|
|
|
lastEntry *Entry
|
|
|
|
|
lastTable *Table
|
|
|
|
|
lastInline *Table
|
|
|
|
|
lastArrayElems []*Table
|
|
|
|
|
|
|
|
|
|
// footer holds the comment lines that follow the last statement, which
|
|
|
|
|
// belong to the document rather than to any table or key.
|
|
|
|
|
footer []string
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
2026-09-19 19:36:24 +02:00
|
|
|
// maxNestingDepth bounds how deeply arrays and inline tables may nest when no
|
|
|
|
|
// limit is set. It matches the default encoding/json uses for the same reason,
|
|
|
|
|
// and sits far above any document a person writes.
|
|
|
|
|
const maxNestingDepth = 10000
|
|
|
|
|
|
|
|
|
|
// enterNesting counts one level of array or inline-table nesting and reports a
|
|
|
|
|
// document that nests deeper than the limit allows.
|
|
|
|
|
func (p *parser) enterNesting() error {
|
|
|
|
|
p.depth++
|
|
|
|
|
if p.depth > p.maxDepth {
|
|
|
|
|
return p.errf("nesting exceeds the limit of %d", p.maxDepth)
|
|
|
|
|
}
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) leaveNesting() { p.depth-- }
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
func (p *parser) parse() (map[string]any, error) {
|
|
|
|
|
p.root = map[string]any{}
|
|
|
|
|
p.current = p.root
|
2026-09-20 22:15:10 +02:00
|
|
|
// The definition maps start unallocated: a document with no headers, no
|
|
|
|
|
// dotted keys and no inline tables never pays for them, and a nil map
|
|
|
|
|
// reads as empty. Each is created on its first write.
|
2026-08-19 09:47:00 +02:00
|
|
|
p.currentPath = nil
|
2026-09-20 10:40:57 +02:00
|
|
|
if p.wantDoc {
|
|
|
|
|
p.doc = newTable(p.root)
|
|
|
|
|
p.currentNode = p.doc
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
|
|
|
|
|
for i := 0; ; i++ {
|
|
|
|
|
if i%ctxCheckInterval == 0 {
|
|
|
|
|
if err := p.checkCtx(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if err := p.skipBlank(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
break
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
p.lastEntry, p.lastTable = nil, nil
|
2026-08-19 09:47:00 +02:00
|
|
|
c := p.peek()
|
|
|
|
|
switch {
|
|
|
|
|
case c == '[':
|
|
|
|
|
if err := p.parseTableHeader(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
default:
|
|
|
|
|
if err := p.parseKeyValue(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if err := p.expectLineEnd(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
p.attachComments()
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
p.attachFooter()
|
2026-08-19 09:47:00 +02:00
|
|
|
return p.root, nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
// attachComments hands the collected comments to the statement just parsed:
|
|
|
|
|
// the lines above it to its entry or table, the comment on its own line as the
|
|
|
|
|
// trailing one.
|
|
|
|
|
func (p *parser) attachComments() {
|
|
|
|
|
if p.doc != nil {
|
|
|
|
|
switch {
|
|
|
|
|
case p.lastEntry != nil:
|
|
|
|
|
p.lastEntry.comments = p.pending
|
|
|
|
|
p.lastEntry.trailing = p.trailing
|
|
|
|
|
case p.lastTable != nil:
|
|
|
|
|
p.lastTable.comments = p.pending
|
|
|
|
|
p.lastTable.trailing = p.trailing
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
p.pending, p.trailing = nil, ""
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// attachFooter hands the comment lines that follow the last statement to the
|
|
|
|
|
// document, which is where a comment block at the end of a file belongs.
|
|
|
|
|
func (p *parser) attachFooter() {
|
|
|
|
|
if p.doc != nil && len(p.pending) > 0 {
|
|
|
|
|
p.footer = p.pending
|
|
|
|
|
}
|
|
|
|
|
p.pending = nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
// checkCtx returns ctx.Err() when the context has been cancelled, nil
|
|
|
|
|
// otherwise. The call is a no-op when ctx is nil or the zero Background
|
|
|
|
|
// context, both of which never cancel.
|
|
|
|
|
func (p *parser) checkCtx() error {
|
|
|
|
|
if p.ctx == nil {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
return p.ctx.Err()
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// --- definition maps --------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
// The definition maps record what a document has already defined, so a later
|
|
|
|
|
// statement cannot redefine it. Each is created on first write: reads on a
|
|
|
|
|
// nil map answer false, which is exactly the state of a map never written.
|
|
|
|
|
|
|
|
|
|
func (p *parser) markHeader(pk string) {
|
|
|
|
|
if p.headers == nil {
|
|
|
|
|
p.headers = make(map[string]bool, 4)
|
|
|
|
|
}
|
|
|
|
|
p.headers[pk] = true
|
2026-09-22 21:15:00 +02:00
|
|
|
p.trackScope(pk)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) markFrozen(pk string) {
|
|
|
|
|
if p.frozen == nil {
|
|
|
|
|
p.frozen = make(map[string]bool, 4)
|
|
|
|
|
}
|
|
|
|
|
p.frozen[pk] = true
|
2026-09-22 21:15:00 +02:00
|
|
|
p.trackScope(pk)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) markDotted(pk string) {
|
|
|
|
|
if p.dotted == nil {
|
|
|
|
|
p.dotted = make(map[string]bool, 4)
|
|
|
|
|
}
|
|
|
|
|
p.dotted[pk] = true
|
2026-09-22 21:15:00 +02:00
|
|
|
p.trackScope(pk)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) markArray(pk string) {
|
|
|
|
|
if p.arrays == nil {
|
|
|
|
|
p.arrays = make(map[string]bool, 2)
|
|
|
|
|
}
|
|
|
|
|
p.arrays[pk] = true
|
2026-09-22 21:15:00 +02:00
|
|
|
p.trackScope(pk)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// trackScope records a definition entry under every array of tables it falls
|
|
|
|
|
// inside, so resetScopeUnder can drop it when a later element opens. An entry
|
|
|
|
|
// under no array, such as every definition before the first header, needs no
|
|
|
|
|
// record: no reset can ever name it.
|
|
|
|
|
func (p *parser) trackScope(pk string) {
|
|
|
|
|
for arr := range p.arrays {
|
|
|
|
|
if strings.HasPrefix(pk, arr+"\x00") {
|
|
|
|
|
if p.scopeMarks == nil {
|
|
|
|
|
p.scopeMarks = make(map[string][]string, 2)
|
|
|
|
|
}
|
|
|
|
|
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// internKey returns the shared string for key bytes. The lookup works on the
|
|
|
|
|
// bytes directly, which the compiler lets run without allocating, so a
|
|
|
|
|
// repeated key costs no allocation at all and the tree stores one string per
|
|
|
|
|
// distinct key.
|
|
|
|
|
func (p *parser) internKey(b []byte) string {
|
|
|
|
|
if p.keys == nil {
|
|
|
|
|
p.keys = make(map[string]string, 16)
|
|
|
|
|
}
|
|
|
|
|
if s, ok := p.keys[string(b)]; ok {
|
|
|
|
|
return s
|
|
|
|
|
}
|
|
|
|
|
s := string(b)
|
|
|
|
|
p.keys[s] = s
|
|
|
|
|
return s
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
// --- table headers ---------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseTableHeader() error {
|
|
|
|
|
array := false
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // consume '['
|
2026-08-19 09:47:00 +02:00
|
|
|
if !p.eof() && p.peek() == '[' {
|
|
|
|
|
array = true
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
first, rest, err := p.parseKeyPath()
|
2026-08-19 09:47:00 +02:00
|
|
|
if err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
p.skipInline()
|
|
|
|
|
if p.eof() || p.peek() != ']' {
|
|
|
|
|
return p.errf("expected ']' to close table header")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
if array {
|
|
|
|
|
if p.eof() || p.peek() != ']' {
|
|
|
|
|
return p.errf("expected ']]' to close array-of-tables header")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// The key the rest of the header handling reads. parseKeyPath hands back
|
|
|
|
|
// a transient buffer for the single-segment case, the shape every
|
|
|
|
|
// repeated array-of-tables header has; anything longer is copied once.
|
|
|
|
|
key := p.keyBuf[:1]
|
|
|
|
|
key[0] = first
|
|
|
|
|
if len(rest) > 0 {
|
|
|
|
|
key = append([]string{first}, rest...)
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
if array {
|
2026-09-20 10:40:57 +02:00
|
|
|
tbl, elem, err := p.appendArrayTable(key)
|
2026-08-19 09:47:00 +02:00
|
|
|
if err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
// A new array-of-tables element starts a fresh scope: sub-table headers
|
|
|
|
|
// and inline-table freezes from the previous element no longer apply.
|
|
|
|
|
p.resetScopeUnder(key)
|
2026-09-20 22:15:10 +02:00
|
|
|
p.markArray(pathKey(key))
|
2026-08-19 09:47:00 +02:00
|
|
|
p.current = tbl
|
2026-09-20 22:15:10 +02:00
|
|
|
p.currentPath = p.retainPath(key)
|
2026-09-20 10:40:57 +02:00
|
|
|
p.currentNode = elem
|
|
|
|
|
p.lastTable = elem
|
2026-08-19 09:47:00 +02:00
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
pk := pathKey(key)
|
|
|
|
|
if p.headers[pk] || p.dotted[pk] || p.arrays[pk] {
|
|
|
|
|
return p.errf("table %q is defined more than once", strings.Join(key, "."))
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
p.markHeader(pk)
|
2026-08-19 09:47:00 +02:00
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
tbl, node, err := p.tableAt(key)
|
2026-08-19 09:47:00 +02:00
|
|
|
if err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
p.current = tbl
|
2026-09-20 22:15:10 +02:00
|
|
|
p.currentPath = p.retainPath(key)
|
2026-09-20 10:40:57 +02:00
|
|
|
p.currentNode = node
|
|
|
|
|
p.lastTable = node
|
2026-08-19 09:47:00 +02:00
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// retainPath copies key into the parser-owned storage currentPath holds, so
|
|
|
|
|
// the transient key buffer is free to serve the next statement.
|
|
|
|
|
func (p *parser) retainPath(key []string) []string {
|
|
|
|
|
if cap(p.currentPath) < len(key) {
|
|
|
|
|
p.currentPath = make([]string, len(key))
|
|
|
|
|
} else {
|
|
|
|
|
p.currentPath = p.currentPath[:len(key)]
|
|
|
|
|
}
|
|
|
|
|
copy(p.currentPath, key)
|
|
|
|
|
return p.currentPath
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
// tableAt walks (creating intermediate tables) to the table named by key,
|
|
|
|
|
// relative to the document root, rejecting any step into a frozen inline table.
|
2026-09-20 10:40:57 +02:00
|
|
|
func (p *parser) tableAt(key []string) (map[string]any, *Table, error) {
|
2026-08-19 09:47:00 +02:00
|
|
|
cur := p.root
|
2026-09-20 10:40:57 +02:00
|
|
|
node := p.doc
|
2026-09-20 22:15:10 +02:00
|
|
|
// The intermediate-path bookkeeping allocates only when the key actually
|
|
|
|
|
// has intermediate segments; a single-segment key checks its own name.
|
|
|
|
|
var path []string
|
|
|
|
|
if len(key) > 1 {
|
|
|
|
|
path = make([]string, 0, len(key))
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
for _, k := range key {
|
2026-09-20 22:15:10 +02:00
|
|
|
if len(key) == 1 {
|
|
|
|
|
if p.frozen[k] {
|
|
|
|
|
return nil, nil, p.errf("cannot extend inline table %q", k)
|
|
|
|
|
}
|
|
|
|
|
} else {
|
|
|
|
|
path = append(path, k)
|
|
|
|
|
if p.frozen[pathKey(path)] {
|
|
|
|
|
return nil, nil, p.errf("cannot extend inline table %q", strings.Join(path, "."))
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
existing, ok := cur[k]
|
|
|
|
|
if !ok {
|
|
|
|
|
next := map[string]any{}
|
|
|
|
|
cur[k] = next
|
|
|
|
|
cur = next
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
node = node.addTable(k, next)
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
switch v := existing.(type) {
|
|
|
|
|
case map[string]any:
|
|
|
|
|
cur = v
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
node = node.addTable(k, v)
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
case []map[string]any:
|
|
|
|
|
if len(v) == 0 {
|
2026-09-20 10:40:57 +02:00
|
|
|
return nil, nil, p.errf("key %q is an empty array of tables", k)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
cur = v[len(v)-1]
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
node = node.lastElement(k)
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
default:
|
2026-09-20 10:40:57 +02:00
|
|
|
return nil, nil, p.errf("key %q is not a table", k)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
return cur, node, nil
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
func (p *parser) appendArrayTable(key []string) (map[string]any, *Table, error) {
|
2026-08-19 09:47:00 +02:00
|
|
|
parent := p.root
|
2026-09-20 10:40:57 +02:00
|
|
|
node := p.doc
|
2026-09-20 22:15:10 +02:00
|
|
|
// As in tableAt, the path slice exists only for a multi-segment key; the
|
|
|
|
|
// loop below runs for those alone.
|
|
|
|
|
var path []string
|
|
|
|
|
if len(key) > 1 {
|
|
|
|
|
path = make([]string, 0, len(key))
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
for _, k := range key[:len(key)-1] {
|
2026-09-17 23:04:49 +02:00
|
|
|
path = append(path, k)
|
|
|
|
|
if p.frozen[pathKey(path)] {
|
2026-09-20 10:40:57 +02:00
|
|
|
return nil, nil, p.errf("cannot extend inline table %q", strings.Join(path, "."))
|
2026-09-17 23:04:49 +02:00
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
existing, ok := parent[k]
|
|
|
|
|
if !ok {
|
|
|
|
|
next := map[string]any{}
|
|
|
|
|
parent[k] = next
|
|
|
|
|
parent = next
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
node = node.addTable(k, next)
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
switch v := existing.(type) {
|
|
|
|
|
case map[string]any:
|
|
|
|
|
parent = v
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
node = node.addTable(k, v)
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
case []map[string]any:
|
|
|
|
|
parent = v[len(v)-1]
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
node = node.lastElement(k)
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
default:
|
2026-09-20 10:40:57 +02:00
|
|
|
return nil, nil, p.errf("key %q is not a table", k)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
leaf := key[len(key)-1]
|
|
|
|
|
tbl := map[string]any{}
|
|
|
|
|
switch existing := parent[leaf].(type) {
|
|
|
|
|
case nil:
|
|
|
|
|
parent[leaf] = []map[string]any{tbl}
|
|
|
|
|
case []map[string]any:
|
|
|
|
|
parent[leaf] = append(existing, tbl)
|
|
|
|
|
default:
|
2026-09-20 10:40:57 +02:00
|
|
|
return nil, nil, p.errf("key %q is not an array of tables", leaf)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
var elem *Table
|
|
|
|
|
if node != nil {
|
|
|
|
|
elem = node.addElement(leaf, tbl)
|
|
|
|
|
}
|
|
|
|
|
return tbl, elem, nil
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// --- key/value -------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseKeyValue() error {
|
2026-09-20 22:15:10 +02:00
|
|
|
first, rest, err := p.parseKeyPath()
|
2026-08-19 09:47:00 +02:00
|
|
|
if err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
p.skipInline()
|
|
|
|
|
if p.eof() || p.peek() != '=' {
|
|
|
|
|
return p.errf("expected '=' after key")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
p.skipInline()
|
|
|
|
|
|
|
|
|
|
val, err := p.parseValue()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
dest := p.current
|
2026-09-20 22:15:10 +02:00
|
|
|
// The absolute path of the key drives the dotted-key bookkeeping and the
|
|
|
|
|
// inline-table freeze. A single top-level key needs it only for the
|
|
|
|
|
// freeze, where a one-element path sits in the parser's scratch.
|
|
|
|
|
var abs []string
|
|
|
|
|
if len(rest) > 0 || len(p.currentPath) > 0 {
|
|
|
|
|
abs = make([]string, 0, len(p.currentPath)+len(rest)+1)
|
|
|
|
|
abs = append(abs, p.currentPath...)
|
|
|
|
|
abs = append(abs, first)
|
|
|
|
|
} else {
|
|
|
|
|
abs = append(p.absScratch[:0], first)
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
// dests collects the map each dotted key descended into, which the node
|
|
|
|
|
// tree needs to build the matching tables around the value.
|
|
|
|
|
var dests []map[string]any
|
2026-09-20 22:15:10 +02:00
|
|
|
leaf := first
|
|
|
|
|
if len(rest) > 0 {
|
|
|
|
|
if err := p.descendKey(&dest, first, abs, &dests); err != nil {
|
|
|
|
|
return err
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
for _, k := range rest[:len(rest)-1] {
|
|
|
|
|
abs = append(abs, k)
|
|
|
|
|
if err := p.descendKey(&dest, k, abs, &dests); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
leaf = rest[len(rest)-1]
|
|
|
|
|
abs = append(abs, leaf)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
if _, exists := dest[leaf]; exists {
|
|
|
|
|
return p.errf("duplicate key %q", leaf)
|
|
|
|
|
}
|
|
|
|
|
dest[leaf] = val
|
2026-09-20 10:40:57 +02:00
|
|
|
if p.doc != nil {
|
|
|
|
|
node := p.currentNode
|
2026-09-20 22:15:10 +02:00
|
|
|
if len(rest) > 0 {
|
2026-09-22 21:15:00 +02:00
|
|
|
// The tables a dotted key builds hold the position of a line, so
|
|
|
|
|
// the write side marks them and gives each leaf back as a dotted
|
|
|
|
|
// key rather than a header that would swallow the lines after it.
|
2026-09-20 22:15:10 +02:00
|
|
|
node = node.addTable(first, dests[0])
|
2026-09-22 21:15:00 +02:00
|
|
|
node.dotted = true
|
2026-09-20 22:15:10 +02:00
|
|
|
for i, k := range rest[:len(rest)-1] {
|
|
|
|
|
node = node.addTable(k, dests[i+1])
|
2026-09-22 21:15:00 +02:00
|
|
|
node.dotted = true
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
}
|
|
|
|
|
_, inline := val.(map[string]any)
|
|
|
|
|
entry := node.addValue(leaf, val, inline)
|
|
|
|
|
if inline {
|
|
|
|
|
entry.child = p.takeInline(val)
|
|
|
|
|
}
|
|
|
|
|
if nodes := p.takeArrayElems(val); nodes != nil {
|
|
|
|
|
entry.elements = nodes
|
|
|
|
|
}
|
|
|
|
|
p.lastEntry = entry
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
p.freezeInline(abs, val)
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// descendKey walks dest into the sub-table named key on the dotted path abs,
|
|
|
|
|
// recording the path in the definition maps; dests collects the maps
|
|
|
|
|
// descended into.
|
|
|
|
|
func (p *parser) descendKey(dest *map[string]any, key string, abs []string, dests *[]map[string]any) error {
|
|
|
|
|
ak := pathKey(abs)
|
|
|
|
|
if p.frozen[ak] {
|
|
|
|
|
return p.errf("cannot extend inline table %q", strings.Join(abs, "."))
|
|
|
|
|
}
|
|
|
|
|
if p.headers[ak] {
|
|
|
|
|
return p.errf("cannot extend table %q with a dotted key", strings.Join(abs, "."))
|
|
|
|
|
}
|
|
|
|
|
p.markDotted(ak)
|
|
|
|
|
existing, ok := (*dest)[key]
|
|
|
|
|
if !ok {
|
|
|
|
|
next := map[string]any{}
|
|
|
|
|
(*dest)[key] = next
|
|
|
|
|
*dest = next
|
|
|
|
|
*dests = append(*dests, next)
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
m, ok := existing.(map[string]any)
|
|
|
|
|
if !ok {
|
|
|
|
|
return p.errf("key %q is not a table", key)
|
|
|
|
|
}
|
|
|
|
|
*dest = m
|
|
|
|
|
*dests = append(*dests, m)
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
// takeInline returns the node of the inline table just parsed, when v is that
|
|
|
|
|
// table's value, and clears it so a later value cannot pick it up.
|
|
|
|
|
func (p *parser) takeInline(v any) *Table {
|
|
|
|
|
node := p.lastInline
|
|
|
|
|
p.lastInline = nil
|
|
|
|
|
if _, ok := v.(map[string]any); !ok {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
return node
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// takeArrayElems returns the element nodes of the array just parsed, when v is
|
|
|
|
|
// that array's value, and clears them.
|
|
|
|
|
func (p *parser) takeArrayElems(v any) []*Table {
|
|
|
|
|
nodes := p.lastArrayElems
|
|
|
|
|
p.lastArrayElems = nil
|
|
|
|
|
if _, ok := v.([]any); !ok {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
return nodes
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
// freezeInline marks the path of an inline table (and any nested inline tables)
|
2026-09-20 22:15:10 +02:00
|
|
|
// as immutable, so a later header or dotted key cannot extend it. The
|
|
|
|
|
// recursion appends into the caller's path slice; the frozen map keeps the
|
|
|
|
|
// joined strings, never the slice, so the backing is free to be reused.
|
2026-08-19 09:47:00 +02:00
|
|
|
func (p *parser) freezeInline(path []string, val any) {
|
|
|
|
|
m, ok := val.(map[string]any)
|
|
|
|
|
if !ok {
|
|
|
|
|
return
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
p.markFrozen(pathKey(path))
|
2026-08-19 09:47:00 +02:00
|
|
|
for k, v := range m {
|
2026-09-20 22:15:10 +02:00
|
|
|
p.freezeInline(append(path, k), v)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-17 23:05:06 +02:00
|
|
|
// resetScopeUnder forgets the definition records nested under key, which
|
|
|
|
|
// belong to the previous element of an array of tables: headers, frozen
|
|
|
|
|
// inline tables, dotted-key paths, and nested arrays of tables all start
|
2026-09-22 21:15:00 +02:00
|
|
|
// fresh in the new element. The records to drop are the ones the element
|
|
|
|
|
// added, which scopeMarks holds; the array's own entry, and everything
|
|
|
|
|
// outside it, keep their place.
|
2026-08-19 09:47:00 +02:00
|
|
|
func (p *parser) resetScopeUnder(key []string) {
|
2026-09-22 21:15:00 +02:00
|
|
|
pk := pathKey(key)
|
|
|
|
|
for _, k := range p.scopeMarks[pk] {
|
|
|
|
|
delete(p.headers, k)
|
|
|
|
|
delete(p.frozen, k)
|
|
|
|
|
delete(p.dotted, k)
|
|
|
|
|
delete(p.arrays, k)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
2026-09-22 21:15:00 +02:00
|
|
|
if p.scopeMarks != nil {
|
|
|
|
|
p.scopeMarks[pk] = nil
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// parseKeyPath parses a dotted key. The first component comes back directly
|
|
|
|
|
// and the rest as a usually nil slice, because a single-component key is the
|
|
|
|
|
// common shape and a fresh slice per statement is what the allocation profile
|
|
|
|
|
// showed. The single-key slice a caller sees is parser-owned and transient.
|
|
|
|
|
func (p *parser) parseKeyPath() (string, []string, error) {
|
|
|
|
|
p.skipInline()
|
|
|
|
|
first, err := p.parseKeyComponent()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return "", nil, err
|
|
|
|
|
}
|
|
|
|
|
p.skipInline()
|
|
|
|
|
if p.eof() || p.peek() != '.' {
|
|
|
|
|
return first, nil, nil
|
|
|
|
|
}
|
|
|
|
|
p.pos++
|
|
|
|
|
var rest []string
|
2026-08-19 09:47:00 +02:00
|
|
|
for {
|
|
|
|
|
p.skipInline()
|
|
|
|
|
part, err := p.parseKeyComponent()
|
|
|
|
|
if err != nil {
|
2026-09-20 22:15:10 +02:00
|
|
|
return "", nil, err
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
rest = append(rest, part)
|
2026-08-19 09:47:00 +02:00
|
|
|
p.skipInline()
|
2026-09-20 22:15:10 +02:00
|
|
|
if p.eof() || p.peek() != '.' {
|
|
|
|
|
return first, rest, nil
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseKeyComponent() (string, error) {
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return "", p.errf("expected a key")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
switch p.peek() {
|
2026-08-19 09:47:00 +02:00
|
|
|
case '"':
|
|
|
|
|
if p.lookahead(`"""`) {
|
|
|
|
|
return "", p.errf("multiline strings are not allowed in keys")
|
|
|
|
|
}
|
|
|
|
|
return p.parseBasicString()
|
|
|
|
|
case '\'':
|
|
|
|
|
if p.lookahead(`'''`) {
|
|
|
|
|
return "", p.errf("multiline strings are not allowed in keys")
|
|
|
|
|
}
|
|
|
|
|
return p.parseLiteralString()
|
|
|
|
|
default:
|
|
|
|
|
start := p.pos
|
|
|
|
|
for !p.eof() {
|
|
|
|
|
c := p.peek()
|
|
|
|
|
if (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
|
|
|
|
|
(c >= '0' && c <= '9') || c == '_' || c == '-' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
break
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
// The stopping byte decides the message: a multi-byte sequence that
|
|
|
|
|
// does not decode names that, before any grammar message can.
|
|
|
|
|
if !p.eof() && p.peek() >= utf8.RuneSelf {
|
|
|
|
|
if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 {
|
2026-09-21 23:55:58 +02:00
|
|
|
return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
if p.pos == start {
|
2026-09-17 23:11:50 +02:00
|
|
|
r, _ := utf8.DecodeRune(p.src[p.pos:])
|
|
|
|
|
return "", p.errf("invalid key character %q", string(r))
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
return p.internKey(p.src[start:p.pos]), nil
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// --- values ----------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseValue() (any, error) {
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return nil, p.errf("expected a value")
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
// A container value leaves its node behind for the caller to pick up; a
|
|
|
|
|
// value that follows must not find the previous one.
|
|
|
|
|
p.lastInline, p.lastArrayElems = nil, nil
|
2026-08-19 09:47:00 +02:00
|
|
|
switch c := p.peek(); {
|
|
|
|
|
case c == '"':
|
|
|
|
|
return p.parseBasicString()
|
|
|
|
|
case c == '\'':
|
|
|
|
|
return p.parseLiteralString()
|
|
|
|
|
case c == '[':
|
|
|
|
|
return p.parseArray()
|
|
|
|
|
case c == '{':
|
|
|
|
|
return p.parseInlineTable()
|
|
|
|
|
case c == 't' || c == 'f':
|
|
|
|
|
return p.parseBool()
|
|
|
|
|
default:
|
|
|
|
|
return p.parseAtom()
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseBool() (any, error) {
|
|
|
|
|
if p.match("true") {
|
|
|
|
|
return true, nil
|
|
|
|
|
}
|
|
|
|
|
if p.match("false") {
|
|
|
|
|
return false, nil
|
|
|
|
|
}
|
|
|
|
|
return nil, p.errf("invalid value")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// parseAtom handles numbers, inf/nan, and date-times.
|
|
|
|
|
func (p *parser) parseAtom() (any, error) {
|
|
|
|
|
start := p.pos
|
|
|
|
|
p.scanBareToken()
|
|
|
|
|
tok := string(p.src[start:p.pos])
|
|
|
|
|
if tok == "" {
|
|
|
|
|
return nil, p.errf("expected a value")
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
if hasHighByte(tok) && !utf8.ValidString(tok) {
|
2026-09-22 21:15:00 +02:00
|
|
|
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
// A date may be followed by a space and a time, forming one date-time.
|
|
|
|
|
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
|
|
|
|
|
if next, ok := p.peekAt(1); ok && next >= '0' && next <= '9' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // consume the separating space
|
2026-08-19 09:47:00 +02:00
|
|
|
timeStart := p.pos
|
|
|
|
|
p.scanBareToken()
|
|
|
|
|
tok = tok + " " + string(p.src[timeStart:p.pos])
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-09-22 21:15:00 +02:00
|
|
|
v, isDT, dterr := parseDateTime(tok)
|
|
|
|
|
if dterr != nil {
|
|
|
|
|
return nil, p.errf("%s", dterr)
|
|
|
|
|
}
|
|
|
|
|
if isDT {
|
2026-08-19 09:47:00 +02:00
|
|
|
return v, nil
|
|
|
|
|
}
|
|
|
|
|
v, err := decodeNumber(tok)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, p.errf("%s", err)
|
|
|
|
|
}
|
2026-09-22 18:42:18 +02:00
|
|
|
// The token's shape is validated either way; NumbersAsLiterals only keeps the
|
2026-09-21 23:49:39 +02:00
|
|
|
// literal instead of the evaluated value.
|
|
|
|
|
if p.useNumber {
|
|
|
|
|
return Number(tok), nil
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
return v, nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// scanBareToken advances past a bare value token (number, bool, or date-time),
|
|
|
|
|
// stopping at whitespace, a separator, or a comment.
|
|
|
|
|
func (p *parser) scanBareToken() {
|
|
|
|
|
for !p.eof() {
|
|
|
|
|
c := p.peek()
|
|
|
|
|
if c == ' ' || c == '\t' || c == '\n' || c == '\r' ||
|
|
|
|
|
c == ',' || c == ']' || c == '}' || c == '#' {
|
|
|
|
|
return
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// hasHighByte reports whether s holds any byte outside ASCII, the cheap gate
|
|
|
|
|
// in front of a full UTF-8 check.
|
|
|
|
|
func hasHighByte(s string) bool {
|
|
|
|
|
for i := range len(s) {
|
|
|
|
|
if s[i] >= utf8.RuneSelf {
|
|
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-22 21:15:00 +02:00
|
|
|
// invalidUTF8Offset returns the offset of the first byte in s that does not
|
|
|
|
|
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
|
|
|
|
|
// that is invalid rather than the end of the token around it.
|
|
|
|
|
func invalidUTF8Offset(s string) int {
|
|
|
|
|
for i := 0; i < len(s); {
|
|
|
|
|
r, size := utf8.DecodeRuneInString(s[i:])
|
|
|
|
|
if r == utf8.RuneError && size == 1 {
|
|
|
|
|
return i
|
|
|
|
|
}
|
|
|
|
|
i += size
|
|
|
|
|
}
|
|
|
|
|
return -1
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
// --- strings ---------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseBasicString() (string, error) {
|
|
|
|
|
if p.lookahead(`"""`) {
|
|
|
|
|
return p.parseMultilineString('"', true)
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // opening quote
|
2026-09-20 22:15:10 +02:00
|
|
|
start := p.pos
|
|
|
|
|
// A run of plain characters up to the closing quote needs no builder, only
|
|
|
|
|
// one copy at the end; escapes, controls and multi-byte runes fall through
|
|
|
|
|
// to the builder loop, which validates them on the spot.
|
|
|
|
|
for p.pos < len(p.src) {
|
|
|
|
|
c := p.src[p.pos]
|
|
|
|
|
if c == '"' {
|
|
|
|
|
s := string(p.src[start:p.pos])
|
|
|
|
|
p.pos++
|
|
|
|
|
return s, nil
|
|
|
|
|
}
|
|
|
|
|
if c == '\\' || c == '\n' || c == '\r' || c >= utf8.RuneSelf ||
|
|
|
|
|
(c < 0x20 && c != '\t') || c == 0x7f {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
p.pos++
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return "", p.errf("unterminated string")
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
var b strings.Builder
|
2026-09-20 22:15:10 +02:00
|
|
|
b.Grow(p.pos - start)
|
|
|
|
|
b.Write(p.src[start:p.pos])
|
|
|
|
|
return p.parseBasicStringRest(&b)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// parseBasicStringRest continues a basic string whose fast scan has met a byte
|
|
|
|
|
// it does not handle: an escape, a control character, a multi-byte rune, or a
|
|
|
|
|
// bare newline, which the loop rejects.
|
|
|
|
|
func (p *parser) parseBasicStringRest(b *strings.Builder) (string, error) {
|
2026-08-19 09:47:00 +02:00
|
|
|
for {
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return "", p.errf("unterminated string")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
c := p.peek()
|
2026-08-19 09:47:00 +02:00
|
|
|
switch c {
|
|
|
|
|
case '"':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return b.String(), nil
|
|
|
|
|
case '\n':
|
|
|
|
|
return "", p.errf("unterminated string")
|
|
|
|
|
case '\r':
|
|
|
|
|
return "", p.errf("bare carriage return is not allowed in a string")
|
|
|
|
|
case '\\':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
r, err := p.readEscape()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return "", err
|
|
|
|
|
}
|
|
|
|
|
b.WriteRune(r)
|
|
|
|
|
default:
|
2026-09-20 22:15:10 +02:00
|
|
|
if err := p.writeContentRune(b); err != nil {
|
2026-09-17 23:11:50 +02:00
|
|
|
return "", err
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseLiteralString() (string, error) {
|
|
|
|
|
if p.lookahead(`'''`) {
|
|
|
|
|
return p.parseMultilineString('\'', false)
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // opening quote
|
2026-09-20 22:15:10 +02:00
|
|
|
start := p.pos
|
|
|
|
|
// The same fast scan as the basic string, without the escape case.
|
|
|
|
|
for p.pos < len(p.src) {
|
|
|
|
|
c := p.src[p.pos]
|
|
|
|
|
if c == '\'' {
|
|
|
|
|
s := string(p.src[start:p.pos])
|
|
|
|
|
p.pos++
|
|
|
|
|
return s, nil
|
|
|
|
|
}
|
|
|
|
|
if c == '\n' || c == '\r' || c >= utf8.RuneSelf ||
|
|
|
|
|
(c < 0x20 && c != '\t') || c == 0x7f {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
p.pos++
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return "", p.errf("unterminated literal string")
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
var b strings.Builder
|
2026-09-20 22:15:10 +02:00
|
|
|
b.Grow(p.pos - start)
|
|
|
|
|
b.Write(p.src[start:p.pos])
|
2026-08-19 09:47:00 +02:00
|
|
|
for {
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return "", p.errf("unterminated literal string")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
c := p.peek()
|
|
|
|
|
switch c {
|
|
|
|
|
case '\'':
|
|
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return b.String(), nil
|
2026-09-17 23:11:50 +02:00
|
|
|
case '\n':
|
2026-08-19 09:47:00 +02:00
|
|
|
return "", p.errf("unterminated literal string")
|
2026-09-17 23:11:50 +02:00
|
|
|
case '\r':
|
2026-08-19 09:47:00 +02:00
|
|
|
return "", p.errf("bare carriage return is not allowed in a string")
|
2026-09-17 23:11:50 +02:00
|
|
|
default:
|
|
|
|
|
if err := p.writeContentRune(&b); err != nil {
|
|
|
|
|
return "", err
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-17 23:11:50 +02:00
|
|
|
// writeContentRune appends the rune at the cursor to b and advances past it.
|
|
|
|
|
// An ASCII byte, which includes every control character the grammar forbids,
|
2026-09-20 22:15:10 +02:00
|
|
|
// is checked and written directly; a multi-byte rune is decoded, and a
|
|
|
|
|
// sequence that does not decode is the UTF-8 error reported where it sits.
|
2026-09-17 23:11:50 +02:00
|
|
|
func (p *parser) writeContentRune(b *strings.Builder) error {
|
|
|
|
|
c := p.peek()
|
|
|
|
|
if c < utf8.RuneSelf {
|
|
|
|
|
if isControlRune(rune(c)) {
|
|
|
|
|
return p.errf("control character U+%04X is not allowed in a string", c)
|
|
|
|
|
}
|
|
|
|
|
p.pos++
|
|
|
|
|
b.WriteByte(c)
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
r, size := utf8.DecodeRune(p.src[p.pos:])
|
2026-09-20 22:15:10 +02:00
|
|
|
if r == utf8.RuneError && size == 1 {
|
2026-09-21 23:55:58 +02:00
|
|
|
return p.errf("invalid UTF-8 in string at byte offset %d", p.pos)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos += size
|
|
|
|
|
b.WriteRune(r)
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
|
2026-08-19 09:47:00 +02:00
|
|
|
p.skipN(3) // opening delimiter
|
2026-09-22 21:15:00 +02:00
|
|
|
// A newline immediately after the opening delimiter is trimmed, and it is
|
|
|
|
|
// a newline: a bare CR here is the bare-CR error like anywhere else, not
|
|
|
|
|
// a newline to trim.
|
2026-08-19 09:47:00 +02:00
|
|
|
if !p.eof() && p.peek() == '\r' {
|
2026-09-22 21:15:00 +02:00
|
|
|
if next, ok := p.peekAt(1); !ok || next != '\n' {
|
|
|
|
|
return "", p.errf("bare carriage return is not allowed in a string")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
if !p.eof() && p.peek() == '\n' {
|
|
|
|
|
p.line++
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var b strings.Builder
|
|
|
|
|
for {
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return "", p.errf("unterminated multiline string")
|
|
|
|
|
}
|
|
|
|
|
if p.peek() == quote {
|
|
|
|
|
// Count the run of delimiter characters. The last three close the
|
|
|
|
|
// string; up to two extra ones belong to the content.
|
|
|
|
|
n := 0
|
|
|
|
|
for p.pos+n < len(p.src) && p.src[p.pos+n] == quote {
|
|
|
|
|
n++
|
|
|
|
|
}
|
|
|
|
|
if n >= 3 {
|
|
|
|
|
if n > 5 {
|
|
|
|
|
return "", p.errf("too many '%c' before the closing delimiter", quote)
|
|
|
|
|
}
|
|
|
|
|
for range n - 3 {
|
2026-09-17 23:11:50 +02:00
|
|
|
b.WriteByte(quote)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
p.skipN(n)
|
|
|
|
|
return b.String(), nil
|
|
|
|
|
}
|
|
|
|
|
for range n {
|
2026-09-17 23:11:50 +02:00
|
|
|
b.WriteByte(quote)
|
|
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
continue
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
c := p.peek()
|
|
|
|
|
switch {
|
|
|
|
|
case c == '\n':
|
2026-08-19 09:47:00 +02:00
|
|
|
p.line++
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
|
|
|
|
b.WriteByte(c)
|
|
|
|
|
case c == '\r':
|
|
|
|
|
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
|
|
|
|
|
b.WriteByte(c)
|
|
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
return "", p.errf("bare carriage return is not allowed in a string")
|
2026-09-17 23:11:50 +02:00
|
|
|
case escapes && c == '\\':
|
|
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
// Line-ending backslash trims the following whitespace/newlines.
|
|
|
|
|
if p.trimLineEndingBackslash() {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
r, err := p.readEscape()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return "", err
|
|
|
|
|
}
|
|
|
|
|
b.WriteRune(r)
|
2026-09-17 23:11:50 +02:00
|
|
|
default:
|
|
|
|
|
if err := p.writeContentRune(&b); err != nil {
|
|
|
|
|
return "", err
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// trimLineEndingBackslash consumes whitespace through the next newline (and the
|
|
|
|
|
// blank lines that follow) when a backslash is the last token on a line.
|
|
|
|
|
// It reports whether it did so.
|
|
|
|
|
func (p *parser) trimLineEndingBackslash() bool {
|
|
|
|
|
save, saveLine := p.pos, p.line
|
|
|
|
|
for !p.eof() {
|
|
|
|
|
c := p.peek()
|
|
|
|
|
if c == ' ' || c == '\t' || c == '\r' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
if c == '\n' {
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
// Not a line-ending backslash; restore.
|
|
|
|
|
p.pos, p.line = save, saveLine
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
p.pos, p.line = save, saveLine
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
// Consume the newline and all following whitespace.
|
|
|
|
|
for !p.eof() {
|
|
|
|
|
c := p.peek()
|
|
|
|
|
if c == '\n' {
|
|
|
|
|
p.line++
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
if c == ' ' || c == '\t' || c == '\r' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) readEscape() (rune, error) {
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return 0, p.errf("unterminated escape sequence")
|
|
|
|
|
}
|
|
|
|
|
c := p.next()
|
|
|
|
|
switch c {
|
|
|
|
|
case 'b':
|
|
|
|
|
return '\b', nil
|
|
|
|
|
case 't':
|
|
|
|
|
return '\t', nil
|
|
|
|
|
case 'n':
|
|
|
|
|
return '\n', nil
|
|
|
|
|
case 'f':
|
|
|
|
|
return '\f', nil
|
|
|
|
|
case 'r':
|
|
|
|
|
return '\r', nil
|
2026-09-17 22:06:26 +02:00
|
|
|
case 'e':
|
|
|
|
|
// TOML 1.1: the escape character.
|
|
|
|
|
return '\x1b', nil
|
2026-08-19 09:47:00 +02:00
|
|
|
case '"':
|
|
|
|
|
return '"', nil
|
|
|
|
|
case '\\':
|
|
|
|
|
return '\\', nil
|
2026-09-17 22:06:26 +02:00
|
|
|
case 'x':
|
|
|
|
|
// TOML 1.1: two hex digits, code points 0x00 through 0xFF.
|
|
|
|
|
return p.readUnicode(2)
|
2026-08-19 09:47:00 +02:00
|
|
|
case 'u':
|
|
|
|
|
return p.readUnicode(4)
|
|
|
|
|
case 'U':
|
|
|
|
|
return p.readUnicode(8)
|
|
|
|
|
default:
|
2026-09-17 23:11:50 +02:00
|
|
|
// The byte just consumed starts a rune: the backslash before it is a
|
|
|
|
|
// boundary, and the input is valid UTF-8.
|
|
|
|
|
r, _ := utf8.DecodeRune(p.src[p.pos-1:])
|
|
|
|
|
return 0, p.errf("invalid escape sequence \\%c", r)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) readUnicode(n int) (rune, error) {
|
|
|
|
|
if p.pos+n > len(p.src) {
|
|
|
|
|
return 0, p.errf("invalid unicode escape")
|
|
|
|
|
}
|
|
|
|
|
hex := string(p.src[p.pos : p.pos+n])
|
|
|
|
|
p.pos += n
|
2026-09-22 21:15:00 +02:00
|
|
|
// ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
|
|
|
|
|
// read would let "\U-0000001" through the range checks below only to
|
|
|
|
|
// write U+FFFD for a document the grammar rejects.
|
|
|
|
|
v, err := strconv.ParseUint(hex, 16, 32)
|
2026-08-19 09:47:00 +02:00
|
|
|
if err != nil {
|
|
|
|
|
return 0, p.errf("invalid unicode escape \\%s", hex)
|
|
|
|
|
}
|
|
|
|
|
if v > 0x10FFFF || (v >= 0xD800 && v <= 0xDFFF) {
|
|
|
|
|
return 0, p.errf("escape \\%s is not a valid Unicode scalar value", hex)
|
|
|
|
|
}
|
|
|
|
|
return rune(v), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// --- arrays and inline tables ---------------------------------------------
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
func (p *parser) parseArray() (val any, err error) {
|
2026-09-19 19:36:24 +02:00
|
|
|
if err := p.enterNesting(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
defer p.leaveNesting()
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // '['
|
2026-09-20 22:15:10 +02:00
|
|
|
// A small presize covers the arrays documents actually hold, and trades a
|
|
|
|
|
// little capacity on tiny arrays for the growth chain an append-from-nil
|
|
|
|
|
// costs per array.
|
|
|
|
|
arr := make([]any, 0, 4)
|
2026-09-20 10:40:57 +02:00
|
|
|
// elems carries the node of each element that is an inline table, so the
|
|
|
|
|
// caller can keep its key order; the entries are nil for other values.
|
|
|
|
|
var elems []*Table
|
|
|
|
|
if p.doc != nil {
|
|
|
|
|
defer func() {
|
|
|
|
|
if err == nil {
|
|
|
|
|
p.lastArrayElems = elems
|
|
|
|
|
}
|
|
|
|
|
}()
|
|
|
|
|
}
|
2026-09-22 00:44:23 +02:00
|
|
|
for i := 0; ; i++ {
|
|
|
|
|
// A container the size of memory should answer cancellation inside the
|
|
|
|
|
// value, not only between statements, so the element loops check the
|
|
|
|
|
// context on their own cadence.
|
|
|
|
|
if i%ctxCheckInterval == 0 {
|
|
|
|
|
if err := p.checkCtx(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-09-17 22:06:26 +02:00
|
|
|
if err := p.skipNestedSpace(); err != nil {
|
2026-08-19 09:47:00 +02:00
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return nil, p.errf("unterminated array")
|
|
|
|
|
}
|
|
|
|
|
if p.peek() == ']' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return arr, nil
|
|
|
|
|
}
|
|
|
|
|
v, err := p.parseValue()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
if p.doc != nil {
|
|
|
|
|
elems = append(elems, p.takeInline(v))
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
arr = append(arr, v)
|
2026-09-17 22:06:26 +02:00
|
|
|
if err := p.skipNestedSpace(); err != nil {
|
2026-08-19 09:47:00 +02:00
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return nil, p.errf("unterminated array")
|
|
|
|
|
}
|
|
|
|
|
switch p.peek() {
|
|
|
|
|
case ',':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
case ']':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return arr, nil
|
|
|
|
|
default:
|
|
|
|
|
return nil, p.errf("expected ',' or ']' in array")
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
func (p *parser) parseInlineTable() (val any, err error) {
|
2026-09-19 19:36:24 +02:00
|
|
|
if err := p.enterNesting(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
defer p.leaveNesting()
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // '{'
|
2026-08-19 09:47:00 +02:00
|
|
|
tbl := map[string]any{}
|
2026-09-20 22:15:10 +02:00
|
|
|
// assigned tracks the dotted paths written into this table. It is created
|
|
|
|
|
// on the first key, so an empty inline table allocates nothing for it.
|
|
|
|
|
var assigned map[string]bool
|
2026-09-20 10:40:57 +02:00
|
|
|
// The inline table is a node of its own, so the keys keep their order; the
|
|
|
|
|
// caller picks the node up when the table parses.
|
|
|
|
|
var node *Table
|
|
|
|
|
if p.doc != nil {
|
|
|
|
|
node = newTable(tbl)
|
|
|
|
|
node.inline = true
|
|
|
|
|
defer func() {
|
|
|
|
|
if err == nil {
|
|
|
|
|
p.lastInline = node
|
|
|
|
|
}
|
|
|
|
|
}()
|
|
|
|
|
}
|
2026-09-17 22:06:26 +02:00
|
|
|
// TOML 1.1 lets an inline table span lines: interior whitespace includes
|
|
|
|
|
// newlines and comments, and a trailing comma is allowed before the
|
|
|
|
|
// closing brace.
|
|
|
|
|
if err := p.skipNestedSpace(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
if !p.eof() && p.peek() == '}' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return tbl, nil
|
|
|
|
|
}
|
2026-09-22 00:44:23 +02:00
|
|
|
for i := 0; ; i++ {
|
|
|
|
|
if i%ctxCheckInterval == 0 {
|
|
|
|
|
if err := p.checkCtx(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-09-17 22:06:26 +02:00
|
|
|
if err := p.skipNestedSpace(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
first, rest, err := p.parseKeyPath()
|
2026-08-19 09:47:00 +02:00
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
p.skipInline()
|
|
|
|
|
if p.eof() || p.peek() != '=' {
|
|
|
|
|
return nil, p.errf("expected '=' in inline table")
|
|
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
p.skipInline()
|
|
|
|
|
val, err := p.parseValue()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
dest := tbl
|
2026-09-20 22:15:10 +02:00
|
|
|
var path []string
|
2026-09-20 10:40:57 +02:00
|
|
|
var dests []map[string]any
|
2026-09-20 22:15:10 +02:00
|
|
|
leaf := first
|
|
|
|
|
if len(rest) > 0 {
|
|
|
|
|
path = append(p.absScratch[:0], first)
|
|
|
|
|
d, err := p.descendInline(&dest, first, path, assigned)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
dests = append(dests, d)
|
|
|
|
|
for _, k := range rest[:len(rest)-1] {
|
|
|
|
|
path = append(path, k)
|
|
|
|
|
d, err := p.descendInline(&dest, k, path, assigned)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
dests = append(dests, d)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
leaf = rest[len(rest)-1]
|
|
|
|
|
path = append(path, leaf)
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
if _, exists := dest[leaf]; exists {
|
|
|
|
|
return nil, p.errf("duplicate key %q in inline table", leaf)
|
|
|
|
|
}
|
|
|
|
|
dest[leaf] = val
|
2026-09-20 22:15:10 +02:00
|
|
|
if assigned == nil {
|
|
|
|
|
assigned = make(map[string]bool, 4)
|
|
|
|
|
}
|
|
|
|
|
if len(rest) == 0 {
|
|
|
|
|
assigned[first] = true
|
|
|
|
|
} else {
|
|
|
|
|
assigned[pathKey(path)] = true
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
if node != nil {
|
|
|
|
|
child := node
|
2026-09-20 22:15:10 +02:00
|
|
|
if len(rest) > 0 {
|
|
|
|
|
child = child.addTable(first, dests[0])
|
|
|
|
|
for i, k := range rest[:len(rest)-1] {
|
|
|
|
|
child = child.addTable(k, dests[i+1])
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
}
|
|
|
|
|
_, inline := val.(map[string]any)
|
|
|
|
|
entry := child.addValue(leaf, val, inline)
|
|
|
|
|
if inline {
|
|
|
|
|
entry.child = p.takeInline(val)
|
|
|
|
|
}
|
|
|
|
|
if nodes := p.takeArrayElems(val); nodes != nil {
|
|
|
|
|
entry.elements = nodes
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
|
2026-09-17 22:06:26 +02:00
|
|
|
if err := p.skipNestedSpace(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
if p.eof() {
|
|
|
|
|
return nil, p.errf("unterminated inline table")
|
|
|
|
|
}
|
|
|
|
|
switch p.peek() {
|
|
|
|
|
case ',':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-09-17 22:06:26 +02:00
|
|
|
if err := p.skipNestedSpace(); err != nil {
|
|
|
|
|
return nil, err
|
|
|
|
|
}
|
|
|
|
|
if !p.eof() && p.peek() == '}' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-09-17 22:06:26 +02:00
|
|
|
return tbl, nil
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
case '}':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return tbl, nil
|
|
|
|
|
default:
|
|
|
|
|
return nil, p.errf("expected ',' or '}' in inline table")
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 22:15:10 +02:00
|
|
|
// descendInline walks dest into the sub-table named key inside an inline
|
|
|
|
|
// table, rejecting a dotted segment the table has already defined.
|
|
|
|
|
func (p *parser) descendInline(dest *map[string]any, key string, path []string, assigned map[string]bool) (map[string]any, error) {
|
|
|
|
|
pk := pathKey(path)
|
|
|
|
|
if assigned[pk] {
|
|
|
|
|
return nil, p.errf("key %q is already defined", strings.Join(path, "."))
|
|
|
|
|
}
|
|
|
|
|
existing, ok := (*dest)[key]
|
|
|
|
|
if !ok {
|
|
|
|
|
m := map[string]any{}
|
|
|
|
|
(*dest)[key] = m
|
|
|
|
|
*dest = m
|
|
|
|
|
return m, nil
|
|
|
|
|
}
|
|
|
|
|
m, isMap := existing.(map[string]any)
|
|
|
|
|
if !isMap {
|
|
|
|
|
return nil, p.errf("key %q is already defined", key)
|
|
|
|
|
}
|
|
|
|
|
*dest = m
|
|
|
|
|
return m, nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-19 09:47:00 +02:00
|
|
|
// --- scanning helpers ------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
func (p *parser) eof() bool { return p.pos >= len(p.src) }
|
2026-09-17 23:11:50 +02:00
|
|
|
func (p *parser) peek() byte { return p.src[p.pos] }
|
2026-08-19 09:47:00 +02:00
|
|
|
|
2026-09-17 23:11:50 +02:00
|
|
|
// peekAt returns the byte at offset n from the current position and whether the
|
2026-08-19 09:47:00 +02:00
|
|
|
// offset is within the source. Use it instead of indexing p.src directly when
|
|
|
|
|
// the offset may sit past the end.
|
2026-09-17 23:11:50 +02:00
|
|
|
func (p *parser) peekAt(n int) (byte, bool) {
|
2026-08-19 09:47:00 +02:00
|
|
|
i := p.pos + n
|
|
|
|
|
if i < 0 || i >= len(p.src) {
|
|
|
|
|
return 0, false
|
|
|
|
|
}
|
|
|
|
|
return p.src[i], true
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-17 23:11:50 +02:00
|
|
|
func (p *parser) next() byte {
|
2026-08-19 09:47:00 +02:00
|
|
|
c := p.src[p.pos]
|
|
|
|
|
p.pos++
|
|
|
|
|
return c
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) skipN(n int) {
|
|
|
|
|
for i := 0; i < n && !p.eof(); i++ {
|
|
|
|
|
p.next()
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *parser) match(word string) bool {
|
|
|
|
|
if p.lookahead(word) {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.skipN(len(word))
|
2026-08-19 09:47:00 +02:00
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-17 23:11:50 +02:00
|
|
|
// lookahead reports whether s follows the cursor. Every lookahead argument in
|
|
|
|
|
// the grammar is ASCII, so comparing bytes is exact.
|
2026-08-19 09:47:00 +02:00
|
|
|
func (p *parser) lookahead(s string) bool {
|
2026-09-17 23:11:50 +02:00
|
|
|
return p.pos+len(s) <= len(p.src) && string(p.src[p.pos:p.pos+len(s)]) == s
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// skipInline consumes spaces and tabs only.
|
|
|
|
|
func (p *parser) skipInline() {
|
|
|
|
|
for !p.eof() {
|
|
|
|
|
if c := p.peek(); c == ' ' || c == '\t' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
break
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-17 22:06:26 +02:00
|
|
|
// skipNestedSpace consumes whitespace, newlines, and comments inside a value
|
|
|
|
|
// container (an array, or an inline table under TOML 1.1).
|
|
|
|
|
func (p *parser) skipNestedSpace() error {
|
2026-08-19 09:47:00 +02:00
|
|
|
for !p.eof() {
|
|
|
|
|
switch p.peek() {
|
|
|
|
|
case ' ', '\t':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
case '\r':
|
|
|
|
|
if err := p.expectCRLF(); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
case '\n':
|
|
|
|
|
p.line++
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
case '#':
|
2026-09-20 10:40:57 +02:00
|
|
|
// A comment between values inside an array or an inline table is
|
|
|
|
|
// skipped; the Document does not carry those yet.
|
|
|
|
|
if _, err := p.skipComment(); err != nil {
|
2026-08-19 09:47:00 +02:00
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
default:
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// skipBlank consumes whitespace, blank lines, and comments between statements.
|
|
|
|
|
func (p *parser) skipBlank() error {
|
|
|
|
|
for !p.eof() {
|
|
|
|
|
switch p.peek() {
|
|
|
|
|
case ' ', '\t':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
case '\r':
|
|
|
|
|
if err := p.expectCRLF(); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
case '\n':
|
|
|
|
|
p.line++
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
case '#':
|
2026-09-20 10:40:57 +02:00
|
|
|
line, err := p.skipComment()
|
|
|
|
|
if err != nil {
|
2026-08-19 09:47:00 +02:00
|
|
|
return err
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
p.pending = append(p.pending, line)
|
2026-08-19 09:47:00 +02:00
|
|
|
default:
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-20 10:40:57 +02:00
|
|
|
func (p *parser) skipComment() (string, error) {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // consume '#'
|
2026-09-20 10:40:57 +02:00
|
|
|
start := p.pos
|
2026-08-19 09:47:00 +02:00
|
|
|
for !p.eof() {
|
|
|
|
|
c := p.peek()
|
|
|
|
|
switch {
|
|
|
|
|
case c == '\n':
|
2026-09-20 10:40:57 +02:00
|
|
|
return commentText(string(p.src[start:p.pos])), nil
|
2026-08-19 09:47:00 +02:00
|
|
|
case c == '\r':
|
|
|
|
|
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
|
2026-09-20 10:40:57 +02:00
|
|
|
return commentText(string(p.src[start:p.pos])), nil
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
return "", p.errf("bare carriage return is not allowed")
|
2026-08-19 09:47:00 +02:00
|
|
|
case c == '\t':
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
case c < 0x20 || c == 0x7f:
|
2026-09-20 10:40:57 +02:00
|
|
|
return "", p.errf("control character U+%04X is not allowed in a comment", c)
|
2026-09-20 22:15:10 +02:00
|
|
|
case c < utf8.RuneSelf:
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-09-20 22:15:10 +02:00
|
|
|
default:
|
|
|
|
|
r, size := utf8.DecodeRune(p.src[p.pos:])
|
|
|
|
|
if r == utf8.RuneError && size == 1 {
|
2026-09-21 23:55:58 +02:00
|
|
|
return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
|
|
|
|
p.pos += size
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
return commentText(string(p.src[start:p.pos])), nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// commentText drops the one space that usually follows the '#', so a line
|
|
|
|
|
// stored in a Document reads as the comment itself.
|
|
|
|
|
func commentText(s string) string {
|
|
|
|
|
return strings.TrimPrefix(s, " ")
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// expectCRLF consumes a carriage return that must be immediately followed by a
|
|
|
|
|
// line feed; a bare CR is invalid.
|
|
|
|
|
func (p *parser) expectCRLF() error {
|
|
|
|
|
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++ // consume CR; the LF is handled by the caller
|
2026-08-19 09:47:00 +02:00
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
return p.errf("bare carriage return is not allowed")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// expectLineEnd consumes trailing inline whitespace and an optional comment,
|
|
|
|
|
// then requires a newline or end of input.
|
|
|
|
|
func (p *parser) expectLineEnd() error {
|
|
|
|
|
p.skipInline()
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
if p.peek() == '#' {
|
2026-09-20 10:40:57 +02:00
|
|
|
line, err := p.skipComment()
|
|
|
|
|
if err != nil {
|
2026-08-19 09:47:00 +02:00
|
|
|
return err
|
|
|
|
|
}
|
2026-09-20 10:40:57 +02:00
|
|
|
if p.doc != nil {
|
|
|
|
|
p.trailing = line
|
|
|
|
|
}
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
if p.peek() == '\r' {
|
|
|
|
|
if err := p.expectCRLF(); err != nil {
|
|
|
|
|
return err
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if p.eof() {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
if p.peek() == '\n' {
|
|
|
|
|
p.line++
|
2026-09-17 23:11:50 +02:00
|
|
|
p.pos++
|
2026-08-19 09:47:00 +02:00
|
|
|
return nil
|
|
|
|
|
}
|
2026-09-20 22:15:10 +02:00
|
|
|
r, size := utf8.DecodeRune(p.src[p.pos:])
|
|
|
|
|
if r == utf8.RuneError && size == 1 {
|
2026-09-21 23:55:58 +02:00
|
|
|
return p.errf("invalid UTF-8 after value at byte offset %d", p.pos)
|
2026-09-20 22:15:10 +02:00
|
|
|
}
|
2026-09-17 23:11:50 +02:00
|
|
|
return p.errf("unexpected %q after value", string(r))
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
2026-09-21 23:55:58 +02:00
|
|
|
// errf builds the SyntaxError with the position the scan stopped at: the line,
|
|
|
|
|
// the byte offset in the input, and the 1-based column on that line. The
|
|
|
|
|
// offset is the cursor, which on an escape or a delimiter run sits just after
|
|
|
|
|
// the bytes that caused the complaint; SourceLine renders the caret there.
|
2026-08-19 09:47:00 +02:00
|
|
|
func (p *parser) errf(format string, args ...any) error {
|
2026-09-21 23:55:58 +02:00
|
|
|
col := p.pos + 1
|
|
|
|
|
if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 {
|
|
|
|
|
col = p.pos - start
|
|
|
|
|
}
|
|
|
|
|
return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)}
|
2026-08-19 09:47:00 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// pathKey joins key components with a NUL separator so a dotted path can be
|
2026-09-20 22:15:10 +02:00
|
|
|
// used as a map key for tracking defined tables. A single component comes
|
|
|
|
|
// back as it is, with no join and no copy.
|
2026-08-19 09:47:00 +02:00
|
|
|
func pathKey(parts []string) string {
|
|
|
|
|
return strings.Join(parts, "\x00")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// isControlRune reports whether r is a control character disallowed in a string
|
|
|
|
|
// literal. Tab, line feed, and carriage return are permitted (handled
|
|
|
|
|
// elsewhere); everything else below U+0020, plus U+007F, is rejected.
|
|
|
|
|
func isControlRune(r rune) bool {
|
|
|
|
|
if r == '\t' || r == '\n' || r == '\r' {
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
return r < 0x20 || r == 0x7f
|
|
|
|
|
}
|