430 lines
17 KiB
Go
430 lines
17 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
// Package interpres is a dependency-free TOML parser for Go.
|
|
//
|
|
// interpres reads and writes TOML documents using only the standard library.
|
|
// It exposes a small, encoding/json-style API:
|
|
//
|
|
// var cfg Config
|
|
// err := interpres.Unmarshal(data, &cfg)
|
|
//
|
|
// out, err := interpres.Marshal(cfg)
|
|
//
|
|
// or, for the document with its key order and comments:
|
|
//
|
|
// doc, err := interpres.Parse(data)
|
|
// tree := doc.Map()
|
|
//
|
|
// A Decoder allows strict decoding that rejects keys without a matching
|
|
// struct field, mirroring (*json.Decoder).DisallowUnknownFields.
|
|
package interpres
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
)
|
|
|
|
// A SyntaxError describes a malformed TOML document, including the 1-based
|
|
// line on which the problem was detected.
|
|
type SyntaxError struct {
|
|
Line int
|
|
Msg string
|
|
}
|
|
|
|
func (e *SyntaxError) Error() string {
|
|
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
|
|
}
|
|
|
|
// A DecodeError wraps a decoding failure with the key path at which it
|
|
// happened. Path lists one segment per level from the document root, the
|
|
// outermost key first: a key contributes its name and an array element its
|
|
// bracketed index, so the path of the weight field in the first item reads
|
|
// ["items", "[0]", "weight"]. The rendered message is unchanged by the type;
|
|
// read it programmatically with errors.AsType:
|
|
//
|
|
// if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
|
|
// fmt.Println(de.Path, de.Err)
|
|
// }
|
|
type DecodeError struct {
|
|
// Path is the key path from the document root, outermost key first.
|
|
Path []string
|
|
// Err is the failure at that path.
|
|
Err error
|
|
}
|
|
|
|
func (e *DecodeError) Error() string { return e.Path[0] + ": " + e.Err.Error() }
|
|
|
|
// Unwrap returns the failure the path points at.
|
|
func (e *DecodeError) Unwrap() error { return e.Err }
|
|
|
|
// newDecodeError wraps err with one path segment. The rest of the path comes
|
|
// from the DecodeError err already carries, if any: the decoder wraps each
|
|
// key and index on its way down, so the innermost wrap holds the deepest
|
|
// segments and each outer wrap prepends one.
|
|
func newDecodeError(key string, err error) *DecodeError {
|
|
path := make([]string, 0, 4)
|
|
path = append(path, key)
|
|
if de, ok := errors.AsType[*DecodeError](err); ok {
|
|
path = append(path, de.Path...)
|
|
}
|
|
return &DecodeError{Path: path, Err: err}
|
|
}
|
|
|
|
// An EncodeError wraps an encoding failure with the key path of the value
|
|
// that failed, in the notation of a TOML document: fields join with dots and
|
|
// an array element carries its bracketed index, so the path of the third
|
|
// port under server reads "server.ports[2]". The rendered message is
|
|
// unchanged by the type; read it programmatically with errors.AsType.
|
|
type EncodeError struct {
|
|
// Path is the key path of the failing value.
|
|
Path string
|
|
// Err is the failure at that path.
|
|
Err error
|
|
}
|
|
|
|
func (e *EncodeError) Error() string { return "interpres: " + e.Path + ": " + e.Err.Error() }
|
|
|
|
// Unwrap returns the failure the path points at.
|
|
func (e *EncodeError) Unwrap() error { return e.Err }
|
|
|
|
// Parse decodes a TOML document into a Document: the values, the order the
|
|
// keys were written in, whether a table was written inline, and the comments.
|
|
// ParseMap gives the plain value tree instead.
|
|
//
|
|
// Values are mapped to Go types as follows: strings to string, integers to
|
|
// int64, floats to float64, booleans to bool, offset date-times to
|
|
// OffsetDateTime, the local date-time kinds to their wrappers, arrays to
|
|
// []any, and tables (including inline tables) to map[string]any.
|
|
//
|
|
// Parse is equivalent to ParseContext with context.Background.
|
|
func Parse(data []byte) (*Document, error) {
|
|
return ParseContext(context.Background(), data)
|
|
}
|
|
|
|
// ParseContext decodes a TOML document into a Document, obeying ctx. The
|
|
// context is checked between top-level statements so cancellation is honoured
|
|
// before the parser has done substantial work.
|
|
func ParseContext(ctx context.Context, data []byte) (*Document, error) {
|
|
_, doc, err := parseWithOptions(ctx, data, parseOptions{}, true)
|
|
return doc, err
|
|
}
|
|
|
|
// ParseMap decodes a TOML document into a nested map[string]any, the value
|
|
// tree without the order and the comments a Document carries. It is the shape
|
|
// this package parsed into before [Document] existed.
|
|
//
|
|
// ParseMap is equivalent to ParseMapContext with context.Background.
|
|
func ParseMap(data []byte) (map[string]any, error) {
|
|
return ParseMapContext(context.Background(), data)
|
|
}
|
|
|
|
// ParseMapContext is the cancellable variant of ParseMap.
|
|
func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error) {
|
|
tree, _, err := parseWithOptions(ctx, data, parseOptions{}, false)
|
|
return tree, err
|
|
}
|
|
|
|
// parseOptions bound the work one parse may do. A zero field takes the
|
|
// default.
|
|
type parseOptions struct {
|
|
maxDepth int
|
|
maxInputSize int
|
|
}
|
|
|
|
// parseWithOptions parses data, building the node tree of a Document when
|
|
// wantDoc asks for it, and returns both the value tree and that document.
|
|
func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantDoc bool) (map[string]any, *Document, error) {
|
|
if err := ctx.Err(); err != nil {
|
|
return nil, nil, err
|
|
}
|
|
if opts.maxInputSize > 0 && len(data) > opts.maxInputSize {
|
|
return nil, nil, fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), opts.maxInputSize)
|
|
}
|
|
// UTF-8 validity is not checked in a pass of its own: the scanner
|
|
// validates the multi-byte sequences where it meets them, so an invalid
|
|
// byte is reported on its own line instead of always on line 1.
|
|
maxDepth := opts.maxDepth
|
|
if maxDepth <= 0 {
|
|
maxDepth = maxNestingDepth
|
|
}
|
|
// The parser scans data in place; it only reads the buffer, and every
|
|
// string it stores in the tree is copied out of it.
|
|
p := &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, wantDoc: wantDoc}
|
|
tree, err := p.parse()
|
|
if err != nil {
|
|
return nil, nil, err
|
|
}
|
|
if !wantDoc {
|
|
return tree, nil, nil
|
|
}
|
|
return tree, &Document{root: p.doc, footer: p.footer}, nil
|
|
}
|
|
|
|
// Unmarshal parses a TOML document and stores the result in the value pointed
|
|
// to by v. v is typically a pointer to a struct or to a map[string]any.
|
|
//
|
|
// Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a
|
|
// case-insensitive match on the field name when no tag is present. A tag of
|
|
// "-" skips the field.
|
|
//
|
|
// A destination implementing Unmarshaler receives the parsed value as it is,
|
|
// a TOML string fills a destination implementing encoding.TextUnmarshaler, and
|
|
// a time.Duration destination takes a duration literal such as `1h30m` or a
|
|
// bare integer as its nanosecond count.
|
|
//
|
|
// Unmarshal is equivalent to UnmarshalContext with context.Background.
|
|
func Unmarshal(data []byte, v any) error {
|
|
return UnmarshalContext(context.Background(), data, v)
|
|
}
|
|
|
|
// UnmarshalContext is the cancellable variant of Unmarshal.
|
|
func UnmarshalContext(ctx context.Context, data []byte, v any) error {
|
|
tree, err := ParseMapContext(ctx, data)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
return newDecoder().decode(tree, v)
|
|
}
|
|
|
|
// A Decoder decodes a TOML document into a Go value with configurable
|
|
// strictness and configurable limits on the parse it performs.
|
|
type Decoder struct {
|
|
disallowUnknown bool
|
|
maxDepth int
|
|
maxInputSize int
|
|
}
|
|
|
|
// NewDecoder returns a Decoder.
|
|
func NewDecoder() *Decoder { return &Decoder{} }
|
|
|
|
// DisallowUnknownFields causes Decode to return an error when the document
|
|
// contains a key with no matching destination struct field.
|
|
func (d *Decoder) DisallowUnknownFields() *Decoder {
|
|
d.disallowUnknown = true
|
|
return d
|
|
}
|
|
|
|
// MaxDepth bounds how deeply arrays and inline tables may nest in a document
|
|
// this decoder accepts. The parser is a recursive descent, so a document that
|
|
// nests without bound would exhaust the stack; one that nests deeper than the
|
|
// limit is rejected with a SyntaxError naming it instead. Use 0 or any
|
|
// negative value for the default of 10000, which no hand-written document
|
|
// approaches.
|
|
func (d *Decoder) MaxDepth(depth int) *Decoder {
|
|
d.maxDepth = depth
|
|
return d
|
|
}
|
|
|
|
// MaxInputSize bounds the size of a document this decoder accepts, in bytes; a
|
|
// larger one is rejected before parsing starts. Use 0 or any negative value for
|
|
// no limit, which is the default: the caller already holds the bytes, so the
|
|
// size is a policy the caller sets rather than a protection the library
|
|
// imposes on its own. Parse and ParseContext take no limit beyond the nesting
|
|
// default.
|
|
func (d *Decoder) MaxInputSize(size int) *Decoder {
|
|
d.maxInputSize = size
|
|
return d
|
|
}
|
|
|
|
// Decode parses data and stores the result in the value pointed to by v,
|
|
// honouring the decoder's strictness settings.
|
|
//
|
|
// Decode is equivalent to DecodeContext with context.Background.
|
|
func (d *Decoder) Decode(data []byte, v any) error {
|
|
return d.DecodeContext(context.Background(), data, v)
|
|
}
|
|
|
|
// DecodeContext is the cancellable variant of Decode.
|
|
func (d *Decoder) DecodeContext(ctx context.Context, data []byte, v any) error {
|
|
tree, _, err := parseWithOptions(ctx, data, parseOptions{
|
|
maxDepth: d.maxDepth,
|
|
maxInputSize: d.maxInputSize,
|
|
}, false)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
dec := newDecoder()
|
|
dec.disallowUnknown = d.disallowUnknown
|
|
return dec.decode(tree, v)
|
|
}
|
|
|
|
// Marshaler is the interface implemented by types that can produce a custom
|
|
// TOML representation of themselves. MarshalTOML returns a value that Marshal
|
|
// then encodes as if the returned value had been passed in its place, which
|
|
// is useful for emitting a Go type as a different TOML shape (for example, a
|
|
// struct as an inline table or a primitive alias as a richer value).
|
|
//
|
|
// MarshalTOML wins over encoding.TextMarshaler when a type implements both.
|
|
// A type that implements only encoding.TextMarshaler is encoded as a TOML
|
|
// string holding its text, and needs no method here.
|
|
type Marshaler interface {
|
|
MarshalTOML() (any, error)
|
|
}
|
|
|
|
// Unmarshaler is the inverse of Marshaler: a type that wants control over
|
|
// how it is decoded from a TOML value may implement UnmarshalTOML. The data
|
|
// argument is whatever the parser produced for that key: one of string,
|
|
// bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime,
|
|
// []any, or map[string]any. A tree built by hand may carry a plain time.Time
|
|
// where the parser would put an OffsetDateTime.
|
|
//
|
|
// UnmarshalTOML may parse, inspect, or transform the value however it likes,
|
|
// then store the result by mutating its receiver through the standard
|
|
// pointer-indirection rules of the reflect package (i.e. via
|
|
// reflect.Value.Set or by reassigning fields through a pointer the receiver
|
|
// holds).
|
|
//
|
|
// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the
|
|
// destination type implements the interface. The decoder does not need to
|
|
// consult the concrete return value; whatever the receiver stores is kept.
|
|
//
|
|
// UnmarshalTOML wins over encoding.TextUnmarshaler when a type implements
|
|
// both. A type that implements only encoding.TextUnmarshaler is filled from a
|
|
// TOML string holding its text, and needs no method here.
|
|
type Unmarshaler interface {
|
|
UnmarshalTOML(data any) error
|
|
}
|
|
|
|
// Marshal returns the TOML encoding of v. The output is valid TOML 1.1.
|
|
//
|
|
// Marshal traverses v using reflection and applies the following rules:
|
|
//
|
|
// - The top-level value must be a struct or a map[string]V. Pointers are
|
|
// followed; a nil top-level pointer is an error.
|
|
// - Struct fields are matched by `toml:"name"` tag (case-insensitive
|
|
// fallback to field name; `-` skips). The tag options `omitzero` (skip
|
|
// the zero value of the field's type) and `omitempty` (skip an empty
|
|
// slice, array, or map) drop a field from the output on encode; the
|
|
// decoder ignores them. Anonymous (embedded) fields without a tag are
|
|
// inlined.
|
|
// - Maps use sorted keys for deterministic output.
|
|
// - Slices and arrays of structs or maps become TOML arrays of tables; a
|
|
// nil or empty array of tables is omitted (TOML forbids an empty `[[a]]`),
|
|
// while other empty arrays emit as `key = []`.
|
|
// - Other slices and arrays become TOML arrays; a table element inside a
|
|
// value array (for example an inline table in a mixed array) emits as an
|
|
// inline table.
|
|
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
|
|
// and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/
|
|
// LocalTime (local variants). A date-time writes its seconds only when the value carries
|
|
// them, and drops the trailing zeros of a fractional second.
|
|
// - A table element of a value array, and a sub-table inlined by
|
|
// Encoder.InlineTables, is written as an inline table, across lines when it
|
|
// does not fit one.
|
|
// - Values implementing Marshaler are encoded by calling MarshalTOML and
|
|
// using its result.
|
|
// - Values implementing encoding.TextMarshaler, and not one of the
|
|
// date-time types, encode as a TOML string holding the text the method
|
|
// returns. time.Duration is written in its canonical Go form, `1h30m0s`.
|
|
// - nil pointer fields are omitted.
|
|
//
|
|
// Marshal cannot encode cyclic data structures; passing one will loop until
|
|
// the stack overflows. The output is not guaranteed to be byte-identical to
|
|
// the input that produced v: comments, whitespace, key order (for maps),
|
|
// string quoting style, and the choice between `[table]` headers and inline
|
|
// tables are not preserved.
|
|
//
|
|
// Marshal is equivalent to MarshalContext with context.Background.
|
|
func Marshal(v any) ([]byte, error) {
|
|
return MarshalContext(context.Background(), v)
|
|
}
|
|
|
|
// MarshalContext is the cancellable variant of Marshal.
|
|
func MarshalContext(ctx context.Context, v any) ([]byte, error) {
|
|
if err := ctx.Err(); err != nil {
|
|
return nil, err
|
|
}
|
|
return NewEncoder().MarshalContext(ctx, v)
|
|
}
|
|
|
|
// An Encoder encodes Go values into TOML.
|
|
//
|
|
// All options default to the behaviour that passes the toml-test compliance
|
|
// suite in both directions:
|
|
//
|
|
// GroupByKind: true (scalars first, then tables, then arrays of tables)
|
|
// OmitEmptyArrays: false (a nil/empty []string slice emits [] as a value;
|
|
// a nil/empty []Item struct slice is still skipped)
|
|
// LiteralMultilineAt: 0 (always emit the escaped basic form, never a
|
|
// literal one)
|
|
// InlineTablesAt: 0 (always emit a table header, never an inline
|
|
// table)
|
|
//
|
|
// Use the chainable option methods to opt out. The option state is private;
|
|
// callers that need the underlying knobs reach for the methods rather than
|
|
// reading or mutating fields.
|
|
type Encoder struct {
|
|
groupByKind bool // default true; set via (*Encoder).GroupByKind
|
|
omitEmptyArrays bool // default false; set via (*Encoder).OmitEmptyArrays
|
|
literalMultilineAt int // default 0; set via (*Encoder).UseLiteralMultiline
|
|
inlineTablesAt int // default 0; set via (*Encoder).InlineTables
|
|
}
|
|
|
|
// NewEncoder returns an Encoder with default options.
|
|
func NewEncoder() *Encoder { return &Encoder{groupByKind: true} }
|
|
|
|
// GroupByKind toggles whether fields at the same TOML level are reordered
|
|
// into the group-by-kind layout (scalars first, then tables, then arrays of
|
|
// tables). When set to false, the emitter preserves the source declaration
|
|
// order (struct field order, or sorted key order for maps).
|
|
func (e *Encoder) GroupByKind(v bool) *Encoder {
|
|
e.groupByKind = v
|
|
return e
|
|
}
|
|
|
|
// OmitEmptyArrays opts in to skipping empty (non-nil, length 0) TOML arrays
|
|
// of scalars. The default emits them as "key = []". Nil slices and empty
|
|
// arrays of tables are already always omitted.
|
|
func (e *Encoder) OmitEmptyArrays() *Encoder {
|
|
e.omitEmptyArrays = true
|
|
return e
|
|
}
|
|
|
|
// UseLiteralMultiline sets the length threshold at which a multi-line string
|
|
// is emitted as a literal triple-quoted string instead of the escaped form.
|
|
// Use 0 or any negative value to disable (always escaped). The literal form
|
|
// is selected only when the value contains an internal newline; otherwise the
|
|
// single-line basic form is used regardless of this setting.
|
|
func (e *Encoder) UseLiteralMultiline(threshold int) *Encoder {
|
|
e.literalMultilineAt = threshold
|
|
return e
|
|
}
|
|
|
|
// InlineTables sets the size limit, in bytes of the single-line rendering, at
|
|
// which a sub-table is written as an inline table instead of a table header,
|
|
// which makes a document of small tables shorter. Use 0 or any negative value
|
|
// to disable (always emit a header).
|
|
//
|
|
// A sub-table is inlined only when doing so keeps every value's type: an array
|
|
// of tables keeps its header form, because its inline form would re-parse as a
|
|
// value array. An inlined table that does not fit the line is written across
|
|
// lines, which TOML 1.1 allows.
|
|
//
|
|
// With GroupByKind(false) the layout is already for presentation only, and an
|
|
// inlined table follows the same rule as any other value line: it lands in the
|
|
// section of the header that precedes it.
|
|
func (e *Encoder) InlineTables(threshold int) *Encoder {
|
|
e.inlineTablesAt = threshold
|
|
return e
|
|
}
|
|
|
|
// Marshal encodes v to TOML bytes. It is equivalent to calling Marshal with v.
|
|
//
|
|
// Marshal is equivalent to MarshalContext with context.Background.
|
|
func (e *Encoder) Marshal(v any) ([]byte, error) {
|
|
return e.MarshalContext(context.Background(), v)
|
|
}
|
|
|
|
// MarshalContext is the cancellable variant of Marshal.
|
|
func (e *Encoder) MarshalContext(ctx context.Context, v any) ([]byte, error) {
|
|
enc := newEncoder()
|
|
enc.ctx = ctx
|
|
enc.opts = *e
|
|
if err := enc.encode(v); err != nil {
|
|
return nil, err
|
|
}
|
|
return enc.bytes(), nil
|
|
}
|