// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT // Package interpres is a dependency-free TOML parser for Go. // // interpres reads and writes TOML documents using only the standard library. // It exposes a small, encoding/json-style API: // // var cfg Config // err := interpres.Unmarshal(data, &cfg) // // out, err := interpres.Marshal(cfg) // // or, for the document with its key order and comments: // // doc, err := interpres.Parse(data) // tree := doc.Map() // // A Decoder allows strict decoding that rejects keys without a matching // struct field, mirroring (*json.Decoder).DisallowUnknownFields. package interpres import ( "bytes" "context" "errors" "fmt" "os" "reflect" "slices" "strings" "time" ) // A SyntaxError describes a malformed TOML document. Line is the 1-based line // the problem was detected on. Offset is the byte offset in the input the scan // stopped at, and Column is the 1-based column on that line; both are new in // 2.0 and a struct literal that names Line and Msg alone still builds. type SyntaxError struct { Line int Offset int Column int Msg string } func (e *SyntaxError) Error() string { return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg) } // SourceLine returns the source line the error points at, rendered from src, // followed by a caret line marking the column. It is meant for a message the // reader sees under the input: // // port = = 8080 // ^ // // The caret sits at Offset when it falls inside src, and at the start of the // line when the error carries no position. func (e *SyntaxError) SourceLine(src []byte) string { off := min(e.Offset, len(src)) start := 0 if i := bytes.LastIndexByte(src[:off], '\n'); i >= 0 { start = i + 1 } end := len(src) if i := bytes.IndexByte(src[start:], '\n'); i >= 0 { end = start + i } return string(src[start:end]) + "\n" + strings.Repeat(" ", off-start) + "^" } // A Path names a value in a document, one segment per level from the root: // a key contributes its name and an array element its bracketed index, so the // path of the weight field of the first item is the segments // ["items", "[0]", "weight"]. String renders the TOML notation, // "items[0].weight". type Path []string // String renders the path the way a TOML document writes it: keys join with // dots and an index attaches to the previous segment in brackets. func (p Path) String() string { var b strings.Builder for _, s := range p { if strings.HasPrefix(s, "[") { b.WriteString(s) continue } if b.Len() > 0 { b.WriteByte('.') } b.WriteString(s) } return b.String() } // A DecodeError wraps a decoding failure with the key path at which it // happened. Read the path programmatically with errors.AsType: // // if de, ok := errors.AsType[*interpres.DecodeError](err); ok { // fmt.Println(de.Path.String(), de.Err) // } type DecodeError struct { // Path is the key path from the document root, outermost key first. Path Path // Err is the failure at that path. Err error } func (e *DecodeError) Error() string { msg := strings.TrimPrefix(e.Err.Error(), "interpres: ") if p := e.Path.String(); p != "" { return "interpres: " + p + ": " + msg } return "interpres: " + msg } // Unwrap returns the failure the path points at. func (e *DecodeError) Unwrap() error { return e.Err } // newDecodeError wraps err with one path segment. The rest of the path comes // from the DecodeError err already carries, if any: the decoder wraps each // key and index on its way down, so the wrap flattens that inner error's // segments onto the front and keeps the failure it pointed at, leaving one // path and one failure to render. func newDecodeError(key string, err error) *DecodeError { path := make(Path, 0, 4) path = append(path, key) if de, ok := errors.AsType[*DecodeError](err); ok { path = append(path, de.Path...) err = de.Err } return &DecodeError{Path: path, Err: err} } // An EncodeError wraps an encoding failure with the key path of the value // that failed, in the notation of a TOML document: fields join with dots and // an array element carries its bracketed index, so the path of the third // port under server reads "server.ports[2]". The rendered message is // unchanged by the type; read it programmatically with errors.AsType. type EncodeError struct { // Path is the key path of the failing value. Path Path // Err is the failure at that path. Err error } func (e *EncodeError) Error() string { msg := strings.TrimPrefix(e.Err.Error(), "interpres: ") if p := e.Path.String(); p != "" { return "interpres: " + p + ": " + msg } return "interpres: " + msg } // Unwrap returns the failure the path points at. func (e *EncodeError) Unwrap() error { return e.Err } // Parse decodes a TOML document into a Document: the values, the order the // keys were written in, whether a table was written inline, and the comments. // ParseMap gives the plain value tree instead. // // Values are mapped to Go types as follows: strings to string, integers to // int64, floats to float64, booleans to bool, offset date-times to // OffsetDateTime, the local date-time kinds to their wrappers, arrays to // []any, and tables (including inline tables) to map[string]any. // // Parse is equivalent to ParseContext with context.Background. func Parse(data []byte) (*Document, error) { return ParseContext(context.Background(), data) } // ParseContext decodes a TOML document into a Document, obeying ctx. The // context is checked between top-level statements so cancellation is honoured // before the parser has done substantial work. func ParseContext(ctx context.Context, data []byte) (*Document, error) { _, doc, err := parseWithOptions(ctx, data, parseOptions{}, true) return doc, err } // ParseMap decodes a TOML document into a nested map[string]any, the value // tree without the order and the comments a Document carries. It is the shape // this package parsed into before [Document] existed. // // ParseMap is equivalent to ParseMapContext with context.Background. func ParseMap(data []byte) (map[string]any, error) { return ParseMapContext(context.Background(), data) } // ParseMapContext is the cancellable variant of ParseMap. func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error) { tree, _, err := parseWithOptions(ctx, data, parseOptions{}, false) return tree, err } // ParseFile reads the TOML document at path and parses it into a Document, // the shape Parse gives. Every error names the file it came from: a read // failure and a parse failure alike carry the path as their first words, // wrapped so errors.AsType still reaches the SyntaxError inside. func ParseFile(path string) (*Document, error) { data, err := os.ReadFile(path) if err != nil { return nil, fmt.Errorf("%s: %w", path, err) } doc, err := Parse(data) if err != nil { return nil, fmt.Errorf("%s: %w", path, err) } return doc, nil } // Valid reports whether data is a valid TOML document: nil when the parser // accepts it, and the parse error when it does not. It is the library call // the -validate mode of interpres-decode is built on, and it reads nothing // but the bytes it is given. func Valid(data []byte) error { _, err := ParseMapContext(context.Background(), data) return err } // parseOptions bound the work one parse may do and the shape it produces. A // zero field takes the default. type parseOptions struct { maxDepth int maxInputSize int useNumber bool } // parseWithOptions parses data, building the node tree of a Document when // wantDoc asks for it, and returns both the value tree and that document. func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantDoc bool) (map[string]any, *Document, error) { if err := ctx.Err(); err != nil { return nil, nil, err } if opts.maxInputSize > 0 && len(data) > opts.maxInputSize { return nil, nil, fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), opts.maxInputSize) } // UTF-8 validity is not checked in a pass of its own: the scanner // validates the multi-byte sequences where it meets them, so an invalid // byte is reported on its own line instead of always on line 1. maxDepth := opts.maxDepth if maxDepth <= 0 { maxDepth = maxNestingDepth } // The parser scans data in place; it only reads the buffer, and every // string it stores in the tree is copied out of it. p := &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, wantDoc: wantDoc, useNumber: opts.useNumber} tree, err := p.parse() if err != nil { return nil, nil, err } if !wantDoc { return tree, nil, nil } return tree, &Document{root: p.doc, footer: p.footer}, nil } // Unmarshal parses a TOML document and stores the result in the value pointed // to by v. v is typically a pointer to a struct or to a map[string]any. // // Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a // case-insensitive match on the field name when no tag is present. A tag of // "-" skips the field. // // A destination implementing Unmarshaler receives the parsed value as it is, // a TOML string fills a destination implementing encoding.TextUnmarshaler, and // a time.Duration destination takes a duration literal such as `1h30m` or a // bare integer as its nanosecond count. // // Unmarshal is equivalent to UnmarshalContext with context.Background. func Unmarshal(data []byte, v any) error { return UnmarshalContext(context.Background(), data, v) } // ParseAs decodes a TOML document into T in one call, the generic shorthand // for Unmarshal with a destination variable: // // cfg, err := interpres.ParseAs[Config](data) // // The zero T comes back with the error. func ParseAs[T any](data []byte) (T, error) { var v T err := Unmarshal(data, &v) return v, err } // NewSchema precompiles the codec for T: the struct schema both directions // walk and the interface flags the decoder and the encoder resolve through // are built once and cached, so the first document pays the cost instead of // the hot path. A T that is not a struct warms nothing; there is nothing to // precompute for a map or a slice. func NewSchema[T any]() { t := reflect.TypeFor[T]() if t.Kind() != reflect.Struct { return } cachedStructSchema(t) _ = typeFlags(t) _ = encTypeFlags(t) pt := reflect.PointerTo(t) _ = typeFlags(pt) _ = encTypeFlags(pt) } // UnmarshalContext is the cancellable variant of Unmarshal. func UnmarshalContext(ctx context.Context, data []byte, v any) error { // Only a destination that can reach an OrderedMap needs the node tree the // written key order is read from; every other decode skips building it. tree, doc, err := parseWithOptions(ctx, data, parseOptions{}, typeWantsOrder(reflect.TypeOf(v))) if err != nil { return err } dec := newDecoder() dec.ctx = ctx dec.nodes = indexNodes(doc.Root()) return dec.decode(tree, v) } // A Decoder decodes a TOML document into a Go value with configurable // strictness and configurable limits on the parse it performs. type Decoder struct { disallowUnknown bool useNumber bool maxDepth int maxInputSize int localLoc *time.Location } // NewDecoder returns a Decoder. func NewDecoder() *Decoder { return &Decoder{} } // DisallowUnknownFields causes Decode to return an error when the document // contains a key with no matching destination struct field. func (d *Decoder) DisallowUnknownFields() *Decoder { d.disallowUnknown = true return d } // UseNumber causes the numbers of the document to reach the value tree as a // Number carrying the literal the document wrote, so 0x1f, 1_000, +1.0 and // inf survive a round trip with their spelling intact. A destination of a // concrete numeric kind still takes the evaluated value; the literal is kept // only where a Number, or an any, receives it. func (d *Decoder) UseNumber() *Decoder { d.useNumber = true return d } // LocalTimeLocation sets the zone a local date-time is placed in when it // decodes into a time.Time destination. Without the option a local date-time // fills only its own wrapper type (LocalDateTime, LocalDate, LocalTime), // whose embedded time.Time is UTC; with the option, a time.Time destination // takes the value too, carried in the location given. A nil location restores // the default. func (d *Decoder) LocalTimeLocation(loc *time.Location) *Decoder { d.localLoc = loc return d } // MaxDepth bounds how deeply arrays and inline tables may nest in a document // this decoder accepts. The parser is a recursive descent, so a document that // nests without bound would exhaust the stack; one that nests deeper than the // limit is rejected with a SyntaxError naming it instead. Use 0 or any // negative value for the default of 10000, which no hand-written document // approaches. func (d *Decoder) MaxDepth(depth int) *Decoder { d.maxDepth = depth return d } // MaxInputSize bounds the size of a document this decoder accepts, in bytes; a // larger one is rejected before parsing starts. Use 0 or any negative value for // no limit, which is the default: the caller already holds the bytes, so the // size is a policy the caller sets rather than a protection the library // imposes on its own. Parse and ParseContext take no limit beyond the nesting // default. func (d *Decoder) MaxInputSize(size int) *Decoder { d.maxInputSize = size return d } // Decode parses data and stores the result in the value pointed to by v, // honouring the decoder's strictness settings. // // Decode is equivalent to DecodeContext with context.Background. func (d *Decoder) Decode(data []byte, v any) error { return d.DecodeContext(context.Background(), data, v) } // DecodeContext is the cancellable variant of Decode. func (d *Decoder) DecodeContext(ctx context.Context, data []byte, v any) error { opts := parseOptions{ maxDepth: d.maxDepth, maxInputSize: d.maxInputSize, useNumber: d.useNumber, } tree, doc, err := parseWithOptions(ctx, data, opts, typeWantsOrder(reflect.TypeOf(v))) if err != nil { return err } dec := newDecoder() dec.disallowUnknown = d.disallowUnknown dec.ctx = ctx dec.nodes = indexNodes(doc.Root()) dec.loc = d.localLoc return dec.decode(tree, v) } // DecodeOptions gathers the options a one-shot decode call can set, the // struct-shaped alternative to building a Decoder for a single document. The // zero value decodes with the defaults: unknown keys ignored, numbers // evaluated, and no limit beyond the nesting default. type DecodeOptions struct { // DisallowUnknownFields rejects a key with no matching struct field. DisallowUnknownFields bool // UseNumber keeps the numbers of the document as Number literals. UseNumber bool // MaxDepth bounds how deeply arrays and inline tables may nest; 0 takes // the default of 10000. MaxDepth int // MaxInputSize bounds the document size in bytes; 0 takes no limit. MaxInputSize int } // UnmarshalWithOptions decodes data into v with the options set, the one-shot // form of building a Decoder. See DecodeOptions for the fields and their // defaults. func UnmarshalWithOptions(data []byte, v any, opts DecodeOptions) error { dec := &Decoder{ disallowUnknown: opts.DisallowUnknownFields, useNumber: opts.UseNumber, maxDepth: opts.MaxDepth, maxInputSize: opts.MaxInputSize, } return dec.DecodeContext(context.Background(), data, v) } // Marshaler is the interface implemented by types that can produce a custom // TOML representation of themselves. MarshalTOML returns a value that Marshal // then encodes as if the returned value had been passed in its place, which // is useful for emitting a Go type as a different TOML shape (for example, a // struct as an inline table or a primitive alias as a richer value). // // MarshalTOML wins over encoding.TextMarshaler when a type implements both. // A type that implements only encoding.TextMarshaler is encoded as a TOML // string holding its text, and needs no method here. type Marshaler interface { MarshalTOML() (any, error) } // Unmarshaler is the inverse of Marshaler: a type that wants control over // how it is decoded from a TOML value may implement UnmarshalTOML. The data // argument is whatever the parser produced for that key: one of string, // bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime, // []any, or map[string]any. A tree built by hand may carry a plain time.Time // where the parser would put an OffsetDateTime, and a Decoder configured with // UseNumber a Number. // // UnmarshalTOML may parse, inspect, or transform the value however it likes, // then store the result by mutating its receiver through the standard // pointer-indirection rules of the reflect package (i.e. via // reflect.Value.Set or by reassigning fields through a pointer the receiver // holds). // // UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the // destination type implements the interface. The decoder does not need to // consult the concrete return value; whatever the receiver stores is kept. // // UnmarshalTOML wins over encoding.TextUnmarshaler when a type implements // both. A type that implements only encoding.TextUnmarshaler is filled from a // TOML string holding its text, and needs no method here. type Unmarshaler interface { UnmarshalTOML(data any) error } // UnmarshalerContext is Unmarshaler with the decode's context handed in. A // type that implements both interfaces gets UnmarshalTOMLContext, so a long // custom decode can abort on cancellation instead of running to completion. // The context a non-cancellable entry point carries is context.Background, // never nil. type UnmarshalerContext interface { UnmarshalTOMLContext(ctx context.Context, data any) error } // Marshal returns the TOML encoding of v. The output is valid TOML 1.1. // // Marshal traverses v using reflection and applies the following rules: // // - The top-level value must be a struct or a map[string]V. Pointers are // followed; a nil top-level pointer is an error. // - Struct fields are matched by `toml:"name"` tag (case-insensitive // fallback to field name; `-` skips). The tag options `omitzero` (skip // the zero value of the field's type) and `omitempty` (skip an empty // slice, array, or map) drop a field from the output on encode; the // decoder ignores them. Anonymous (embedded) fields without a tag are // inlined. // - Maps use sorted keys for deterministic output. // - Slices and arrays of structs or maps become TOML arrays of tables; a // nil or empty array of tables is omitted (TOML forbids an empty `[[a]]`), // while other empty arrays emit as `key = []`. // - Other slices and arrays become TOML arrays; a table element inside a // value array (for example an inline table in a mixed array) emits as an // inline table. // - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time // and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/ // LocalTime (local variants). A date-time writes its seconds only when the value carries // them, and drops the trailing zeros of a fractional second. // - A table element of a value array, and a sub-table inlined by // Encoder.InlineTables, is written as an inline table, across lines when it // does not fit one. // - Values implementing Marshaler are encoded by calling MarshalTOML and // using its result. // - Values implementing encoding.TextMarshaler, and not one of the // date-time types, encode as a TOML string holding the text the method // returns. time.Duration is written in its canonical Go form, `1h30m0s`. // - nil pointer fields are omitted. // // Marshal rejects a value that nests deeper than 10000 levels with an error // naming the limit, so cyclic data is reported instead of running the stack // out. The output is not guaranteed to be byte-identical to // the input that produced v: comments, whitespace, key order (for maps), // string quoting style, and the choice between `[table]` headers and inline // tables are not preserved. // // Marshal is equivalent to MarshalContext with context.Background. func Marshal(v any) ([]byte, error) { return MarshalContext(context.Background(), v) } // MarshalAppend appends the TOML encoding of v to buf and returns the extended // buffer, the shape json.MarshalAppend has. A failed encoding leaves buf // untouched and comes back with a nil slice. func MarshalAppend(buf []byte, v any) ([]byte, error) { out, err := Marshal(v) if err != nil { return nil, err } return append(buf, out...), nil } // MarshalContext is the cancellable variant of Marshal. func MarshalContext(ctx context.Context, v any) ([]byte, error) { if err := ctx.Err(); err != nil { return nil, err } return NewEncoder().MarshalContext(ctx, v) } // An Encoder encodes Go values into TOML. // // All options default to the behaviour that passes the toml-test compliance // suite in both directions: // // GroupByKind: true (scalars first, then tables, then arrays of tables) // OmitEmptyArrays: false (a nil/empty []string slice emits [] as a value; // a nil/empty []Item struct slice is still skipped) // LiteralMultilineAt: 0 (always emit the escaped basic form, never a // literal one) // InlineTablesAt: 0 (always emit a table header, never an inline // table) // // Use the chainable option methods to opt out. The option state is private; // callers that need the underlying knobs reach for the methods rather than // reading or mutating fields. type Encoder struct { groupByKind bool // default true; set via (*Encoder).GroupByKind omitEmptyArrays bool // default false; set via (*Encoder).OmitEmptyArrays literalMultilineAt int // default 0; set via (*Encoder).UseLiteralMultiline inlineTablesAt int // default 0; set via (*Encoder).InlineTables emitFieldComments bool // default false; set via (*Encoder).EmitFieldComments } // NewEncoder returns an Encoder with default options. func NewEncoder() *Encoder { return &Encoder{groupByKind: true} } // GroupByKind toggles whether fields at the same TOML level are reordered // into the group-by-kind layout (scalars first, then tables, then arrays of // tables). When set to false, the emitter preserves the source declaration // order (struct field order, or sorted key order for maps). func (e *Encoder) GroupByKind(v bool) *Encoder { e.groupByKind = v return e } // OmitEmptyArrays opts in to skipping empty (non-nil, length 0) TOML arrays // of scalars. The default emits them as "key = []". Nil slices and empty // arrays of tables are already always omitted. func (e *Encoder) OmitEmptyArrays() *Encoder { e.omitEmptyArrays = true return e } // UseLiteralMultiline sets the length threshold at which a multi-line string // is emitted as a literal triple-quoted string instead of the escaped form. // Use 0 or any negative value to disable (always escaped). The literal form // is selected only when the value contains an internal newline; otherwise the // single-line basic form is used regardless of this setting. func (e *Encoder) UseLiteralMultiline(threshold int) *Encoder { e.literalMultilineAt = threshold return e } // InlineTables sets the size limit, in bytes of the single-line rendering, at // which a sub-table is written as an inline table instead of a table header, // which makes a document of small tables shorter. Use 0 or any negative value // to disable (always emit a header). // // A sub-table is inlined only when doing so keeps every value's type: an array // of tables keeps its header form, because its inline form would re-parse as a // value array. An inlined table that does not fit the line is written across // lines, which TOML 1.1 allows. // // With GroupByKind(false) the layout is already for presentation only, and an // inlined table follows the same rule as any other value line: it lands in the // section of the header that precedes it. func (e *Encoder) InlineTables(threshold int) *Encoder { e.inlineTablesAt = threshold return e } // EmitFieldComments turns on printing the comment a field's `toml` tag // carries in a `comment=` option, above the field's line or header, the // comments a round trip through the Go type would otherwise drop: // // Port int `toml:"port,comment=The port to listen on"` // // Go doc comments are not visible to reflection, so the tag is the channel // that carries the text. Off by default, and a field without a `comment=` // option prints none. Multi-line comments carry newlines in the tag, each // line printed with its own "# " marker. func (e *Encoder) EmitFieldComments() *Encoder { e.emitFieldComments = true return e } // Marshal encodes v to TOML bytes. It is equivalent to calling Marshal with v. // // Marshal is equivalent to MarshalContext with context.Background. func (e *Encoder) Marshal(v any) ([]byte, error) { return e.MarshalContext(context.Background(), v) } // MarshalContext is the cancellable variant of Marshal. func (e *Encoder) MarshalContext(ctx context.Context, v any) ([]byte, error) { enc := newEncoder() enc.ctx = ctx enc.opts = *e if err := enc.encode(v); err != nil { enc.release() return nil, err } // The output leaves the pooled buffer as a copy, so the next Marshal // reuses the buffer without touching what the caller holds. out := slices.Clone(enc.buf.Bytes()) enc.release() return out, nil }