diff --git a/api_test.go b/api_test.go new file mode 100644 index 0000000..0830962 --- /dev/null +++ b/api_test.go @@ -0,0 +1,89 @@ +// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) +// SPDX-License-Identifier: MIT + +package interpres + +import ( + "errors" + "strings" + "testing" +) + +// errReader fails every read with a fixed error. +type errReader struct{ err error } + +func (r errReader) Read([]byte) (int, error) { return 0, r.err } + +// TestUnmarshalRead covers the streaming entry: the happy path with options, +// a failing reader, and MaxInputSize bounding what a reader is drained into. +func TestUnmarshalRead(t *testing.T) { + var got struct { + Name string `toml:"name"` + N int `toml:"n"` + } + err := UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, RejectUnknownFields(true)) + if err != nil { + t.Fatalf("UnmarshalRead: %v", err) + } + if got.Name != "x" { + t.Errorf("Name = %q", got.Name) + } + + readErr := errors.New("boom") + if err := UnmarshalRead(errReader{readErr}, &got); !errors.Is(err, readErr) { + t.Errorf("err = %v, want the read error wrapped", err) + } + + err = UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, MaxInputSize(4)) + if err == nil || !strings.Contains(err.Error(), "over the limit") { + t.Errorf("err = %v, want the size limit", err) + } + // The limit bounds the read itself: a reader that would supply far more + // than the limit is not drained into memory first. + big := strings.Repeat("x", 1<<20) + if err := UnmarshalRead(strings.NewReader(big), &got, MaxInputSize(16)); err == nil || !strings.Contains(err.Error(), "over the limit") { + t.Errorf("err = %v, want the size limit before the read completes", err) + } +} + +// TestParseAsWithOptions covers the generic shorthand carrying options. +func TestParseAsWithOptions(t *testing.T) { + type cfg struct { + Name string `toml:"name"` + } + got, err := ParseAs[cfg]([]byte("name = \"x\"\nrogue = 1\n"), RejectUnknownFields(true)) + if err == nil || !strings.Contains(err.Error(), "unknown field") { + t.Errorf("err = %v, want the strict failure", err) + } + // The statements before the failure stay written, the contract the + // targeted path documents and encoding/json follows. + if got.Name != "x" { + t.Errorf("Name = %q, want the statement before the failure kept", got.Name) + } +} + +// TestStatementsValueArrays pins that a value array is one statement, a +// scalar array and an array of inline tables alike; only an array of tables +// yields per element. +func TestStatementsValueArrays(t *testing.T) { + src := strings.NewReader("port = [8080, 9090]\nmix = [{y = 1, x = 2}]\n[[items]]\nn = 1\n") + var got []Statement + for stmt, err := range Statements(src) { + if err != nil { + t.Fatal(err) + } + got = append(got, stmt) + } + if len(got) != 3 { + t.Fatalf("got %d statements, want 3", len(got)) + } + if got[0].Index != -1 || got[0].Table != nil { + t.Errorf("port statement = %+v, want one plain key/value", got[0]) + } + if got[1].Index != -1 || got[1].Table != nil { + t.Errorf("mix statement = %+v, want one plain key/value", got[1]) + } + if got[2].Index != 0 || got[2].Table == nil { + t.Errorf("items statement = %+v, want the element with its node", got[2]) + } +} diff --git a/interpres.go b/interpres.go index 9bfaadc..346685f 100644 --- a/interpres.go +++ b/interpres.go @@ -16,8 +16,10 @@ // doc, err := interpres.Parse(data) // tree := doc.Map() // -// A Decoder allows strict decoding that rejects keys without a matching -// struct field, mirroring the RejectUnknownFields option of encoding/json/v2. +// Strict decoding that rejects keys without a matching struct field is an +// option, mirroring the RejectUnknownMembers option of encoding/json/v2: +// +// err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true)) package interpres import ( @@ -212,7 +214,7 @@ func ParseFile(path string) (*Document, error) { // Valid reports whether data is a valid TOML document: nil when the parser // accepts it, and the parse error when it does not. It is the library call -// the -validate mode of interpres-decode is built on, and it reads nothing +// the --validate mode of interpres-decode is built on, and it reads nothing // but the bytes it is given. func Valid(data []byte) error { _, err := ParseMapContext(context.Background(), data) @@ -256,9 +258,6 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD return tree, &Document{root: p.doc, footer: p.footer}, nil } -// Unmarshal parses a TOML document and stores the result in the value pointed -// to by v. v is typically a pointer to a struct or to a map[string]any. -// // Unmarshal parses a TOML document and stores the result in the value pointed // to by v. v is typically a pointer to a struct or to a map[string]any. // Options tune the call; with none, unknown keys are ignored, numbers are @@ -266,7 +265,9 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD // // Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a // case-insensitive match on the field name when no tag is present. A tag of -// "-" skips the field. +// "-" skips the field. Two document keys that differ only in case and both +// match one field resolve deterministically: the lexicographically greater +// one wins, the same key winning every run. // // A destination implementing Unmarshaler receives the parsed value as it is, // a TOML string fills a destination implementing encoding.TextUnmarshaler, and @@ -281,12 +282,12 @@ func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error { // ParseAs decodes a TOML document into T in one call, the generic shorthand // for Unmarshal with a destination variable: // -// cfg, err := interpres.ParseAs[Config](data) +// cfg, err := interpres.ParseAs[Config](data, interpres.RejectUnknownFields(true)) // -// The zero T comes back with the error. -func ParseAs[T any](data []byte) (T, error) { +// The options are Unmarshal's. The zero T comes back with the error. +func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error) { var v T - err := Unmarshal(data, &v) + err := Unmarshal(data, &v, opts...) return v, err } @@ -315,10 +316,19 @@ func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...Unmarshal // UnmarshalRead reads the document from r and decodes it into v, the // streaming-shaped entry the json/v2 vocabulary uses. The reader is -// consumed in full, because the parser scans its source in place; the -// options and the behaviour are Unmarshal's. +// consumed in full, because the parser scans its source in place; with +// MaxInputSize set, reading stops one byte past the limit so the size the +// option bounds is the memory held, not what a reader is drained into first. +// The options and the behaviour are Unmarshal's. func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error { - data, err := io.ReadAll(r) + s := settingsFor(opts) + var data []byte + var err error + if s.maxInputSize > 0 { + data, err = io.ReadAll(io.LimitReader(r, int64(s.maxInputSize)+1)) + } else { + data, err = io.ReadAll(r) + } if err != nil { return fmt.Errorf("interpres: read: %w", err) } @@ -338,18 +348,19 @@ func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error { // tree path, so every option means the same thing on every document. type UnmarshalOption func(*decodeSettings) -// decodeSettings is the option carrier of one decode call. +// decodeSettings is the option carrier of one decode call. The context is +// not one: it arrives as its own argument, because every entry point names it +// explicitly. type decodeSettings struct { disallowUnknown bool useNumber bool maxDepth int maxInputSize int localLoc *time.Location - ctx context.Context } func settingsFor(opts []UnmarshalOption) *decodeSettings { - s := &decodeSettings{ctx: context.Background()} + s := &decodeSettings{} for _, opt := range opts { opt(s) } @@ -452,8 +463,8 @@ type Marshaler interface { // argument is whatever the parser produced for that key: one of string, // bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime, // []any, or map[string]any. A tree built by hand may carry a plain time.Time -// where the parser would put an OffsetDateTime, and a Decoder configured with -// UseNumber a Number. +// where the parser would put an OffsetDateTime, and NumbersAsLiterals a +// Number. // // UnmarshalTOML may parse, inspect, or transform the value however it likes, // then store the result by mutating its receiver through the standard @@ -461,7 +472,7 @@ type Marshaler interface { // reflect.Value.Set or by reassigning fields through a pointer the receiver // holds). // -// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the +// UnmarshalTOML is invoked from Unmarshal and its siblings when the // destination type implements the interface. The decoder does not need to // consult the concrete return value; whatever the receiver stores is kept. // @@ -482,16 +493,21 @@ type UnmarshalerContext interface { } // Marshal returns the TOML encoding of v. The output is valid TOML 1.1. +// Options tune the emission; with none, the layout groups entries by kind, +// empty arrays emit and sub-tables take the header form. // // Marshal traverses v using reflection and applies the following rules: // -// - The top-level value must be a struct or a map[string]V. Pointers are -// followed; a nil top-level pointer is an error. +// - The top-level value must be a struct, a map[string]V or an OrderedMap +// (or a non-nil pointer to one). A Document writes itself back, and a nil +// one is an error. // - Struct fields are matched by `toml:"name"` tag (case-insensitive -// fallback to field name; `-` skips). The tag options `omitzero` (skip -// the zero value of the field's type) and `omitempty` (skip an empty -// slice, array, or map) drop a field from the output on encode; the -// decoder ignores them. Anonymous (embedded) fields without a tag are +// fallback to field name; `-` skips). The tag option `omitzero` skips a +// field holding the zero value of its type (a type with an IsZero method +// decides through it), and `omitempty` skips a value that is empty in +// the encoding/json sense: an empty string, a zero number, false, a nil +// pointer or interface, and an empty slice, array or map. The decoder +// ignores both options. Anonymous (embedded) fields without a tag are // inlined. // - Maps use sorted keys for deterministic output. // - Slices and arrays of structs or maps become TOML arrays of tables; a @@ -502,10 +518,12 @@ type UnmarshalerContext interface { // inline table. // - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time // and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/ -// LocalTime (local variants). A date-time writes its seconds only when the value carries -// them, and drops the trailing zeros of a fractional second. -// - A table element of a value array, and a sub-table inlined by -// Encoder.InlineTables, is written as an inline table, across lines when it +// LocalTime (local variants). A date-time writes its seconds only when +// the value carries them, and drops the trailing zeros of a fractional +// second. A zone offset that is not a whole number of minutes is refused, +// because TOML has no form that carries its seconds. +// - A table element of a value array, and a sub-table the InlineTables +// option inlines, is written as an inline table, across lines when it // does not fit one. // - Values implementing Marshaler are encoded by calling MarshalTOML and // using its result. @@ -516,14 +534,10 @@ type UnmarshalerContext interface { // // Marshal rejects a value that nests deeper than 10000 levels with an error // naming the limit, so cyclic data is reported instead of running the stack -// out. The output is not guaranteed to be byte-identical to -// the input that produced v: comments, whitespace, key order (for maps), -// string quoting style, and the choice between `[table]` headers and inline -// tables are not preserved. -// -// Marshal returns the TOML encoding of v. Options tune the emission; with -// none, the layout groups entries by kind, empty arrays emit and sub-tables -// take the header form. +// out. The output is not guaranteed to be byte-identical to the input that +// produced v: comments, whitespace, key order (for maps), string quoting +// style, and the choice between `[table]` headers and inline tables are not +// preserved. // // Marshal is equivalent to MarshalContext with context.Background. func Marshal(v any, opts ...MarshalOption) ([]byte, error) { @@ -548,12 +562,13 @@ type Statement struct { } // Statements reads a TOML document from r and returns an iterator over its -// top-level statements in written order: key/value statements, a [table] -// header as one statement carrying its Table node, and an [[array of -// tables]] as one statement per element, each with the element's node and -// its Index. Iteration stops at the first error, which arrives as the second -// value, and at a false yield: a caller that breaks after the statement it -// wanted reads no further ones. +// top-level statements in written order: key/value statements, including a +// value that is an array or an inline table, a [table] header as one +// statement carrying its Table node, and an [[array of tables]] as one +// statement per element, each with the element's node and its Index. +// Iteration stops at the first error, which arrives as the second value, and +// at a false yield: a caller that breaks after the statement it wanted reads +// no further ones. // // The reader is consumed in full before the first statement is yielded, // because the parser scans the source in place; processing the yielded @@ -572,8 +587,12 @@ func Statements(r io.Reader) iter.Seq2[Statement, error] { return } for _, e := range doc.Root().Entries() { - if els := e.Elements(); len(els) > 0 { - for i, el := range els { + // Only an array of tables yields per element, the branch the + // write side takes too: a value array is one statement whatever + // its elements, and an emptied array of tables holds no element + // to yield. + if _, isTables := e.Value().([]map[string]any); isTables && len(e.Elements()) > 0 { + for i, el := range e.Elements() { if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: el, Index: i}, nil) { return } @@ -648,14 +667,14 @@ const ( // interpres.InlineTables(60)) type MarshalOption func(*encodeSettings) -// encodeSettings is the option carrier of one encode call. +// encodeSettings is the option carrier of one encode call. As on the decode +// side, the context arrives as its own argument. type encodeSettings struct { - ctx context.Context cfg encodeConfig } func settingsForEncode(opts []MarshalOption) *encodeSettings { - s := &encodeSettings{ctx: context.Background(), cfg: encodeConfig{layout: LayoutKindGrouped}} + s := &encodeSettings{cfg: encodeConfig{layout: LayoutKindGrouped}} for _, opt := range opts { opt(s) } @@ -681,6 +700,7 @@ func (s *encodeSettings) marshal(ctx context.Context, v any) ([]byte, error) { // Layout sets the layout the encoder writes a document's entries in: // LayoutKindGrouped, the default, reorders them scalars first, then tables, // then arrays of tables; LayoutKindDeclaration preserves declaration order. +// A value the two constants do not name behaves as LayoutKindGrouped. func Layout(kind LayoutKind) MarshalOption { return func(s *encodeSettings) { s.cfg.layout = kind } } @@ -727,7 +747,9 @@ func InlineTables(threshold int) MarshalOption { // Go doc comments are not visible to reflection, so the tag is the channel // that carries the text. Off by default, and a field without a `comment=` // option prints none. Multi-line comments carry newlines in the tag, each -// line printed with its own "# " marker. +// line printed with its own "# " marker. The tag's options separate with +// commas, so the comment text itself cannot carry one; the first comma ends +// it. func EmitFieldComments(v bool) MarshalOption { return func(s *encodeSettings) { s.cfg.emitFieldComments = v } }