12 Commits
Author SHA1 Message Date
petrbalvin eb6ac1ab9d ci: drop the schedule triggers, race and fuzz run on dispatch alone
Test / test (push) Successful in 1m54s
Assisted-by: GLM 5.3
2026-09-22 21:28:30 +02:00
petrbalvin b0f40739c7 chore: keep the coverage floor at the documented 80 percent
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin fba53440c7 ci: state where the nightly race and fuzz sweeps run
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 332cd01d44 chore: check the suite colour before comparing documented counts
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 2fa075de00 docs: align every document with the reviewed behaviour
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 4900367970 fix(cmd): long-form flags, honest counts and safer inference
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin b7f39435e1 fix(document): rebuild set nodes and write documents back round-trip
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 0ded34da3c fix(encode): pointer table arrays, whole-minute offsets and emission checks
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin b45f4d65da fix(decode): keep the targeted parse on the tree path's contract
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin f7e427ae3f fix(decode): allocate embedded pointer maps and settle case collisions
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin e677e34508 fix(api): one statement per value array, parseas options and bounded reads
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 2efdb2d059 fix(parse): reject the lenient grammar edges and name out-of-range date-times
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
34 changed files with 2753 additions and 775 deletions
+5 -9
View File
@@ -1,20 +1,16 @@
# Fuzz smoke, Go. A nightly time-boxed run of the fuzz targets, and a hand # Fuzz smoke, Go. Dispatched by hand when a change asks for it.
# dispatch when a change asks for it.
# #
# Fuzzing is exploration, so it never belongs to the push pipeline; a 30 second # Fuzzing is exploration, so it never belongs to the push pipeline; a 30 second
# smoke per target, run nightly, is the compromise that catches a new crash # smoke per target on a hand dispatch checks a change without holding the
# within a day without holding the shared box. The targets run the seeds and # shared box. The targets run the seeds and whatever the corpus has gathered; a
# whatever the corpus has gathered; a failure leaves its crashing input in # failure leaves its crashing input in testdata/fuzz, which the ordinary suite
# testdata/fuzz, which the ordinary suite then reproduces on every push. # then reproduces on every push.
# #
# Every step is one command, so the step that fails is the gate that failed. # Every step is one command, so the step that fails is the gate that failed.
name: Fuzz name: Fuzz
on: on:
workflow_dispatch: workflow_dispatch:
schedule:
# Nightly at 03:30 UTC, after the race sweep has had the box first.
- cron: "30 3 * * *"
env: env:
# One core: parallelism buys no speed here and costs memory the box does not have. # One core: parallelism buys no speed here and costs memory the box does not have.
+5 -9
View File
@@ -1,20 +1,16 @@
# Race, Go. Dispatched by hand, run nightly on a schedule, and run as part of # Race, Go. Dispatched by hand.
# the release gates.
# #
# The race detector roughly doubles both time and memory, which the shared runner box # The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per # cannot afford on every push. Locally it belongs to `just gates`, which runs it once
# task; here it is an explicit decision rather than a routine. The schedule is the # per task; here it is an explicit decision rather than a routine, a hand dispatch
# nightly sweep: a failing race on development is known by morning without anyone # when a change asks for one. Development carries its race gate on every push through
# remembering to dispatch it. # that local gate.
# #
# Every step is one command, so the step that fails is the gate that failed. # Every step is one command, so the step that fails is the gate that failed.
name: Race name: Race
on: on:
workflow_dispatch: workflow_dispatch:
schedule:
# Nightly at 03:00 UTC, the quietest hours of the shared box.
- cron: "0 3 * * *"
env: env:
# One core: parallelism buys no speed here and costs memory the box does not have. # One core: parallelism buys no speed here and costs memory the box does not have.
+3 -2
View File
@@ -1,8 +1,9 @@
# Test, Go. Push and pull request to development. Never on main. # Test, Go. Push and pull request to development. Never on main.
# #
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared # The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so race runs once inside # runner box cannot afford the race detector on every push. Race has its own
# the release pipeline instead. The box is one core and 2 GB beside Gitea, so # pipeline, dispatched by hand, and the local `just gates` runs it once per
# task. The box is one core and 2 GB beside Gitea, so
# parallelism is bounded on purpose and everything runs in one job. Extra jobs would # parallelism is bounded on purpose and everything runs in one job. Extra jobs would
# duplicate the checkout, the Go setup and the dependency download three times without # duplicate the checkout, the Go setup and the dependency download three times without
# buying any parallelism. # buying any parallelism.
+1 -1
View File
@@ -1,7 +1,7 @@
.idea/ .idea/
.zcode/ .zcode/
# Build artifacts # Build artefacts
bin/ bin/
*.test *.test
*.out *.out
+16 -1
View File
@@ -155,7 +155,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
interface, or a nil or empty slice, array or map. In 1.x the option covered interface, or a nil or empty slice, array or map. In 1.x the option covered
only the collections. only the collections.
- The `toml` tag gained the `inline` option: a struct or map field tagged - The `toml` tag gained the `inline` option: a struct or map field tagged
`toml:"retry,inline"` emits as `retry = {…}` instead of a header section, `toml:"retry,inline"` emits as `retry = {…}` instead of a header section,
whatever its size, a named embedded struct included. Forcing it on an array whatever its size, a named embedded struct included. Forcing it on an array
of tables is an error, because the inline form would re-parse as a value of tables is an error, because the inline form would re-parse as a value
array and change the value's Go type. array and change the value's Go type.
@@ -239,6 +239,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- A top-level value the encoder could not normalise reported its path with - A top-level value the encoder could not normalise reported its path with
a leading dot, `interpres: .port: ...`; the message now reads a leading dot, `interpres: .port: ...`; the message now reads
`interpres: port: ...`, the shape `EncodeError.Path` already used. `interpres: port: ...`, the shape `EncodeError.Path` already used.
- A token shaped like a date-time with a component out of range, such as an
hour of 24 or a February the 30th, fell through to the number decoder and
failed with the number complaint `invalid character "-" in number`; it now
fails as the date-time it visibly is, `invalid date-time "..."`.
- A `time.Time` or `OffsetDateTime` whose zone offset is not a whole number
of minutes wrote only the minutes, silently shifting the instant by the
seconds dropped; the encoder now refuses such an offset, which TOML has no
form for, instead of corrupting the value.
- An empty array of tables over pointer elements, `[]*T{}`, emitted as
`key = []` while its value form was omitted; it is omitted too now, the
rule TOML forces, because an empty `[[a]]` has no valid form.
- Two lenient grammar edges are closed: a sign in a `\u` or `\U` escape,
which is not a hex digit, is rejected instead of evaluating, and a bare
carriage return right after a multi-line string's opening delimiter is
the bare-CR error instead of a newline trimmed silently.
### Migration from 1.x ### Migration from 1.x
+1
View File
@@ -116,6 +116,7 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
|---|---|---| |---|---|---|
| Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite | | Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite |
| Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs | | Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs |
| Fuzz | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector | | Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector |
The local equivalent is `just gates`, which is the same set plus the race The local equivalent is `just gates`, which is the same set plus the race
+7 -5
View File
@@ -13,16 +13,17 @@ the entire official [toml-test](https://github.com/toml-lang/toml-test) suite:
`\xHH` escapes; integers in the four radixes with `_` separators; floats with `\xHH` escapes; integers in the four radixes with `_` separators; floats with
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds
optional as of 1.1; arrays and inline tables, multi-line as of 1.1. optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and - **Decoding and encoding**: `Unmarshal` and `Marshal` for structs and maps,
`Marshal` for structs and maps, mirroring `encoding/json`. mirroring `encoding/json`; `Parse` and `ParseMap` for the document with its
key order and the plain untyped tree.
- **Strict decoding**: `RejectUnknownFields(true)` rejects keys that - **Strict decoding**: `RejectUnknownFields(true)` rejects keys that
match no destination field, at every struct depth. match no destination field, at every struct depth.
- **Custom types**: `Marshaler` and `Unmarshaler` let a type control its own - **Custom types**: `Marshaler` and `Unmarshaler` let a type control its own
TOML representation in both directions, and `encoding.TextMarshaler` and TOML representation in both directions, and `encoding.TextMarshaler` and
`TextUnmarshaler` are honoured by default, so `net.IP`, `time.Duration` and `TextUnmarshaler` are honoured by default, so `net.IP`, `time.Duration` and
user types with text methods need no configuration. user types with text methods need no configuration.
- **Cancellation**: every entry point has a `*Context` sibling that honours a - **Cancellation**: the parse, decode and marshal entries have `*Context`
`context.Context`. siblings that honour a `context.Context`, checked while the work runs.
- **Ordered documents**: `Parse` gives a `*Document` that keeps the key order, - **Ordered documents**: `Parse` gives a `*Document` that keeps the key order,
tells an inline table from a header one, and carries the comments; `ParseMap` tells an inline table from a header one, and carries the comments; `ParseMap`
gives the plain `map[string]any` tree. gives the plain `map[string]any` tree.
@@ -173,7 +174,8 @@ See [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the full workflow, and
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow - [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/API.md](docs/API.md): the API reference, decoding and encoding rules - [docs/API.md](docs/API.md): the API reference, decoding and encoding rules
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator - [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator,
also shipped as the manual page `man/interpres-decode.1`
## Licence ## Licence
+89
View File
@@ -0,0 +1,89 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"errors"
"strings"
"testing"
)
// errReader fails every read with a fixed error.
type errReader struct{ err error }
func (r errReader) Read([]byte) (int, error) { return 0, r.err }
// TestUnmarshalRead covers the streaming entry: the happy path with options,
// a failing reader, and MaxInputSize bounding what a reader is drained into.
func TestUnmarshalRead(t *testing.T) {
var got struct {
Name string `toml:"name"`
N int `toml:"n"`
}
err := UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, RejectUnknownFields(true))
if err != nil {
t.Fatalf("UnmarshalRead: %v", err)
}
if got.Name != "x" {
t.Errorf("Name = %q", got.Name)
}
readErr := errors.New("boom")
if err := UnmarshalRead(errReader{readErr}, &got); !errors.Is(err, readErr) {
t.Errorf("err = %v, want the read error wrapped", err)
}
err = UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, MaxInputSize(4))
if err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit", err)
}
// The limit bounds the read itself: a reader that would supply far more
// than the limit is not drained into memory first.
big := strings.Repeat("x", 1<<20)
if err := UnmarshalRead(strings.NewReader(big), &got, MaxInputSize(16)); err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit before the read completes", err)
}
}
// TestParseAsWithOptions covers the generic shorthand carrying options.
func TestParseAsWithOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name"`
}
got, err := ParseAs[cfg]([]byte("name = \"x\"\nrogue = 1\n"), RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), "unknown field") {
t.Errorf("err = %v, want the strict failure", err)
}
// The statements before the failure stay written, the contract the
// targeted path documents and encoding/json follows.
if got.Name != "x" {
t.Errorf("Name = %q, want the statement before the failure kept", got.Name)
}
}
// TestStatementsValueArrays pins that a value array is one statement, a
// scalar array and an array of inline tables alike; only an array of tables
// yields per element.
func TestStatementsValueArrays(t *testing.T) {
src := strings.NewReader("port = [8080, 9090]\nmix = [{y = 1, x = 2}]\n[[items]]\nn = 1\n")
var got []Statement
for stmt, err := range Statements(src) {
if err != nil {
t.Fatal(err)
}
got = append(got, stmt)
}
if len(got) != 3 {
t.Fatalf("got %d statements, want 3", len(got))
}
if got[0].Index != -1 || got[0].Table != nil {
t.Errorf("port statement = %+v, want one plain key/value", got[0])
}
if got[1].Index != -1 || got[1].Table != nil {
t.Errorf("mix statement = %+v, want one plain key/value", got[1])
}
if got[2].Index != 0 || got[2].Table == nil {
t.Errorf("items statement = %+v, want the element with its node", got[2])
}
}
+99 -41
View File
@@ -6,9 +6,11 @@ package main
import ( import (
"fmt" "fmt"
"io" "io"
"strconv"
"strings" "strings"
"time" "time"
"unicode" "unicode"
"unicode/utf8"
"sourcedock.dev/petrbalvin/interpres/v2" "sourcedock.dev/petrbalvin/interpres/v2"
) )
@@ -17,63 +19,115 @@ import (
// like the document: one field per key in written order, nested tables as // like the document: one field per key in written order, nested tables as
// nested struct types, an array of tables as a slice, and the field names // nested struct types, an array of tables as a slice, and the field names
// invented from the keys. It is the onboarding aid: the printed type compiles // invented from the keys. It is the onboarding aid: the printed type compiles
// and decodes the document it came from. // and decodes the document it came from. The definition is built whole and
// written with a single call, so a failing standard output surfaces as one
// error instead of being dropped mid-print.
func inferStruct(data []byte, stdout io.Writer) error { func inferStruct(data []byte, stdout io.Writer) error {
doc, err := interpres.Parse(data) doc, err := interpres.Parse(data)
if err != nil { if err != nil {
return err return err
} }
fmt.Fprintln(stdout, "// Generated by interpres-decode -struct; decode with") body := &strings.Builder{}
fmt.Fprintln(stdout, "// sourcedock.dev/petrbalvin/interpres/v2.") fmt.Fprintln(body, "// Generated by interpres-decode --struct; decode with")
fmt.Fprintln(stdout, "type inferred struct {") fmt.Fprintln(body, "// sourcedock.dev/petrbalvin/interpres/v2.")
if err := writeInferredFields(stdout, doc.Root(), map[string]bool{}); err != nil { fmt.Fprintln(body, "type inferred struct {")
writeInferredFields(body, tableFields(doc.Root()), map[string]bool{})
fmt.Fprintln(body, "}")
_, err = io.WriteString(stdout, body.String())
return err return err
} }
fmt.Fprintln(stdout, "}")
return nil
}
// writeInferredFields writes one field per entry of the table. invented // inferredField is one document key with the entry it is inferred from.
// tracks the field names already used at one level, so two keys that clean type inferredField struct {
// to the same name do not collide. key string
func writeInferredFields(w io.Writer, t *interpres.Table, invented map[string]bool) error { entry *interpres.Entry
}
// tableFields lists a table's entries in written order.
func tableFields(t *interpres.Table) []inferredField {
out := make([]inferredField, 0, len(t.Keys()))
for _, key := range t.Keys() { for _, key := range t.Keys() {
entry, _ := t.Get(key) entry, _ := t.Get(key)
name := goFieldName(key, invented) out = append(out, inferredField{key: key, entry: entry})
}
return out
}
// mergedTableFields merges the key sets of an array's elements in first-seen
// order. An array's type has to cover every element, and a key may appear
// only in a later one, so the first element alone does not decide the shape;
// each key is inferred from the first element that carries it.
func mergedTableFields(tables []*interpres.Table) []inferredField {
var out []inferredField
seen := map[string]bool{}
for _, t := range tables {
for _, f := range tableFields(t) {
if seen[f.key] {
continue
}
seen[f.key] = true
out = append(out, f)
}
}
return out
}
// writeInferredFields writes one field per entry, in the order given.
// invented tracks the field names already used at one level, so two keys
// that clean to the same name do not collide.
func writeInferredFields(w *strings.Builder, fields []inferredField, invented map[string]bool) {
for _, f := range fields {
writeInferredField(w, f, invented)
}
}
// writeInferredField writes one field for one entry: an array of tables as a
// slice of structs, a child table as a nested struct, and everything else as
// the scalar or slice the decoded value names.
func writeInferredField(w *strings.Builder, f inferredField, invented map[string]bool) {
name := goFieldName(f.key, invented)
// An array of tables carries a node per element; the nodes of a value // An array of tables carries a node per element; the nodes of a value
// array are nil wherever an element is not a table, so the nils give // array are nil wherever an element is not a table. Every node present
// it away. // is what tells the two apart: [1, {x=1}] stays a value array even
var tables []*interpres.Table // though one of its elements is a table.
for _, el := range entry.Elements() { elements := f.entry.Elements()
if el != nil { allTables := len(elements) > 0
tables = append(tables, el) for _, el := range elements {
if el == nil {
allTables = false
break
} }
} }
if len(tables) > 0 { if allTables {
// The type comes from the first element.
fmt.Fprintf(w, "\t%s []struct {\n", name) fmt.Fprintf(w, "\t%s []struct {\n", name)
if err := writeInferredFields(w, tables[0], map[string]bool{}); err != nil { writeInferredFields(w, mergedTableFields(elements), map[string]bool{})
return err fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
} }
fmt.Fprintf(w, "\t} `toml:%q`\n", key) if child := f.entry.Table(); child != nil {
continue
}
val := entry.Value()
if child := entry.Table(); child != nil {
fmt.Fprintf(w, "\t%s struct {\n", name) fmt.Fprintf(w, "\t%s struct {\n", name)
if err := writeInferredFields(w, child, map[string]bool{}); err != nil { writeInferredFields(w, tableFields(child), map[string]bool{})
return err fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
} return
fmt.Fprintf(w, "\t} `toml:%q`\n", key)
continue
} }
val := f.entry.Value()
if items, ok := val.([]any); ok { if items, ok := val.([]any); ok {
fmt.Fprintf(w, "\t%s []%s `toml:%q`\n", name, inferScalarType(items), key) fmt.Fprintf(w, "\t%s []%s %s\n", name, inferScalarType(items), structTag(f.key))
continue return
} }
fmt.Fprintf(w, "\t%s %s `toml:%q`\n", name, goTypeOf(val), key) fmt.Fprintf(w, "\t%s %s %s\n", name, goTypeOf(val), structTag(f.key))
} }
return nil
// structTag renders the toml tag of one key as a Go string literal. The raw
// backtick literal is the conventional shape, but a key carrying a backtick
// would end that literal early and the printed definition would not compile,
// so such tags are rendered with strconv.Quote instead.
func structTag(key string) string {
tag := `toml:"` + key + `"`
if !strings.ContainsAny(tag, "`\r") {
return "`" + tag + "`"
}
return strconv.Quote(tag)
} }
// goTypeOf names the Go type the decoded value asks for. // goTypeOf names the Go type the decoded value asks for.
@@ -106,8 +160,10 @@ func goTypeOf(val any) string {
} }
// goFieldName cleans a document key into an exported Go identifier: the // goFieldName cleans a document key into an exported Go identifier: the
// words the punctuation splits become capitalised runs, a leading digit gains // words the punctuation splits become capitalised runs, a leading digit
// an underscore, and a collision with an earlier name gains a counter. // gains a Field prefix, because an underscore would leave the field
// unexported and the decoder would skip it, and a collision with an earlier
// name gains a counter.
func goFieldName(key string, invented map[string]bool) string { func goFieldName(key string, invented map[string]bool) string {
var b strings.Builder var b strings.Builder
nextUpper := true nextUpper := true
@@ -127,8 +183,10 @@ func goFieldName(key string, invented map[string]bool) string {
if name == "" { if name == "" {
name = "Field" name = "Field"
} }
if unicode.IsDigit(rune(name[0])) { // The first rune is decoded rather than taken as a byte, because a key
name = "_" + name // may open with a digit beyond ASCII.
if first, _ := utf8.DecodeRuneInString(name); unicode.IsDigit(first) {
name = "Field" + name
} }
for invented[name] { for invented[name] {
name += "2" name += "2"
+73 -31
View File
@@ -4,16 +4,16 @@
// Command interpres-decode is the toml-test harness adapter and a TOML // Command interpres-decode is the toml-test harness adapter and a TOML
// validator. Without flags it reads a TOML document from standard input and // validator. Without flags it reads a TOML document from standard input and
// writes the toml-test "tagged JSON" representation to standard output. With // writes the toml-test "tagged JSON" representation to standard output. With
// -encode it is the reverse: it reads tagged JSON and writes the TOML document // --encode it is the reverse: it reads tagged JSON and writes the TOML document
// it describes. With -validate it checks the named documents, or standard // it describes. With --validate it checks the named documents, or standard
// input when none are named, and exits non-zero on the first invalid one: // input when none are named, and exits non-zero when one is invalid:
// //
// interpres-decode -validate config.toml // interpres-decode --validate config.toml
// interpres-decode -encode < case.json // interpres-decode --encode < case.json
// //
// Run the official suite in both directions against the adapter with: // Run the official suite in both directions against the adapter with:
// //
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode -encode' // toml-test test -decoder=./interpres-decode -encoder='./interpres-decode --encode'
package main package main
import ( import (
@@ -39,11 +39,16 @@ func main() {
} }
// Run runs the command line and returns the process exit code: 0 success, // Run runs the command line and returns the process exit code: 0 success,
// 1 an invalid document, 2 a usage, reading, encoding, or // 1 an invalid document, 2 a usage, reading, writing, encoding, or
// unsupported-value error. // unsupported-value error.
func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int { func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError) fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError)
fs.SetOutput(stderr) // The flag package's own diagnostics and default usage render flags
// with a single dash, while the command spells every flag in its
// two-dash long form, the form the manpage documents. Its output is
// therefore discarded and the usage below is the only one printed.
fs.SetOutput(io.Discard)
fs.Usage = func() {}
version := fs.Bool("version", false, "print the version and exit") version := fs.Bool("version", false, "print the version and exit")
validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON") validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON")
encode := fs.Bool("encode", false, "read tagged JSON from stdin and write TOML instead") encode := fs.Bool("encode", false, "read tagged JSON from stdin and write TOML instead")
@@ -52,12 +57,18 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
schemaType := fs.String("schema", "", "write a TOML template for the named struct type; the source file follows as the first argument") schemaType := fs.String("schema", "", "write a TOML template for the named struct type; the source file follows as the first argument")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) { if errors.Is(err, flag.ErrHelp) {
usage(stdout)
return 0 return 0
} }
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
usage(stderr)
return 2 return 2
} }
if *version { if *version {
fmt.Fprintf(stdout, "interpres-decode %s\n", versionString()) if _, err := fmt.Fprintf(stdout, "interpres-decode %s\n", versionString()); err != nil {
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0 return 0
} }
modes := 0 modes := 0
@@ -70,13 +81,19 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
modes++ modes++
} }
if modes > 1 { if modes > 1 {
fmt.Fprintln(stderr, "interpres-decode: -validate, -encode, -struct and -schema cannot be combined") fmt.Fprintln(stderr, "interpres-decode: --validate, --encode, --struct and --schema cannot be combined")
return 2
}
// --json shapes the decoding output only, so it is rejected with every
// mode uniformly instead of being silently ignored by some of them.
if *plainJSON && modes > 0 {
fmt.Fprintln(stderr, "interpres-decode: --json shapes the decoder output and cannot be combined with --encode, --struct, --validate or --schema")
return 2 return 2
} }
if *schemaType != "" { if *schemaType != "" {
rest := fs.Args() rest := fs.Args()
if len(rest) != 1 { if len(rest) != 1 {
fmt.Fprintln(stderr, "interpres-decode: -schema needs the type name and exactly one Go source file") fmt.Fprintln(stderr, "interpres-decode: --schema needs the type name and exactly one Go source file")
return 2 return 2
} }
if err := runSchema(*schemaType, rest[0], stdout); err != nil { if err := runSchema(*schemaType, rest[0], stdout); err != nil {
@@ -88,12 +105,8 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
if *validate { if *validate {
return validatePaths(fs.Args(), stdin, stderr) return validatePaths(fs.Args(), stdin, stderr)
} }
if *encode && *plainJSON {
fmt.Fprintln(stderr, "interpres-decode: -json shapes the decoder output and cannot be combined with -encode")
return 2
}
if fs.NArg() > 0 { if fs.NArg() > 0 {
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with -validate") fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with --validate")
return 2 return 2
} }
if *encode { if *encode {
@@ -101,19 +114,25 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
} }
data, err := io.ReadAll(stdin) data, err := io.ReadAll(stdin)
if err != nil { if err != nil {
fmt.Fprintln(stderr, "read stdin:", err) fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2 return 2
} }
if *infer { if *infer {
if err := inferStruct(data, stdout); err != nil { if err := inferStruct(data, stdout); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err) fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
// A document that fails to parse keeps the adapter's invalid
// exit; anything else, a failed write among them, is a tool
// failure.
if _, ok := errors.AsType[*interpres.SyntaxError](err); ok {
return 1 return 1
} }
return 2
}
return 0 return 0
} }
tree, err := interpres.ParseMap(data) tree, err := interpres.ParseMap(data)
if err != nil { if err != nil {
fmt.Fprintln(stderr, err) fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 1 return 1
} }
if *plainJSON { if *plainJSON {
@@ -121,25 +140,48 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
enc.SetEscapeHTML(false) enc.SetEscapeHTML(false)
enc.SetIndent("", " ") enc.SetIndent("", " ")
if err := enc.Encode(plainJSONValue(tree)); err != nil { if err := enc.Encode(plainJSONValue(tree)); err != nil {
fmt.Fprintln(stderr, "encode:", err) fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2 return 2
} }
return 0 return 0
} }
tagged, err := tag(tree) tagged, err := tag(tree)
if err != nil { if err != nil {
fmt.Fprintln(stderr, err) fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2 return 2
} }
enc := json.NewEncoder(stdout) enc := json.NewEncoder(stdout)
enc.SetEscapeHTML(false) enc.SetEscapeHTML(false)
if err := enc.Encode(tagged); err != nil { if err := enc.Encode(tagged); err != nil {
fmt.Fprintln(stderr, "encode:", err) fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2 return 2
} }
return 0 return 0
} }
// usage prints the command line summary, with every flag in its two-dash
// long form: the flag package's default usage printer renders a single dash,
// and the manpage and docs/CLI.md spell the flags the way this text does.
func usage(w io.Writer) {
fmt.Fprint(w, `Usage: interpres-decode [flags]
Without a mode flag the command reads one TOML document from standard input
and writes the toml-test tagged JSON representation to standard output.
--encode read tagged JSON from standard input and write TOML
instead
--help print this usage
--json with the default mode, print plain indented JSON
instead of tagged JSON
--schema TYPE write a TOML template for the named struct type; the
Go source file follows as the first argument
--struct infer a Go struct definition from the document on
standard input and print it
--validate validate the documents instead of emitting tagged JSON
--version print the version and exit
`)
}
// versionString names the version the binary was built at: the module // versionString names the version the binary was built at: the module
// version the toolchain recorded, which is the tag when the release pipeline // version the toolchain recorded, which is the tag when the release pipeline
// builds it, and (devel) for an ordinary build from a working tree. // builds it, and (devel) for an ordinary build from a working tree.
@@ -226,8 +268,8 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
return 2 return 2
} }
} }
valid := true
checked := 0 checked := 0
invalid := 0
for _, p := range files { for _, p := range files {
name := p name := p
var data []byte var data []byte
@@ -244,17 +286,17 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
} }
checked++ checked++
if _, err := interpres.ParseMap(data); err != nil { if _, err := interpres.ParseMap(data); err != nil {
fmt.Fprintf(stderr, "%s: %v\n", name, err) fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
valid = false invalid++
} }
} }
// The single-document run stays quiet on success, the contract the // The single-document run stays quiet on success, the contract the
// compliance tooling relies on; a directory walk closes with the // compliance tooling relies on; a directory walk closes with the
// summary that makes the sweep readable. // summary that makes the sweep readable.
if dirs > 0 { if dirs > 0 {
fmt.Fprintf(stderr, "checked %d documents, %d invalid\n", checked, map[bool]int{true: 0, false: 1}[valid]) fmt.Fprintf(stderr, "checked %d documents, %d invalid\n", checked, invalid)
} }
if !valid { if invalid > 0 {
return 1 return 1
} }
return 0 return 0
@@ -265,17 +307,17 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int { func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
data, err := io.ReadAll(stdin) data, err := io.ReadAll(stdin)
if err != nil { if err != nil {
fmt.Fprintln(stderr, "read stdin:", err) fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2 return 2
} }
var desc any var desc any
if err := json.Unmarshal(data, &desc); err != nil { if err := json.Unmarshal(data, &desc); err != nil {
fmt.Fprintln(stderr, "decode JSON:", err) fmt.Fprintln(stderr, "interpres-decode: decode JSON:", err)
return 2 return 2
} }
tree, err := untag(desc) tree, err := untag(desc)
if err != nil { if err != nil {
fmt.Fprintln(stderr, err) fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2 return 2
} }
doc, ok := tree.(map[string]any) doc, ok := tree.(map[string]any)
@@ -285,11 +327,11 @@ func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
} }
out, err := interpres.Marshal(doc) out, err := interpres.Marshal(doc)
if err != nil { if err != nil {
fmt.Fprintln(stderr, err) fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2 return 2
} }
if _, err := stdout.Write(out); err != nil { if _, err := stdout.Write(out); err != nil {
fmt.Fprintln(stderr, "write stdout:", err) fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2 return 2
} }
return 0 return 0
+412 -28
View File
@@ -7,6 +7,8 @@ import (
"bytes" "bytes"
"encoding/json" "encoding/json"
"errors" "errors"
"go/parser"
"go/token"
"os" "os"
"path/filepath" "path/filepath"
"reflect" "reflect"
@@ -217,7 +219,7 @@ func TestTaggedHelper(t *testing.T) {
func TestValidateStdinAcceptsValidDocument(t *testing.T) { func TestValidateStdinAcceptsValidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"ok\"\n")) in := bytes.NewReader([]byte("title = \"ok\"\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 0 { if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
} }
if stdout.Len() != 0 || stderr.Len() != 0 { if stdout.Len() != 0 || stderr.Len() != 0 {
@@ -228,7 +230,7 @@ func TestValidateStdinAcceptsValidDocument(t *testing.T) {
func TestValidateStdinRejectsInvalidDocument(t *testing.T) { func TestValidateStdinRejectsInvalidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"unterminated\n")) in := bytes.NewReader([]byte("title = \"unterminated\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 1 { if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 1 {
t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String())
} }
if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") { if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") {
@@ -250,10 +252,10 @@ func TestValidateFiles(t *testing.T) {
t.Fatal(err) t.Fatal(err)
} }
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", good}, nil, &stdout, &stderr); code != 0 { if code := Run([]string{"--validate", good}, nil, &stdout, &stderr); code != 0 {
t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String()) t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String())
} }
if code := Run([]string{"-validate", good, bad}, nil, &stdout, &stderr); code != 1 { if code := Run([]string{"--validate", good, bad}, nil, &stdout, &stderr); code != 1 {
t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String()) t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String())
} }
if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") { if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") {
@@ -263,7 +265,7 @@ func TestValidateFiles(t *testing.T) {
func TestValidateMissingFileReturnsTwo(t *testing.T) { func TestValidateMissingFileReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 { if code := Run([]string{"--validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
} }
} }
@@ -274,14 +276,14 @@ func TestAdapterModeRejectsPositionalArgument(t *testing.T) {
if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 { if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
} }
if !strings.Contains(stderr.String(), "-validate") { if !strings.Contains(stderr.String(), "--validate") {
t.Fatalf("stderr = %q, want it to point at -validate", stderr.String()) t.Fatalf("stderr = %q, want it to point at --validate", stderr.String())
} }
} }
func TestUnknownFlagReturnsTwo(t *testing.T) { func TestUnknownFlagReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
if code := Run([]string{"-nope"}, nil, &stdout, &stderr); code != 2 { if code := Run([]string{"--nope"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
} }
} }
@@ -303,7 +305,7 @@ func TestRunEncoderScalars(t *testing.T) {
} }
` `
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr) code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
} }
@@ -334,7 +336,7 @@ func TestRunEncoderNested(t *testing.T) {
} }
` `
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr) code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
} }
@@ -355,7 +357,7 @@ func TestRunEncoderFloatTagDecides(t *testing.T) {
// tag decides the type; the output must stay a float. // tag decides the type; the output must stay a float.
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := `{"whole": {"type": "float", "value": "1"}, "exp": {"type": "float", "value": "5e+22"}}` in := `{"whole": {"type": "float", "value": "1"}, "exp": {"type": "float", "value": "5e+22"}}`
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr) code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
} }
@@ -380,7 +382,7 @@ func TestRunEncoderRejectsBadInput(t *testing.T) {
} }
for _, c := range cases { for _, c := range cases {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(c.in), &stdout, &stderr) code := Run([]string{"--encode"}, strings.NewReader(c.in), &stdout, &stderr)
if code != 2 { if code != 2 {
t.Errorf("%s: Run returned %d, want 2; stderr = %q", c.name, code, stderr.String()) t.Errorf("%s: Run returned %d, want 2; stderr = %q", c.name, code, stderr.String())
continue continue
@@ -396,7 +398,7 @@ func TestRunEncoderRejectsBadInput(t *testing.T) {
func TestRunEncoderFlagConflicts(t *testing.T) { func TestRunEncoderFlagConflicts(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
if code := Run([]string{"-encode", "-validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 { if code := Run([]string{"--encode", "--validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for the two modes together", code) t.Errorf("Run returned %d, want 2 for the two modes together", code)
} }
if !strings.Contains(stderr.String(), "cannot be combined") { if !strings.Contains(stderr.String(), "cannot be combined") {
@@ -405,7 +407,7 @@ func TestRunEncoderFlagConflicts(t *testing.T) {
stdout.Reset() stdout.Reset()
stderr.Reset() stderr.Reset()
if code := Run([]string{"-encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 { if code := Run([]string{"--encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for an argument", code) t.Errorf("Run returned %d, want 2 for an argument", code)
} }
} }
@@ -432,7 +434,7 @@ n = "a"
t.Fatalf("decode returned %d, stderr = %q", code, stderr.String()) t.Fatalf("decode returned %d, stderr = %q", code, stderr.String())
} }
var out bytes.Buffer var out bytes.Buffer
if code := Run([]string{"-encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 { if code := Run([]string{"--encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
t.Fatalf("encode returned %d, stderr = %q", code, stderr.String()) t.Fatalf("encode returned %d, stderr = %q", code, stderr.String())
} }
want, err := interpres.ParseMap([]byte(doc)) want, err := interpres.ParseMap([]byte(doc))
@@ -450,7 +452,7 @@ n = "a"
func TestRunVersion(t *testing.T) { func TestRunVersion(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run([]string{"-version"}, strings.NewReader(""), &stdout, &stderr) code := Run([]string{"--version"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
} }
@@ -463,7 +465,7 @@ func TestRunVersion(t *testing.T) {
func TestRunPlainJSON(t *testing.T) { func TestRunPlainJSON(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nwhen = 1979-05-27T07:32:00-07:00\nitems = [1, 2]\n") in := strings.NewReader("host = \"db\"\nwhen = 1979-05-27T07:32:00-07:00\nitems = [1, 2]\n")
code := Run([]string{"-json"}, in, &stdout, &stderr) code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
} }
@@ -481,18 +483,31 @@ func TestRunPlainJSON(t *testing.T) {
func TestValidateDirectorySummary(t *testing.T) { func TestValidateDirectorySummary(t *testing.T) {
dir := t.TempDir() dir := t.TempDir()
os.WriteFile(filepath.Join(dir, "good.toml"), []byte("a = 1\n"), 0o644) if err := os.WriteFile(filepath.Join(dir, "good.toml"), []byte("a = 1\n"), 0o644); err != nil {
os.WriteFile(filepath.Join(dir, "bad.toml"), []byte("a =\n"), 0o644) t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "bad.toml"), []byte("a =\n"), 0o644); err != nil {
t.Fatal(err)
}
sub := filepath.Join(dir, "nested") sub := filepath.Join(dir, "nested")
os.Mkdir(sub, 0o755) if err := os.Mkdir(sub, 0o755); err != nil {
os.WriteFile(filepath.Join(sub, "deep.toml"), []byte("b = true\n"), 0o644) t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(sub, "deep.toml"), []byte("b = true\n"), 0o644); err != nil {
t.Fatal(err)
}
// A second invalid document, so the summary's invalid count is
// exercised beyond the single failure the boolean tracked.
if err := os.WriteFile(filepath.Join(sub, "worse.toml"), []byte("c =\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run([]string{"-validate", dir}, strings.NewReader(""), &stdout, &stderr) code := Run([]string{"--validate", dir}, strings.NewReader(""), &stdout, &stderr)
if code != 1 { if code != 1 {
t.Fatalf("Run returned %d, want 1 for a directory with an invalid file", code) t.Fatalf("Run returned %d, want 1 for a directory with invalid files", code)
} }
if !strings.Contains(stderr.String(), "checked 3 documents, 1 invalid") { if !strings.Contains(stderr.String(), "checked 4 documents, 2 invalid") {
t.Errorf("stderr = %q, want the summary", stderr.String()) t.Errorf("stderr = %q, want the summary", stderr.String())
} }
} }
@@ -500,7 +515,7 @@ func TestValidateDirectorySummary(t *testing.T) {
func TestInferStruct(t *testing.T) { func TestInferStruct(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nport = 5432\ntags = [\"a\"]\n\n[server]\nname = \"edge\"\n\n[[items]]\nn = 1\n") in := strings.NewReader("host = \"db\"\nport = 5432\ntags = [\"a\"]\n\n[server]\nname = \"edge\"\n\n[[items]]\nn = 1\n")
code := Run([]string{"-struct"}, in, &stdout, &stderr) code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
} }
@@ -545,7 +560,7 @@ type Item struct {
t.Fatal(err) t.Fatal(err)
} }
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run([]string{"-schema", "Config", src}, strings.NewReader(""), &stdout, &stderr) code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
} }
@@ -572,7 +587,7 @@ func TestRunPlainJSONShapes(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := strings.NewReader("when = 1979-05-27T07:32:00-07:00\nd = 1979-05-27\nt = 07:32:00\nwall = 1979-05-27T07:32:00\n" + in := strings.NewReader("when = 1979-05-27T07:32:00-07:00\nd = 1979-05-27\nt = 07:32:00\nwall = 1979-05-27T07:32:00\n" +
"items = [1, \"two\"]\n\n[[tables]]\nx = true\n") "items = [1, \"two\"]\n\n[[tables]]\nx = true\n")
code := Run([]string{"-json"}, in, &stdout, &stderr) code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
} }
@@ -594,7 +609,7 @@ func TestRunPlainJSONShapes(t *testing.T) {
func TestInferStructScalarShapes(t *testing.T) { func TestInferStructScalarShapes(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := strings.NewReader("f = 1.5\nb = true\nd = 1979-05-27\nldt = 1979-05-27T07:32:00\nlt = 07:32:00\nnums = [1, 2, 3]\nmixed = [1, \"a\"]\nempty = []\n") in := strings.NewReader("f = 1.5\nb = true\nd = 1979-05-27\nldt = 1979-05-27T07:32:00\nlt = 07:32:00\nnums = [1, 2, 3]\nmixed = [1, \"a\"]\nempty = []\n")
code := Run([]string{"-struct"}, in, &stdout, &stderr) code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 { if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
} }
@@ -614,3 +629,372 @@ func TestInferStructScalarShapes(t *testing.T) {
} }
} }
} }
func TestRunHelpPrintsLongFlags(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--help"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, flag := range []string{"--encode", "--help", "--json", "--schema", "--struct", "--validate", "--version"} {
if !strings.Contains(out, flag) {
t.Errorf("usage output missing %q:\n%s", flag, out)
}
}
// Every flag line of the list names its flag in the two-dash long form
// only, so no line opens with a single dash.
for line := range strings.SplitSeq(strings.TrimRight(out, "\n"), "\n") {
if after, ok := strings.CutPrefix(line, " -"); ok && !strings.HasPrefix(after, "-") {
t.Errorf("usage line %q lists a flag with one dash", line)
}
}
}
func TestGoFieldName(t *testing.T) {
cases := []struct{ key, want string }{
{"host", "Host"},
{"ab", "Ab"},
{"http-host", "HttpHost"},
{"3d", "Field3d"},
{"", "Field"},
}
for _, c := range cases {
if got := goFieldName(c.key, map[string]bool{}); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestGoFieldNameCollision(t *testing.T) {
// Two keys that clean to the same name must not collide; the counter
// keeps the fields apart and both stay exported.
invented := map[string]bool{}
cases := []struct{ key, want string }{
{"a-b", "AB"},
{"a b", "AB2"},
{"a_b", "AB22"},
}
for _, c := range cases {
if got := goFieldName(c.key, invented); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestInferStructDigitLeadingKey(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("3d = true\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Field3d bool") {
t.Errorf("output missing the exported Field3d field:\n%s", stdout.String())
}
}
func TestInferStructBacktickKey(t *testing.T) {
// A backtick in the key would end a raw string literal early, so the
// tag has to be rendered as an interpreted literal instead.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("\"a`b\" = 1\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "AB int64 \"toml:\\\"a`b\\\"\"") {
t.Errorf("output missing the quoted tag:\n%s", out)
}
// The printed definition has to compile; parsing it as Go is the
// syntax half of that proof.
if _, err := parser.ParseFile(token.NewFileSet(), "inferred.go", "package p\n\n"+out, 0); err != nil {
t.Errorf("the printed definition does not parse: %v\n%s", err, out)
}
}
func TestInferStructMergesArrayElements(t *testing.T) {
// The second element carries a key the first lacks, so the slice type
// has to be inferred from both.
var stdout, stderr bytes.Buffer
in := strings.NewReader("[[items]]\nn = 1\n\n[[items]]\nextra = \"late\"\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"Items []struct {",
"N int64 `toml:\"n\"`",
"Extra string `toml:\"extra\"`",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestInferStructMixedArrayStaysValueArray(t *testing.T) {
// One table element does not make the array an array of tables; a
// struct slice would not decode the scalar element.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("arr = [1, {x = 1}]\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "Arr []any") {
t.Errorf("output = %q, want a value array typed []any", out)
}
if strings.Contains(out, "[]struct") {
t.Errorf("output = %q, a mixed array must not become a struct slice", out)
}
}
func TestRunStructParseError(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a =\n"), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 {
t.Errorf("stdout should be empty on parse error, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "line 1") {
t.Errorf("stderr = %q, want the library's line number", stderr.String())
}
}
func TestRunStructWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a = 1\n"), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunSchemaNeedsTypeAndExactlyOneFile(t *testing.T) {
for _, args := range [][]string{{"--schema", "Config"}, {"--schema", "Config", "a.go", "b.go"}} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--schema") {
t.Errorf("Run(%v) stderr = %q, want it to name --schema", args, stderr.String())
}
}
}
func TestRunSchemaUnparsableSource(t *testing.T) {
src := filepath.Join(t.TempDir(), "broken.go")
if err := os.WriteFile(src, []byte("this is not Go\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "broken.go") {
t.Errorf("stderr = %q, want it to name the source file", stderr.String())
}
}
func TestRunSchemaUnknownType(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Missing", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), `no struct type "Missing"`) {
t.Errorf("stderr = %q, want it to name the missing type", stderr.String())
}
}
func TestRunSchemaRecursiveType(t *testing.T) {
// A self-referential struct has no finite template; the generator has
// to name the recursion instead of exhausting the stack.
src := filepath.Join(t.TempDir(), "node.go")
body := "package cfg\n\ntype Node struct {\n\tNext *Node `toml:\"next\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Node", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
for _, want := range []string{"recursive", "Node"} {
if !strings.Contains(stderr.String(), want) {
t.Errorf("stderr = %q, want it to mention %q", stderr.String(), want)
}
}
}
func TestRunSchemaMultiNameField(t *testing.T) {
// A field list may name several fields of one type; each name is one
// TOML key.
src := filepath.Join(t.TempDir(), "range.go")
if err := os.WriteFile(src, []byte("package cfg\n\ntype Range struct {\n\tMin, Max int\n}\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Range", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{"min = 0", "max = 0"} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunSchemaEmbeddedStructs(t *testing.T) {
// The library inlines only untagged embedded structs; a tagged one
// keeps its own section.
src := filepath.Join(t.TempDir(), "embed.go")
body := `package cfg
type Inner struct {
X int ` + "`toml:\"x\"`" + `
}
type Tagged struct {
Inner ` + "`toml:\"inner\"`" + `
Y int ` + "`toml:\"y\"`" + `
}
type Flat struct {
Inner
Z int ` + "`toml:\"z\"`" + `
}
`
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Tagged", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Tagged: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "y = 0") || !strings.Contains(out, "[inner]") {
t.Errorf("Tagged output = %q, want a y scalar and an [inner] section", out)
}
stdout.Reset()
stderr.Reset()
code = Run([]string{"--schema", "Flat", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Flat: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out = stdout.String()
if !strings.Contains(out, "x = 0") || !strings.Contains(out, "z = 0") {
t.Errorf("Flat output = %q, want x and z flattened as scalars", out)
}
if strings.Contains(out, "[inner]") {
t.Errorf("Flat output = %q, an untagged embedded struct must not become a section", out)
}
}
func TestRunVersionWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--version"}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunSchemaWriteFailure(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunEncodeWriteFailure(t *testing.T) {
var stderr bytes.Buffer
in := `{"a": {"type": "integer", "value": "1"}}`
code := Run([]string{"--encode"}, strings.NewReader(in), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunEmptyInput(t *testing.T) {
t.Run("default", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run(nil, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if strings.TrimSpace(stdout.String()) != "{}" {
t.Errorf("stdout = %q, want an empty table", stdout.String())
}
})
t.Run("encode", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2, empty input is not JSON", code)
}
})
t.Run("struct", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "type inferred struct {\n}") {
t.Errorf("stdout = %q, want an empty struct", stdout.String())
}
})
t.Run("validate", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--validate"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
t.Errorf("validate should be quiet, stdout %q stderr %q", stdout.String(), stderr.String())
}
})
}
func TestRunJSONFlagConflicts(t *testing.T) {
for _, args := range [][]string{
{"--encode", "--json"},
{"--struct", "--json"},
{"--validate", "--json"},
{"--json", "--schema", "Config", "config.go"},
} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--json") {
t.Errorf("Run(%v) stderr = %q, want it to explain the --json conflict", args, stderr.String())
}
}
}
+103 -28
View File
@@ -9,8 +9,10 @@ import (
"go/parser" "go/parser"
"go/token" "go/token"
"io" "io"
"maps"
"path/filepath" "path/filepath"
"reflect" "reflect"
"slices"
"strconv" "strconv"
"strings" "strings"
) )
@@ -34,7 +36,9 @@ func runSchema(typeName, sourcePath string, stdout io.Writer) error {
return fmt.Errorf("no struct type %q in %s", typeName, filepath.Base(sourcePath)) return fmt.Errorf("no struct type %q in %s", typeName, filepath.Base(sourcePath))
} }
body := &strings.Builder{} body := &strings.Builder{}
writeSchemaFields(body, st, types, "") if err := writeSchemaFields(body, st, types, "", nil); err != nil {
return err
}
_, err = io.WriteString(stdout, strings.TrimLeft(body.String(), "\n")) _, err = io.WriteString(stdout, strings.TrimLeft(body.String(), "\n"))
return err return err
} }
@@ -74,9 +78,18 @@ type fieldMeta struct {
// writeSchemaFields writes the fields of one struct level: the scalar lines // writeSchemaFields writes the fields of one struct level: the scalar lines
// first, then the sections, so the template re-parses with every value under // first, then the sections, so the template re-parses with every value under
// the header it belongs to. prefix is the dotted path the nested headers // the header it belongs to. prefix is the dotted path the nested headers
// carry. // carry. path holds the struct types of the levels currently being written,
func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]*ast.StructType, prefix string) { // so a type that reaches itself is reported as recursion instead of
metas := metasOf(st, types) // exhausting the stack.
func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]*ast.StructType, prefix string, path []*ast.StructType) error {
if slices.Contains(path, st) {
return recursionError(st, types)
}
path = append(path, st)
metas, err := metasOf(st, types)
if err != nil {
return err
}
for _, m := range metas { for _, m := range metas {
if _, elemSt := elementStruct(m.typ, types); elemSt != nil { if _, elemSt := elementStruct(m.typ, types); elemSt != nil {
continue continue
@@ -98,7 +111,9 @@ func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]
writeComment(w, m.comment) writeComment(w, m.comment)
fmt.Fprintf(w, "[%s%s]\n", prefix, m.key) fmt.Fprintf(w, "[%s%s]\n", prefix, m.key)
if sub := structOf(m.typ, types); sub != nil { if sub := structOf(m.typ, types); sub != nil {
writeSchemaFields(w, sub, types, prefix+m.key+".") if err := writeSchemaFields(w, sub, types, prefix+m.key+".", path); err != nil {
return err
}
} }
fmt.Fprintln(w) fmt.Fprintln(w)
} }
@@ -109,9 +124,12 @@ func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]
} }
writeComment(w, m.comment) writeComment(w, m.comment)
fmt.Fprintf(w, "[[%s%s]]\n", prefix, m.key) fmt.Fprintf(w, "[[%s%s]]\n", prefix, m.key)
writeSchemaFields(w, elemSt, types, "") if err := writeSchemaFields(w, elemSt, types, "", path); err != nil {
return err
}
fmt.Fprintln(w) fmt.Fprintln(w)
} }
return nil
} }
// writeComment writes the comment lines above a binding. // writeComment writes the comment lines above a binding.
@@ -125,46 +143,103 @@ func writeComment(w *strings.Builder, text string) {
} }
// metasOf flattens the exported fields of a struct. The key comes from the // metasOf flattens the exported fields of a struct. The key comes from the
// toml tag, or the lower-cased field name; a `-` key drops the field. // toml tag, or the lower-cased field name; a `-` key drops the field. An
func metasOf(st *ast.StructType, types map[string]*ast.StructType) []fieldMeta { // embedded struct without a tag name flattens into its parent, the way the
// library inlines it, while a tagged one keeps its own section.
func metasOf(st *ast.StructType, types map[string]*ast.StructType) ([]fieldMeta, error) {
return flattenMetas(st, types, nil)
}
// flattenMetas is metasOf with the chain of struct types currently being
// flattened, which stops a struct that embeds itself, directly or through
// another embedded type.
func flattenMetas(st *ast.StructType, types map[string]*ast.StructType, chain map[*ast.StructType]bool) ([]fieldMeta, error) {
if chain[st] {
return nil, recursionError(st, types)
}
// A copy per branch: the chain is the path being flattened now, not the
// set ever visited, so a type embedded in two siblings is not mistaken
// for recursion.
chain = maps.Clone(chain)
if chain == nil {
chain = map[*ast.StructType]bool{}
}
chain[st] = true
var out []fieldMeta var out []fieldMeta
for _, field := range st.Fields.List { for _, field := range st.Fields.List {
if len(field.Names) == 0 {
// An untagged embedded struct flattens into the parent.
if ident, ok := baseType(field.Type).(*ast.Ident); ok {
if inner, ok := types[ident.Name]; ok {
out = append(out, metasOf(inner, types)...)
}
}
continue
}
name := field.Names[0].Name
if !ast.IsExported(name) {
continue
}
tagText := "" tagText := ""
if field.Tag != nil { if field.Tag != nil {
tagText, _ = strconv.Unquote(field.Tag.Value) tagText, _ = strconv.Unquote(field.Tag.Value)
} }
toml := reflect.StructTag(tagText).Get("toml") toml := reflect.StructTag(tagText).Get("toml")
key, opts := "", "" key, opts, _ := strings.Cut(toml, ",")
if toml != "" { if len(field.Names) == 0 {
key, opts, _ = strings.Cut(toml, ",") if key == "" {
// An untagged embedded struct flattens into its parent.
if ident, ok := baseType(field.Type).(*ast.Ident); ok {
if inner, ok := types[ident.Name]; ok {
metas, err := flattenMetas(inner, types, chain)
if err != nil {
return nil, err
}
out = append(out, metas...)
}
}
continue
} }
if key == "-" { if key == "-" {
continue continue
} }
if key == "" { // A tagged embedded struct is a section of its own; the tag
key = strings.ToLower(name) // name is the only name it has.
}
out = append(out, fieldMeta{ out = append(out, fieldMeta{
key: key, key: key,
comment: tagOption(opts, "comment="), comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="), def: tagOption(opts, "default="),
typ: field.Type, typ: field.Type,
}) })
continue
} }
return out if key == "-" {
continue
}
// A field list may name several fields of one type, `Min, Max int`;
// each name is one TOML key.
for _, name := range field.Names {
if !ast.IsExported(name.Name) {
continue
}
fieldKey := key
if fieldKey == "" {
fieldKey = strings.ToLower(name.Name)
}
out = append(out, fieldMeta{
key: fieldKey,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
}
}
return out, nil
}
// recursionError names the struct type that reached itself. Such a type has
// no finite TOML template: every level would nest another copy of the same
// shape.
func recursionError(st *ast.StructType, types map[string]*ast.StructType) error {
return fmt.Errorf("recursive type %s: the struct contains itself, so it has no finite template", typeName(st, types))
}
// typeName names the declared struct type st refers to, and "anonymous
// struct" for a literal one that no declaration names.
func typeName(st *ast.StructType, types map[string]*ast.StructType) string {
for name, t := range types {
if t == st {
return name
}
}
return "anonymous struct"
} }
// tagOption returns the text a `name=` option carries in the option part of // tagOption returns the text a `name=` option carries in the option part of
+32 -15
View File
@@ -79,7 +79,9 @@ func clockString(t time.Time) string {
// offsetString renders an offset date-time, the fourth TOML kind, in the same // offsetString renders an offset date-time, the fourth TOML kind, in the same
// shape: no zero seconds, no trailing zeros in the fraction, and the offset // shape: no zero seconds, no trailing zeros in the fraction, and the offset
// written as "Z" when it is zero. // written as "Z" when it is zero. A zone offset that is not a whole number of
// minutes loses its seconds to this rendering, which is why Marshal refuses
// such a value rather than writing it.
func offsetString(t time.Time) string { func offsetString(t time.Time) string {
buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T") buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
buf = appendClock(buf, t) buf = appendClock(buf, t)
@@ -235,17 +237,21 @@ func normaliseDateTimeToken(tok string, kind dateTimeKind) string {
// parseDateTime classifies and parses a bare token as a TOML date-time value. // parseDateTime classifies and parses a bare token as a TOML date-time value.
// It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or // It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or
// LocalTime) and whether the token was a date-time at all. // LocalTime), whether the token was a date-time at all, and an error for a
func parseDateTime(tok string) (any, bool) { // token whose shape is a date-time a component of which lies outside its
// range: an hour of 24, a day the month does not hold. Such a token is a
// broken date-time, not some other value, so the error names it instead of
// leaving it to the number decoder's complaint.
func parseDateTime(tok string) (any, bool, error) {
if tok == "" || tok[0] < '0' || tok[0] > '9' { if tok == "" || tok[0] < '0' || tok[0] > '9' {
return nil, false return nil, false, nil
} }
if !strings.ContainsAny(tok, "-:") { if !strings.ContainsAny(tok, "-:") {
return nil, false return nil, false, nil
} }
kind, seconds := scanDateTimeShape(tok) kind, seconds := scanDateTimeShape(tok)
if kind == dateTimeNone { if kind == dateTimeNone {
return nil, false return nil, false, nil
} }
norm := normaliseDateTimeToken(tok, kind) norm := normaliseDateTimeToken(tok, kind)
switch kind { switch kind {
@@ -256,7 +262,7 @@ func parseDateTime(tok string) (any, bool) {
} }
t, err := time.Parse(layout, norm) t, err := time.Parse(layout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
// A zero offset carries its own anonymous location from time.Parse, // A zero offset carries its own anonymous location from time.Parse,
// while the written form is "Z" either way; normalising to UTC keeps // while the written form is "Z" either way; normalising to UTC keeps
@@ -264,7 +270,7 @@ func parseDateTime(tok string) (any, bool) {
if _, off := t.Zone(); off == 0 { if _, off := t.Zone(); off == 0 {
t = t.In(time.UTC) t = t.In(time.UTC)
} }
return OffsetDateTime{t}, true return OffsetDateTime{t}, true, nil
case dateTimeLocal: case dateTimeLocal:
layout := localClockLayout layout := localClockLayout
if seconds { if seconds {
@@ -272,15 +278,15 @@ func parseDateTime(tok string) (any, bool) {
} }
t, err := time.Parse(layout, norm) t, err := time.Parse(layout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
return LocalDateTime{t}, true return LocalDateTime{t}, true, nil
case dateTimeDate: case dateTimeDate:
t, err := time.Parse(localDateOnlyLayout, norm) t, err := time.Parse(localDateOnlyLayout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
return LocalDate{t}, true return LocalDate{t}, true, nil
case dateTimeClock: case dateTimeClock:
layout := localTimeClockLayout layout := localTimeClockLayout
if seconds { if seconds {
@@ -288,11 +294,22 @@ func parseDateTime(tok string) (any, bool) {
} }
t, err := time.Parse(layout, norm) t, err := time.Parse(layout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
return LocalTime{t}, true return LocalTime{t}, true, nil
} }
return nil, false return nil, false, nil
}
// wholeMinuteOffset reports an error when the zone offset carries seconds, a
// shape no TOML offset can hold: writing only the minutes would silently
// shift the instant on the way back, so the encoder refuses the value rather
// than corrupting it.
func wholeMinuteOffset(t time.Time) error {
if _, off := t.Zone(); off%60 != 0 {
return fmt.Errorf("interpres: date-time offset of %d seconds is not a whole number of minutes, which TOML cannot write", off)
}
return nil
} }
// isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a // isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a
+21 -4
View File
@@ -7,6 +7,7 @@ import (
"context" "context"
"encoding" "encoding"
"fmt" "fmt"
"maps"
"reflect" "reflect"
"slices" "slices"
"strings" "strings"
@@ -337,7 +338,8 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
if len(schema.required) > 0 { if len(schema.required) > 0 {
seen = make(map[string]bool, len(tbl)) seen = make(map[string]bool, len(tbl))
} }
for key, val := range tbl { for _, key := range d.tableKeys(tbl) {
val := tbl[key]
// A key that is already lowercase, which document keys usually are, // A key that is already lowercase, which document keys usually are,
// hits the map directly; only a miss pays for the case fold. // hits the map directly; only a miss pays for the case fold.
resolved := key resolved := key
@@ -349,12 +351,15 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
if !ok { if !ok {
if schema.embedMaps != nil { if schema.embedMaps != nil {
// Leftover keys land in an untagged embedded map, the inverse // Leftover keys land in an untagged embedded map, the inverse
// of the encoder inlining that map's entries. // of the encoder inlining that map's entries. The assign call
// rather than assignMap itself lets it allocate the embedded
// pointer the field may be, the way any other destination is
// reached.
mv, err := fieldByIndex(dst, schema.embedMaps[0]) mv, err := fieldByIndex(dst, schema.embedMaps[0])
if err != nil { if err != nil {
return newDecodeError(key, err) return newDecodeError(key, err)
} }
if err := d.assignMap(map[string]any{key: val}, mv); err != nil { if err := d.assign(map[string]any{key: val}, mv); err != nil {
return newDecodeError(key, err) return newDecodeError(key, err)
} }
} }
@@ -379,6 +384,18 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
return nil return nil
} }
// tableKeys returns the keys of tbl in the order the document wrote them
// when the node index knows it, and in sorted order otherwise, the order a
// hand-built tree or a node-free parse offers. The order settles which of
// two keys that differ only in case wins one field: the same key wins every
// run, instead of whichever a map iteration happened to hand out.
func (d *decoder) tableKeys(tbl map[string]any) []string {
if node := d.nodeOf(tbl); node != nil {
return node.Keys()
}
return slices.Sorted(maps.Keys(tbl))
}
func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error { func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
if dst.Type().Key().Kind() != reflect.String { if dst.Type().Key().Kind() != reflect.String {
return fmt.Errorf("interpres: map key must be a string, got %s", dst.Type().Key()) return fmt.Errorf("interpres: map key must be a string, got %s", dst.Type().Key())
@@ -499,7 +516,7 @@ func (d *decoder) setLocalTimeValue(t time.Time, dst reflect.Value) error {
t.Hour(), t.Minute(), t.Second(), t.Nanosecond(), d.loc))) t.Hour(), t.Minute(), t.Second(), t.Nanosecond(), d.loc)))
return nil return nil
} }
return fmt.Errorf("interpres: cannot assign local date-time to time.Time; set Decoder.LocalTimeLocation to choose the zone") return fmt.Errorf("interpres: cannot assign local date-time to time.Time; set LocalTimeLocation to choose the zone")
} }
return fmt.Errorf("interpres: cannot assign local date-time to %s", dst.Type()) return fmt.Errorf("interpres: cannot assign local date-time to %s", dst.Type())
} }
+82
View File
@@ -1690,3 +1690,85 @@ func TestErrorMessagesGolden(t *testing.T) {
}) })
} }
} }
// TestLocalTimeLocationLeavesOffsetsAlone pins that the option's zone is
// used for local date-times only: an offset date-time keeps the offset the
// document wrote.
func TestLocalTimeLocationLeavesOffsetsAlone(t *testing.T) {
var cfg struct {
Stamp time.Time `toml:"stamp"`
}
err := Unmarshal([]byte("stamp = 1979-05-27T07:32:00-07:00\n"), &cfg,
LocalTimeLocation(time.FixedZone("Prague", 2*60*60)))
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if _, off := cfg.Stamp.Zone(); off != -7*60*60 {
t.Errorf("offset = %d, want the document's -07:00", off/3600)
}
}
// TestUnmarshalCaseCollisionIsDeterministic pins that two keys differing
// only in case, both matching one field, resolve the same way on every run
// and on both decode paths.
func TestUnmarshalCaseCollisionIsDeterministic(t *testing.T) {
type cfg struct {
Host string `toml:"host"`
}
in := []byte("Host = \"upper\"\nhost = \"lower\"\n")
// The tree path iterates a map, so pin the winner across many runs.
var want string
for range 50 {
var viaTree cfg
if err := treeDecodeInto(in, &viaTree); err != nil {
t.Fatalf("tree decode: %v", err)
}
if want == "" {
want = viaTree.Host
} else if viaTree.Host != want {
t.Fatalf("tree decode is not deterministic: %q then %q", want, viaTree.Host)
}
}
var targeted cfg
if err := Unmarshal(in, &targeted); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if targeted.Host != want {
t.Errorf("targeted Host = %q, tree %q", targeted.Host, want)
}
}
// TestUnmarshalEmbeddedPointerMap pins that leftover keys reach an embedded
// pointer to a map, allocating it, rather than panicking on the pointer.
func TestUnmarshalEmbeddedPointerMap(t *testing.T) {
type Extra map[string]int
type cfg struct {
*Extra
Name string `toml:"name"`
}
var c cfg
err := Unmarshal([]byte("name = \"x\"\nrogue = 7\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Extra == nil || (*c.Extra)["rogue"] != 7 {
t.Errorf("embedded map = %v, want rogue allocated and filled", c.Extra)
}
}
// TestUnmarshalIgnoresEncodeTagOptions pins that the emission-only tag
// options change nothing on the decode side.
func TestUnmarshalIgnoresEncodeTagOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name,omitempty"`
Port int `toml:"port,omitzero,comment=The port"`
}
var c cfg
err := Unmarshal([]byte("name = \"x\"\nport = 8080\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Name != "x" || c.Port != 8080 {
t.Errorf("cfg = %+v, want both fields filled", c)
}
}
+16 -17
View File
@@ -105,7 +105,7 @@ Encode options:
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes | | `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
| `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header | | `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header |
### `func ParseAs[T any](data []byte) (T, error)` ### `func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error)`
The generic shorthand for `Unmarshal` with a destination variable: The generic shorthand for `Unmarshal` with a destination variable:
@@ -125,7 +125,8 @@ that is not a struct warms nothing.
### `func Statements(r io.Reader) iter.Seq2[Statement, error]` ### `func Statements(r io.Reader) iter.Seq2[Statement, error]`
Iterates the top-level statements of the document r carries, in written Iterates the top-level statements of the document r carries, in written
order: key/value statements, a `[table]` header as one statement carrying order: key/value statements, a value array or an inline table among them as
one statement whatever it holds, a `[table]` header as one statement carrying
its `Table` node, and an `[[array of tables]]` as one statement per element its `Table` node, and an `[[array of tables]]` as one statement per element
with the element's node and its `Index`. Iteration stops at the first error with the element's node and its `Index`. Iteration stops at the first error
and at a false yield, so a caller looking for one section reads no further. and at a false yield, so a caller looking for one section reads no further.
@@ -223,11 +224,7 @@ introduced:
and no surrounding space, so `# note` is stored as `note` and a bare `#` as and no surrounding space, so `# note` is stored as `note` and a bare `#` as
`""`. `""`.
A `Document` is not a value to marshal: `Marshal` writes values, so it refuses ### `func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error`
one and points at `doc.Map()`. Writing a document back, with its order and its
comments, belongs with the editing API.
### `func Unmarshal(data []byte, v any) error`
Parses `data` and stores the result in the value pointed to by `v`, typically a Parses `data` and stores the result in the value pointed to by `v`, typically a
pointer to a struct or to `map[string]any`. Equivalent to pointer to a struct or to `map[string]any`. Equivalent to
@@ -240,11 +237,11 @@ if err := interpres.Unmarshal(data, &cfg); err != nil {
} }
``` ```
### `func UnmarshalContext(ctx context.Context, data []byte, v any) error` ### `func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...UnmarshalOption) error`
The cancellable variant of `Unmarshal`. The cancellable variant of `Unmarshal`.
### `func Marshal(v any) ([]byte, error)` ### `func Marshal(v any, opts ...MarshalOption) ([]byte, error)`
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
TOML document. The emission rules are in the [Encoding](#encoding) section TOML document. The emission rules are in the [Encoding](#encoding) section
@@ -254,12 +251,12 @@ below. Equivalent to `MarshalContext(context.Background(), v)`.
out, err := interpres.Marshal(cfg) out, err := interpres.Marshal(cfg)
``` ```
### `func MarshalContext(ctx context.Context, v any) ([]byte, error)` ### `func MarshalContext(ctx context.Context, v any, opts ...MarshalOption) ([]byte, error)`
The cancellable variant of `Marshal`. The context is checked before any work The cancellable variant of `Marshal`. The context is checked before any work
and every 64 fields during the reflection walk. and every 64 fields during the reflection walk.
### `func MarshalAppend(buf []byte, v any) ([]byte, error)` ### `func MarshalAppend(buf []byte, v any, opts ...MarshalOption) ([]byte, error)`
Appends the TOML encoding of `v` to `buf` and returns the extended buffer, the Appends the TOML encoding of `v` to `buf` and returns the extended buffer, the
shape `json.MarshalAppend` has. A failed encoding leaves `buf` untouched. shape `json.MarshalAppend` has. A failed encoding leaves `buf` untouched.
@@ -525,9 +522,9 @@ through the ordinary assignment rules, so every conversion, hook and error the
path is pinned by a differential fuzz target that decodes every generated path is pinned by a differential fuzz target that decodes every generated
document both ways and compares the results. document both ways and compares the results.
A document or destination the direct skeleton cannot model — an unknown table A document or destination the direct skeleton cannot model (an unknown table
under strictness it must sink, a hook that needs the whole parsed value, an under strictness it must sink, a hook that needs the whole parsed value, an
embedded map filler — falls back to the tree path and reruns, so the embedded map filler) falls back to the tree path and reruns, so the
observable behaviour is always the tree path's, exactly. Nothing changes for observable behaviour is always the tree path's, exactly. Nothing changes for
`Parse`, `ParseMap` or the document API: the tree remains theirs. `Parse`, `ParseMap` or the document API: the tree remains theirs.
@@ -561,8 +558,10 @@ sequenceDiagram
### Input constraints ### Input constraints
`Marshal` and `MarshalWrite` accept a `struct`, a `map[string]V`, or a `Marshal` and `MarshalWrite` accept a `struct`, a `map[string]V`, or a
non-nil pointer to one, where `V` is any value `Marshal` itself understands. A non-nil pointer to one, where `V` is any value `Marshal` itself understands.
different top-level value fails: An `OrderedMap` and a `Document` are accepted as themselves: the first in its
written key order, the second written back as it stands. A different
top-level value fails:
| Input | Error | | Input | Error |
|---|---| |---|---|
@@ -736,7 +735,7 @@ across a round-trip.
A nil slice is always omitted. An empty (length 0) array of tables is always A nil slice is always omitted. An empty (length 0) array of tables is always
omitted, because TOML forbids an empty `[[a]]`. Other empty arrays emit as omitted, because TOML forbids an empty `[[a]]`. Other empty arrays emit as
`key = []` by default; `OmitEmptyArrays()` skips them as well, so `key = []` by default; `OmitEmptyArrays(true)` skips them as well, so
`[]string{}` is treated like a nil slice. `[]string{}` is treated like a nil slice.
### Long strings ### Long strings
@@ -863,7 +862,7 @@ encoding/json/v2 made current, with the differences TOML asks for:
| `encoding.TextMarshaler`, `TextUnmarshaler` | honoured, the same | a type that renders itself as text becomes a TOML string, both ways | | `encoding.TextMarshaler`, `TextUnmarshaler` | honoured, the same | a type that renders itself as text becomes a TOML string, both ways |
| `*json.UnmarshalTypeError` | `*DecodeError` | the path is segments with a `String()` renderer, not a dotted string | | `*json.UnmarshalTypeError` | `*DecodeError` | the path is segments with a `String()` renderer, not a dotted string |
| `*json.SyntaxError` | `*SyntaxError` | the TOML error adds the byte `Offset` and the `Column` to the line | | `*json.SyntaxError` | `*SyntaxError` | the TOML error adds the byte `Offset` and the `Column` to the line |
| context support | `*Context` variants of every entry point | encoding/json has none | | context support | `*Context` variants of the parse, decode and marshal entries | encoding/json has none |
## Types ## Types
+26 -11
View File
@@ -5,16 +5,17 @@ source tree; nothing is aspirational.
## Overview ## Overview
interpres is one public library package, one command, and one example. The interpres is one public library package, one command, and two examples. The
library implements the whole of TOML 1.1, decoding and encoding, in the library implements the whole of TOML 1.1, decoding and encoding, in the
standard library alone; the command wraps the parser and the encoder for the standard library alone; the command wraps the parser and the encoder for the
toml-test compliance harness, against which it stands at 214 valid, 467 invalid toml-test compliance harness, against which it stands at 214 valid, 467 invalid
and 214 encoder cases with zero failures; the example demonstrates the API. and 214 encoder cases with zero failures; the examples demonstrate the API: one the document round trip, one the statement iterator.
```mermaid ```mermaid
flowchart TD flowchart TD
CLI[cmd/interpres-decode<br/>toml-test adapter] --> API CLI[cmd/interpres-decode<br/>toml-test adapter] --> API
EX[examples/basic<br/>usage demo] --> API EX[examples/basic<br/>usage demo] --> API
EX2[examples/statements<br/>statement iterator demo] --> API
subgraph Lib [package interpres] subgraph Lib [package interpres]
API[interpres.go<br/>public API and types] API[interpres.go<br/>public API and types]
API --> P[parser.go<br/>recursive-descent parser] API --> P[parser.go<br/>recursive-descent parser]
@@ -35,9 +36,10 @@ strict validation.
| Path | Responsibility | | Path | Responsibility |
|---|---| |---|---|
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, the option constructors, `Marshaler`, `Unmarshaler`, `SyntaxError`, the local date-time types); everything below it is unexported. | | `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, the option constructors, `Marshaler`, `Unmarshaler`, `SyntaxError`, the error and option types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `-encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. | | `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `--encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
| `examples/basic` | A runnable tour of the API. Documentation in executable form, not part of the library. | | `examples/basic` | A runnable tour of the API. Documentation in executable form, not part of the library. |
| `examples/statements` | The `Statements` iterator over a document, the shape a configuration tool reads. Documentation in executable form. |
Inside the library package, one file owns one concern: Inside the library package, one file owns one concern:
@@ -46,18 +48,28 @@ Inside the library package, one file owns one concern:
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree, records the nodes a [Document](API.md#documents) is built from, and enforces the structural rules of TOML 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. | | `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree, records the nodes a [Document](API.md#documents) is built from, and enforces the structural rules of TOML 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
| `document.go` | The parsed-document types: `Document`, `Table` and `Entry`, which carry the key order, whether a table was written inline, and the comments. The values they expose are the parser's own tree, not a copy. | | `document.go` | The parsed-document types: `Document`, `Table` and `Entry`, which carry the key order, whether a table was written inline, and the comments. The values they expose are the parser's own tree, not a copy. |
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. | | `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. | | `datetime.go` | The four date-time types (`OffsetDateTime` and the three local wrappers) and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `orderedmap.go` | `OrderedMap`, the table that keeps its key order, and the node index the decoder reads the written order from. |
| `target.go` | The targeted parse: the struct skeleton resolved against the document while it scans, no intermediate tree. Falls back to the tree path for every shape it does not model. |
| `docwrite.go` | The write side of the document pipeline: `UnmarshalDocument` and the writer that renders a `Document` back with its order and comments. |
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. | | `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
| `encode.go` | The reverse walk: builds an intermediate `tomlDoc` per table (which is what preserves declaration order and enables the group-by-kind partition) and then emits it as TOML. | | `encode.go` | The reverse walk: builds an intermediate `tomlDoc` per table (which is what preserves declaration order and enables the group-by-kind partition) and then emits it as TOML. |
The boundary that matters: `parser.go` produces only untyped trees The boundary that matters: `parser.go` produces only untyped trees
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); `decode.go` and (`map[string]any`, `[]any`, `[]map[string]any`, scalars); the reflection work
`encode.go` are the only files that touch `reflect`; the command never touches lives in `decode.go`, `encode.go`, `target.go` and `orderedmap.go`; the command
either, it consumes `Parse` alone. consumes `ParseMap`, `Parse` and `Marshal`, and owns no parsing or emission
logic of its own.
## Data flow ## Data flow
Decoding is parse, then one reflection walk. `SyntaxError` values are produced Decoding has two paths. The direct one parses straight into a struct
destination: `target.go` resolves the table skeleton against the struct
schema while the document scans, and values assign through the ordinary
decoder rules, so no intermediate tree exists; that is the hot path every
`Unmarshal` into a struct takes. A document or destination the direct
skeleton cannot model falls back to the tree path: parse the whole document,
then one reflection walk over the tree. `SyntaxError` values are produced
inside `parser.go` and returned as-is; conversion errors are produced inside inside `parser.go` and returned as-is; conversion errors are produced inside
`decode.go` and wrapped with the key path as they unwind. `decode.go` and wrapped with the key path as they unwind.
@@ -66,10 +78,13 @@ sequenceDiagram
participant Caller participant Caller
participant API as interpres.go participant API as interpres.go
participant P as parser.go participant P as parser.go
participant T as target.go
participant D as decode.go participant D as decode.go
Caller->>API: Unmarshal(data, v) Caller->>API: Unmarshal(data, v)
API->>P: ParseContext(ctx, data) API->>T: targeted parse into the struct
P->>P: number and datetime atoms T->>P: scanner, grammar, atoms
T-->>API: result, error or fallback
API->>P: on fallback, ParseContext(ctx, data)
P-->>API: map tree or *SyntaxError P-->>API: map tree or *SyntaxError
API->>D: decode(tree, reflect value) API->>D: decode(tree, reflect value)
D-->>API: nil or wrapped field error D-->>API: nil or wrapped field error
+2
View File
@@ -15,6 +15,8 @@ The benchmarks live in `bench_test.go`, next to the code they measure:
| `BenchmarkParseLong` | `ParseMap` over a generated document with about 2000 array-of-tables entries | | `BenchmarkParseLong` | `ParseMap` over a generated document with about 2000 array-of-tables entries |
| `BenchmarkStrictDecodeLong` | `Unmarshal` into a typed document under `RejectUnknownFields`, over the same long document | | `BenchmarkStrictDecodeLong` | `Unmarshal` into a typed document under `RejectUnknownFields`, over the same long document |
| `BenchmarkMarshalLong` | `Marshal` of the tree `ParseMap` produced from the long document | | `BenchmarkMarshalLong` | `Marshal` of the tree `ParseMap` produced from the long document |
| `BenchmarkStrictDecodeTree` | the tree-path reference decode of the representative document: parse, then the reflection walk |
| `BenchmarkStrictDecodeTreeLong` | the tree-path reference decode of the long document, the A/B baseline of the targeted parse |
## Running ## Running
+45 -40
View File
@@ -3,6 +3,7 @@
The reference below is taken from the program itself. `interpres-decode` is The reference below is taken from the program itself. `interpres-decode` is
the toml-test harness adapter in both directions, decoding TOML into tagged the toml-test harness adapter in both directions, decoding TOML into tagged
JSON and encoding tagged JSON back into TOML, and it also validates documents. JSON and encoding tagged JSON back into TOML, and it also validates documents.
The same reference ships as the manual page `man/interpres-decode.1`.
Install it with Go itself, no release assets involved: Install it with Go itself, no release assets involved:
```sh ```sh
@@ -13,50 +14,52 @@ go install sourcedock.dev/petrbalvin/interpres/v2/cmd/interpres-decode@latest
```sh ```sh
interpres-decode [flags] interpres-decode [flags]
interpres-decode -encode interpres-decode --encode
interpres-decode -validate [file ...] interpres-decode --validate [file ...]
interpres-decode -validate [directory ...] interpres-decode --validate [directory ...]
interpres-decode -json interpres-decode --json
interpres-decode -struct interpres-decode --struct
interpres-decode -schema TYPE file.go interpres-decode --schema TYPE file.go
interpres-decode -version interpres-decode --version
``` ```
Without `-validate`, `-encode`, `-json` or `-struct` the program is the Without `--validate`, `--encode`, `--json`, `--struct` or `--schema` the
decoding half of the toml-test adapter: it takes no arguments, reads one TOML program is the decoding half of the toml-test adapter: it takes no arguments,
document from stdin, and writes the toml-test tagged-JSON form to stdout. reads one TOML document from stdin, and writes the toml-test tagged-JSON form
Build it locally with `just build`, which compiles it into to stdout. Build it locally with `just build`, which compiles it into
`bin/interpres-decode`, or run it straight from the module directory with `bin/interpres-decode`, or run it straight from the module directory with
`just run`. `just run`.
With `-encode` the direction is reversed: the program reads a tagged-JSON With `--encode` the direction is reversed: the program reads a tagged-JSON
description from stdin and writes the TOML document it describes to stdout, description from stdin and writes the TOML document it describes to stdout,
which is the shape toml-test expects of an encoder command. It takes no which is the shape toml-test expects of an encoder command. It takes no
arguments either, and the mode flags cannot be combined. arguments either, and the mode flags cannot be combined.
With `-validate` the program parses each named file instead, or stdin when no With `--validate` the program parses each named file instead, or stdin when no
file is named, and prints one line per invalid document to stderr. It is file is named, and prints one line per invalid document to stderr. It is
quiet on valid documents, which is the shape a CI step wants. The `-` name quiet on valid documents, which is the shape a CI step wants. The `-` name
means stdin. A named directory is walked for `.toml` files, every one of them means stdin. A named directory is walked for `.toml` files, every one of them
validated, and the walk closes with a summary on stderr naming how many validated, and the walk closes with a summary on stderr naming how many
documents were checked and how many were invalid. documents were checked and how many were invalid.
With `-json` the decoding half prints plain indented JSON instead of the With `--json` the decoding half prints plain indented JSON instead of the
tagged form, the shape for people and diffs: the values keep their types as tagged form, the shape for people and diffs: the values keep their types as
JSON sees them, and the date-time wrappers print in their TOML form. JSON sees them, and the date-time wrappers print in their TOML form. The flag
shapes the decoding output only, so it is rejected together with the mode
flags.
With `-struct` the program reads a TOML document from stdin and prints a Go With `--struct` the program reads a TOML document from stdin and prints a Go
struct definition shaped like it: one field per key in written order, nested struct definition shaped like it: one field per key in written order, nested
tables as nested struct types, and an array of tables as a slice. The tables as nested struct types, and an array of tables as a slice. The
printed type compiles and decodes the document it came from. printed type compiles and decodes the document it came from.
With `-schema` the program reads a Go source file and writes a TOML template With `--schema` the program reads a Go source file and writes a TOML template
for the named struct type: one key per exported field, the `comment=` tag for the named struct type: one key per exported field, the `comment=` tag
option printed as a comment above it, and the `default=` option as the value option printed as a comment above it, and the `default=` option as the value
where one is set. It is the inverse of `-struct`, for config-driven where one is set. It is the inverse of `--struct`, for config-driven
applications that generate their example configuration from the type. applications that generate their example configuration from the type.
`-version` prints the binary's version and exits. The release pipeline builds `--version` prints the binary's version and exits. The release pipeline builds
at the tag, so a released binary prints its own tag; a build from a working at the tag, so a released binary prints its own tag; a build from a working
tree prints `(devel)`. tree prints `(devel)`.
@@ -64,21 +67,21 @@ tree prints `(devel)`.
| Flag | Effect | | Flag | Effect |
|---|---| |---|---|
| `-validate` | validate the documents instead of emitting tagged JSON; directories are walked for `.toml` files | | `--validate` | validate the documents instead of emitting tagged JSON; directories are walked for `.toml` files |
| `-encode` | read tagged JSON from stdin and write TOML instead | | `--encode` | read tagged JSON from stdin and write TOML instead |
| `-json` | with the default mode, print plain indented JSON instead of tagged JSON | | `--json` | with the default mode, print plain indented JSON instead of tagged JSON |
| `-struct` | infer a Go struct definition from the document on stdin and print it | | `--struct` | infer a Go struct definition from the document on stdin and print it |
| `-schema TYPE` | write a TOML template for the struct type TYPE from the Go source file named as the first argument | | `--schema TYPE` | write a TOML template for the struct type TYPE from the Go source file named as the first argument |
| `-version` | print the version and exit | | `--version` | print the version and exit |
| `-h` | print the usage | | `--help` | print the usage |
## Exit codes ## Exit codes
| Code | Meaning | | Code | Meaning |
|---|---| |---|---|
| `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed; schema, struct, version: the output was written | | `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed; schema, struct, version: the output was written |
| `1` | adapter: parse error; validate: at least one document is invalid | | `1` | adapter: parse error; validate: at least one document is invalid; struct: the document on stdin failed to parse |
| `2` | a usage error, a read failure, malformed tagged JSON, or a value with no TOML representation | | `2` | a usage error, a read or write failure, malformed tagged JSON, or a value with no TOML representation |
## Wire format ## Wire format
@@ -105,7 +108,7 @@ wrapped in an object with a `type` and a `value`:
| local date | `date-local` | `1979-05-27` | | local date | `date-local` | `1979-05-27` |
| local time | `time-local` | `07:32:00.999999` | | local time | `time-local` | `07:32:00.999999` |
The `-encode` mode reads exactly this form back. Two properties of it are The `--encode` mode reads exactly this form back. Two properties of it are
worth knowing. A float whose value has no fraction and no exponent is written worth knowing. A float whose value has no fraction and no exponent is written
as a bare integer string, `{"type": "float", "value": "1"}`, so there the tag as a bare integer string, `{"type": "float", "value": "1"}`, so there the tag
decides the type and not the literal. And the form cannot tell an array of decides the type and not the literal. And the form cannot tell an array of
@@ -126,10 +129,10 @@ port = 9090
``` ```
The output is the equivalent value tree as one JSON object. Turn a description The output is the equivalent value tree as one JSON object. Turn a description
back into TOML with `-encode`: back into TOML with `--encode`:
```sh ```sh
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode -encode echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode --encode
``` ```
```toml ```toml
@@ -139,14 +142,14 @@ title = "hello"
Validate the TOML files of another repository in CI: Validate the TOML files of another repository in CI:
```sh ```sh
interpres-decode -validate config.toml deploy/example.toml interpres-decode --validate config.toml deploy/example.toml
``` ```
An invalid document reports the file and the library's line number: An invalid document reports the file and the library's line number:
```sh ```sh
$ interpres-decode -validate bad.toml $ interpres-decode --validate bad.toml
bad.toml: interpres: line 1: expected a value interpres-decode: bad.toml: interpres: line 1: expected a value
$ echo $? $ echo $?
1 1
``` ```
@@ -155,22 +158,24 @@ Sweep a whole directory tree of configuration, with the summary the walk
closes on: closes on:
```sh ```sh
$ interpres-decode -validate configs/ $ interpres-decode --validate configs/
configs/old.toml: interpres: line 3: duplicate key "port" interpres-decode: configs/old.toml: interpres: line 3: duplicate key "port"
checked 14 documents, 1 invalid checked 14 documents, 1 invalid
$ echo $? $ echo $?
1 1
``` ```
See the document a `-struct` template would decode: See the document a `--struct` template would decode:
```sh ```sh
echo 'host = "db" echo 'host = "db"
port = 5432 port = 5432
' | ./bin/interpres-decode -struct ' | ./bin/interpres-decode --struct
``` ```
```go ```go
// Generated by interpres-decode --struct; decode with
// sourcedock.dev/petrbalvin/interpres/v2.
type inferred struct { type inferred struct {
Host string `toml:"host"` Host string `toml:"host"`
Port int64 `toml:"port"` Port int64 `toml:"port"`
@@ -182,7 +187,7 @@ the Go source declares fields tagged
`toml:"host,comment=The host to dial,default=example.org"`: `toml:"host,comment=The host to dial,default=example.org"`:
```sh ```sh
./bin/interpres-decode -schema Config config.go ./bin/interpres-decode --schema Config config.go
``` ```
```toml ```toml
@@ -193,7 +198,7 @@ host = "example.org"
Print the binary's version: Print the binary's version:
```sh ```sh
$ ./bin/interpres-decode -version $ ./bin/interpres-decode --version
interpres-decode v2.0.0 interpres-decode v2.0.0
``` ```
+4
View File
@@ -46,6 +46,9 @@ prints the same list.
| `just install` | builds, then copies the binary into `~/.local/bin` (`BINDIR` overrides) | | `just install` | builds, then copies the binary into `~/.local/bin` (`BINDIR` overrides) |
| `just uninstall` | removes the installed binary | | `just uninstall` | removes the installed binary |
| `just clean` | removes `bin/` and `coverage.out` | | `just clean` | removes `bin/` and `coverage.out` |
| `just cross` | cross-compile smoke of the library and the command for arm64, loong64, riscv64 and the browser and edge runtimes; a hand-run convenience, not a gate |
| `just release-check X.Y.Z` | the release pre-flight: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release |
| `just docs-drift` | compares the toml-test counts the documentation quotes with a live suite run |
## Running a single test ## Running a single test
@@ -99,6 +102,7 @@ pipeline.
|---|---|---| |---|---|---|
| `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite | | `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite |
| `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally | | `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally |
| `fuzz.yml` | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut | | `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut |
## Releases ## Releases
+121 -25
View File
@@ -34,15 +34,31 @@ func (d *Document) Root() *Table {
} }
// Map returns the value tree, the shape ParseMap gives. It is the tree the // Map returns the value tree, the shape ParseMap gives. It is the tree the
// document was parsed into, not a copy. // document was parsed into, not a copy. A nil document or one with no root
func (d *Document) Map() map[string]any { return d.root.values } // holds no values.
func (d *Document) Map() map[string]any {
if d == nil || d.root == nil {
return nil
}
return d.root.values
}
// Footer returns the comment lines that follow the last statement, and every // Footer returns the comment lines that follow the last statement, and every
// line of a document that holds no statement at all. // line of a document that holds no statement at all.
func (d *Document) Footer() []string { return d.footer } func (d *Document) Footer() []string {
if d == nil {
return nil
}
return d.footer
}
// SetFooter replaces those lines. // SetFooter replaces those lines.
func (d *Document) SetFooter(lines []string) { d.footer = lines } func (d *Document) SetFooter(lines []string) {
if d == nil {
return
}
d.footer = lines
}
// The document-level convenience forms of the Table edit API; they act on // The document-level convenience forms of the Table edit API; they act on
// the root table. // the root table.
@@ -87,6 +103,11 @@ type Table struct {
// rather than under a header or as a dotted key. // rather than under a header or as a dotted key.
inline bool inline bool
// dotted records that a dotted key introduced the table, `a.b = 1`
// building the a around the leaf: the write side gives such a table back
// as dotted key lines, the form that holds the position of a line.
dotted bool
// comments are the lines above the table's header, trailing is the comment // comments are the lines above the table's header, trailing is the comment
// on the header's own line. Both are empty for a table a dotted key // on the header's own line. Both are empty for a table a dotted key
// introduced, which has no line of its own. // introduced, which has no line of its own.
@@ -98,8 +119,12 @@ func newTable(values map[string]any) *Table {
return &Table{values: values, index: map[string]*Entry{}} return &Table{values: values, index: map[string]*Entry{}}
} }
// Keys returns the table's keys in the order they were written. // Keys returns the table's keys in the order they were written. A nil table
// holds none, the answer a document without a root gives through Root.
func (t *Table) Keys() []string { func (t *Table) Keys() []string {
if t == nil {
return nil
}
keys := make([]string, len(t.entries)) keys := make([]string, len(t.entries))
for i, e := range t.entries { for i, e := range t.entries {
keys[i] = e.key keys[i] = e.key
@@ -109,43 +134,75 @@ func (t *Table) Keys() []string {
// Values returns the table's values, which is the map the value tree holds for // Values returns the table's values, which is the map the value tree holds for
// it. // it.
func (t *Table) Values() map[string]any { return t.values } func (t *Table) Values() map[string]any {
if t == nil {
return nil
}
return t.values
}
// Entries returns the table's entries in written order. // Entries returns the table's entries in written order.
func (t *Table) Entries() []*Entry { return t.entries } func (t *Table) Entries() []*Entry {
if t == nil {
return nil
}
return t.entries
}
// Get returns the entry for key, and whether the table has one. // Get returns the entry for key, and whether the table has one.
func (t *Table) Get(key string) (*Entry, bool) { func (t *Table) Get(key string) (*Entry, bool) {
if t == nil {
return nil, false
}
e, ok := t.index[key] e, ok := t.index[key]
return e, ok return e, ok
} }
// Inline reports whether the table was written as an inline table, `{…}`, // Inline reports whether the table was written as an inline table, `{…}`,
// rather than under a header or introduced by a dotted key. // rather than under a header or introduced by a dotted key.
func (t *Table) Inline() bool { return t.inline } func (t *Table) Inline() bool { return t != nil && t.inline }
// Comments returns the comment lines above the table's header, or above the // Comments returns the comment lines above the table's header, or above the
// key that introduced it. Lines carry no leading '#' and no surrounding space. // key that introduced it. Lines carry no leading '#' and no surrounding space.
func (t *Table) Comments() []string { return t.comments } func (t *Table) Comments() []string {
if t == nil {
return nil
}
return t.comments
}
// SetComments replaces those lines. Each line is written back with a "# " in // SetComments replaces those lines. Each line is written back with a "# " in
// front of it, so a line should not carry one. // front of it, so a line should not carry one.
func (t *Table) SetComments(lines []string) { t.comments = lines } func (t *Table) SetComments(lines []string) {
if t == nil {
return
}
t.comments = lines
}
// Trailing returns the comment on the header's own line, without the '#'. // Trailing returns the comment on the header's own line, without the '#'.
func (t *Table) Trailing() string { return t.trailing } func (t *Table) Trailing() string {
if t == nil {
return ""
}
return t.trailing
}
// SetTrailing replaces that comment. // SetTrailing replaces that comment.
func (t *Table) SetTrailing(line string) { t.trailing = line } func (t *Table) SetTrailing(line string) {
if t == nil {
return
}
t.trailing = line
}
// addValue records a key of the table, in written order. // addValue records a key of the table, in written order. The caller gives
// the entry a table node or element nodes when the value has that shape; a
// map value left without a node writes as an inline table.
func (t *Table) addValue(key string, val any, inline bool) *Entry { func (t *Table) addValue(key string, val any, inline bool) *Entry {
e := &Entry{table: t, key: key, inline: inline} e := &Entry{table: t, key: key, inline: inline}
t.entries = append(t.entries, e) t.entries = append(t.entries, e)
t.index[key] = e t.index[key] = e
if node, ok := val.(map[string]any); ok {
e.child = newTable(node)
}
return e return e
} }
@@ -178,6 +235,9 @@ func (t *Table) addElement(key string, values map[string]any) *Table {
// child returns the node of a table-valued key, or nil. // child returns the node of a table-valued key, or nil.
func (t *Table) child(key string) *Table { func (t *Table) child(key string) *Table {
if t == nil {
return nil
}
if e, ok := t.index[key]; ok { if e, ok := t.index[key]; ok {
return e.child return e.child
} }
@@ -239,6 +299,9 @@ func (e *Entry) SetTrailing(line string) { e.trailing = line }
// GetString returns the string the key holds, and whether it holds one. // GetString returns the string the key holds, and whether it holds one.
func (t *Table) GetString(key string) (string, bool) { func (t *Table) GetString(key string) (string, bool) {
if t == nil {
return "", false
}
v, ok := t.values[key] v, ok := t.values[key]
s, ok := v.(string) s, ok := v.(string)
return s, ok return s, ok
@@ -246,6 +309,9 @@ func (t *Table) GetString(key string) (string, bool) {
// GetInt returns the integer the key holds, and whether it holds one. // GetInt returns the integer the key holds, and whether it holds one.
func (t *Table) GetInt(key string) (int64, bool) { func (t *Table) GetInt(key string) (int64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key] v, ok := t.values[key]
i, ok := v.(int64) i, ok := v.(int64)
return i, ok return i, ok
@@ -253,6 +319,9 @@ func (t *Table) GetInt(key string) (int64, bool) {
// GetFloat returns the float the key holds, and whether it holds one. // GetFloat returns the float the key holds, and whether it holds one.
func (t *Table) GetFloat(key string) (float64, bool) { func (t *Table) GetFloat(key string) (float64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key] v, ok := t.values[key]
f, ok := v.(float64) f, ok := v.(float64)
return f, ok return f, ok
@@ -260,6 +329,9 @@ func (t *Table) GetFloat(key string) (float64, bool) {
// GetBool returns the boolean the key holds, and whether it holds one. // GetBool returns the boolean the key holds, and whether it holds one.
func (t *Table) GetBool(key string) (bool, bool) { func (t *Table) GetBool(key string) (bool, bool) {
if t == nil {
return false, false
}
v, ok := t.values[key] v, ok := t.values[key]
b, ok := v.(bool) b, ok := v.(bool)
return b, ok return b, ok
@@ -267,6 +339,9 @@ func (t *Table) GetBool(key string) (bool, bool) {
// GetArray returns the value array the key holds, and whether it holds one. // GetArray returns the value array the key holds, and whether it holds one.
func (t *Table) GetArray(key string) ([]any, bool) { func (t *Table) GetArray(key string) ([]any, bool) {
if t == nil {
return nil, false
}
v, ok := t.values[key] v, ok := t.values[key]
a, ok := v.([]any) a, ok := v.([]any)
return a, ok return a, ok
@@ -282,9 +357,13 @@ func (t *Table) GetTable(key string) (*Table, bool) {
// Set stores value under key. A key the table already has keeps its position // Set stores value under key. A key the table already has keeps its position
// and its comments; a new one joins the end. A value of map[string]any // and its comments; a new one joins the end. A value of map[string]any
// becomes a table node of its own, written under a header like any other // becomes a table node of its own, written under a header like any other
// table; a Go map carries no order, so its keys take sorted order. A value // table, and replaces the node the key held, which belonged to the value the
// key held; a Go map carries no order, so its keys take sorted order. A value
// of []map[string]any becomes an array-of-tables node. // of []map[string]any becomes an array-of-tables node.
func (t *Table) Set(key string, value any) { func (t *Table) Set(key string, value any) {
if t == nil {
return
}
e, ok := t.index[key] e, ok := t.index[key]
if !ok { if !ok {
t.values[key] = value t.values[key] = value
@@ -303,11 +382,10 @@ func (t *Table) Set(key string, value any) {
t.values[key] = value t.values[key] = value
switch v := value.(type) { switch v := value.(type) {
case map[string]any: case map[string]any:
if e.child == nil { // The node is rebuilt rather than patched: the entries and the index
// belong to the table the key held, and writing the new value
// through them would leave the old table's keys in the output.
e.child = newOrderedTable(v) e.child = newOrderedTable(v)
} else {
e.child.values = v
}
e.elements = nil e.elements = nil
case []map[string]any: case []map[string]any:
e.child = nil e.child = nil
@@ -322,15 +400,30 @@ func (t *Table) Set(key string, value any) {
} }
// newOrderedTable builds a table node for a value the caller set, its keys // newOrderedTable builds a table node for a value the caller set, its keys
// entered as entries in sorted order, the order Marshal writes maps in. // entered in sorted order, the order Marshal writes maps in.
func newOrderedTable(m map[string]any) *Table { func newOrderedTable(m map[string]any) *Table {
return orderedTable(m, 0)
}
// orderedTable is newOrderedTable's recursion. The depth bound is the value
// encoder's: a cyclic map stopped here is written by the value writer, which
// reports it instead of running the stack out.
func orderedTable(m map[string]any, depth int) *Table {
t := newTable(m) t := newTable(m)
for _, k := range slices.Sorted(maps.Keys(m)) { for _, k := range slices.Sorted(maps.Keys(m)) {
v := m[k] v := m[k]
_, isMap := v.(map[string]any)
e := t.addValue(k, v, false) e := t.addValue(k, v, false)
if isMap { if depth >= maxEncodeDepth {
e.child = newOrderedTable(v.(map[string]any)) continue
}
switch val := v.(type) {
case map[string]any:
e.child = orderedTable(val, depth+1)
case []map[string]any:
e.elements = make([]*Table, len(val))
for i, item := range val {
e.elements[i] = orderedTable(item, depth+1)
}
} }
} }
return t return t
@@ -338,6 +431,9 @@ func newOrderedTable(m map[string]any) *Table {
// Delete removes key and everything it holds. // Delete removes key and everything it holds.
func (t *Table) Delete(key string) { func (t *Table) Delete(key string) {
if t == nil {
return
}
if _, ok := t.values[key]; !ok { if _, ok := t.values[key]; !ok {
return return
} }
+141
View File
@@ -4,6 +4,7 @@
package interpres package interpres
import ( import (
"reflect"
"slices" "slices"
"strings" "strings"
"testing" "testing"
@@ -416,3 +417,143 @@ func TestDocumentEditPipeline(t *testing.T) {
} }
}) })
} }
// TestMarshalDocumentRoundTrips pins that a parsed document written back
// re-parses to the same tree: arrays of tables keep exactly one header per
// element, dotted keys hold their line position without swallowing the keys
// after them, inline tables inside value arrays keep their written order,
// and comments travel with their statements.
func TestMarshalDocumentRoundTrips(t *testing.T) {
tests := []struct {
name string
src string
}{
{"array of tables", "[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n"},
{"array of tables with comments", "# about items\n[[items]] # first\nname = \"a\"\n"},
{"dotted key before a later key", "a.b = 1\nc = 2\n"},
{"dotted keys grouped", "a.b = 1\na.c = 2\nd = 3\n"},
{"dotted key with a nested leaf", "a.b.c = 1\nz = 2\n"},
{"header section after a dotted key", "a.b = 1\n\n[a.x]\ny = 2\n"},
{"inline tables in a value array keep order", "arr = [{y = 1, x = 2}, {second = true, first = false}]\n"},
{"nested array of tables", "[[items]]\nn = 1\n\n[items.sub]\nk = \"v\"\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
doc, err := Parse([]byte(tt.src))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
reparsed, err := Parse(out)
if err != nil {
t.Fatalf("re-parse of %q: %v", out, err)
}
if !reflect.DeepEqual(doc.Map(), reparsed.Map()) {
t.Errorf("round trip changed the tree:\nin: %#v\nout: %#v", doc.Map(), reparsed.Map())
}
if got, want := reparsed.Root().Keys(), doc.Root().Keys(); !slices.Equal(got, want) {
t.Errorf("root keys = %v, want %v", got, want)
}
})
}
}
// TestMarshalDocumentArrayComments pins where the comments of an array of
// tables land: above and beside the [[header]] itself.
func TestMarshalDocumentArrayComments(t *testing.T) {
doc, err := Parse([]byte("# element one\n[[items]] # trailing\nname = \"a\"\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "# element one\n[[items]] # trailing\nname = \"a\"\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetReplacesTableNode pins that Set over a key holding a table
// rebuilds the node, so the new map's keys are the ones written.
func TestTableSetReplacesTableNode(t *testing.T) {
doc, err := Parse([]byte("[cache]\nz = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("cache", map[string]any{"a": int64(2)})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "[cache]\na = 2\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetNestedArraysOfTables pins that a value set through the edit API
// carries its arrays of tables into the header form.
func TestTableSetNestedArraysOfTables(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("t", map[string]any{"items": []map[string]any{{"n": int64(1)}, {"n": int64(2)}}})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if !strings.Contains(string(out), "[[t.items]]") {
t.Errorf("output = %q, want the array of tables under a header", out)
}
}
// TestTableSetCyclicMapErrors pins that a cyclic map set through the edit API
// reaches the depth limit instead of the stack.
func TestTableSetCyclicMapErrors(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
m := map[string]any{}
m["self"] = m
doc.Set("cyclic", m)
if _, err := Marshal(doc); err == nil || !strings.Contains(err.Error(), "nests deeper") {
t.Errorf("err = %v, want the depth-limit complaint", err)
}
}
// TestDocumentNilSafety pins that the nil document answers its readers
// instead of panicking, the contract Root already carries.
func TestDocumentNilSafety(t *testing.T) {
var doc *Document
if doc.Map() != nil {
t.Errorf("Map = %v", doc.Map())
}
if doc.Footer() != nil {
t.Errorf("Footer = %v", doc.Footer())
}
doc.SetFooter([]string{"x"})
if e, ok := doc.Get("k"); e != nil || ok {
t.Errorf("Get = %v, %v", e, ok)
}
if _, ok := doc.GetString("k"); ok {
t.Error("GetString on a nil document reports a value")
}
if _, ok := doc.GetTable("k"); ok {
t.Error("GetTable on a nil document reports a value")
}
doc.Set("k", 1)
doc.Delete("k")
if keys := doc.Root().Keys(); keys != nil {
t.Errorf("Keys = %v", keys)
}
if doc.Root().Entries() != nil {
t.Errorf("Entries = %v", doc.Root().Entries())
}
}
+168 -22
View File
@@ -43,8 +43,8 @@ func (e *encoder) writeDocument(doc *Document) error {
} }
// writeDocumentFooter writes the comment lines that follow the last // writeDocumentFooter writes the comment lines that follow the last
// statement, each separated from it by a blank line, the shape the parser // statement. The parser collects them wherever they sit after it, so the
// reads them back from. // writer needs no blank line of its own to have them read back.
func (e *encoder) writeDocumentFooter(footer []string) { func (e *encoder) writeDocumentFooter(footer []string) {
for _, line := range footer { for _, line := range footer {
e.buf.WriteString("# ") e.buf.WriteString("# ")
@@ -53,8 +53,9 @@ func (e *encoder) writeDocumentFooter(footer []string) {
} }
} }
// writeTableEntries writes one table's entries in written order at the given // writeTableEntries writes one table at the given header path, nil for the
// header path, nil for the document root, whose keys need no header. // document root, whose keys need no header: the blank line, the comments,
// the header line with its trailing comment, then the body.
func (e *encoder) writeTableEntries(t *Table, path []string) error { func (e *encoder) writeTableEntries(t *Table, path []string) error {
if t == nil { if t == nil {
return nil return nil
@@ -66,45 +67,169 @@ func (e *encoder) writeTableEntries(t *Table, path []string) error {
if err := e.writeKeyPath(path); err != nil { if err := e.writeKeyPath(path); err != nil {
return err return err
} }
e.buf.WriteString("]\n") e.buf.WriteString("]")
if tr := t.Trailing(); tr != "" { if tr := t.Trailing(); tr != "" {
e.buf.WriteString(" # ") e.buf.WriteString(" # ")
e.buf.WriteString(tr) e.buf.WriteString(tr)
}
e.buf.WriteByte('\n') e.buf.WriteByte('\n')
} }
return e.writeTableBody(t, path)
}
// writeTableBody writes one table's entries: the value lines first, in
// written order, then the header sections. In a valid document every line at
// one level precedes the headers below it, so the split reorders nothing;
// what it prevents is a table a dotted key introduced, which the parse nests
// as a sub-table at the position of a line, from swallowing the lines that
// follow it into its header.
func (e *encoder) writeTableBody(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if !e.isLineEntry(entry) {
continue
}
if err := e.writeLineEntry(entry, path); err != nil {
return err
}
} }
for _, entry := range t.Entries() { for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil { if err := e.checkCtx(); err != nil {
return err return err
} }
if _, isTables := entry.Value().([]map[string]any); isTables && len(entry.Elements()) > 0 { if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
for i, el := range entry.Elements() { // A dotted table writes as lines above; its own header-form
// sub-tables are sections the document placed after those lines,
// so the section pass reaches through the dotted entry.
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeDottedSections writes the header-form sub-tables of a dotted table:
// the sections the document placed after the dotted lines, reached through
// the dotted entry itself.
func (e *encoder) writeDottedSections(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeSectionEntry writes one entry the line pass left behind: a table or
// an array of tables under its header, at the path this level carries.
func (e *encoder) writeSectionEntry(entry *Entry, path []string) error {
if _, isTables := entry.Value().([]map[string]any); isTables {
// An array of tables keeps its header form, one element per header
// with the element's own comments above it; the body that follows is
// the element's, with no header of its own to repeat.
elemPath := append(append([]string{}, path...), entry.Key()) elemPath := append(append([]string{}, path...), entry.Key())
for i, el := range entry.Elements() {
e.writeBlankLine() e.writeBlankLine()
if i == 0 { if i == 0 {
e.writeComments(entry.Comments()) e.writeComments(entry.Comments())
} }
e.writeComments(el.Comments())
e.buf.WriteString("[[") e.buf.WriteString("[[")
if err := e.writeKeyPath(elemPath); err != nil { if err := e.writeKeyPath(elemPath); err != nil {
return err return err
} }
e.buf.WriteString("]]\n") e.buf.WriteString("]]")
if err := e.writeTableEntries(el, elemPath); err != nil { if tr := el.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
if err := e.writeTableBody(el, elemPath); err != nil {
return err return err
} }
} }
continue return nil
} }
if child := entry.Table(); child != nil && !entry.Inline() {
headerPath := append(append([]string{}, path...), entry.Key()) headerPath := append(append([]string{}, path...), entry.Key())
if err := e.writeTableEntries(child, headerPath); err != nil { return e.writeTableEntries(entry.Table(), headerPath)
}
// isLineEntry reports whether an entry writes as one or more "key = value"
// lines at its own level: a value, an inline table, or a table a dotted key
// introduced, which goes back as dotted keys. An emptied array of tables
// counts as one only so the line pass can drop it, the omission the value
// encoder applies to an empty array of tables too.
func (e *encoder) isLineEntry(entry *Entry) bool {
if child := entry.Table(); child != nil {
return entry.Inline() || child.dotted
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return len(entry.Elements()) == 0
}
return true
}
// writeLineEntry writes one entry as lines at this level, and drops an
// emptied array of tables, which has no TOML form.
func (e *encoder) writeLineEntry(entry *Entry, path []string) error {
if child := entry.Table(); child != nil && !entry.Inline() {
return e.writeDottedTable(child, append(append([]string{}, path...), entry.Key()))
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return nil
}
return e.writeDocumentEntry(entry)
}
// writeDottedTable writes a table a dotted key introduced as one dotted line
// per leaf, in written order: `a.b = 1`. A sub-table the document added
// under a header stays a section and is left to the section pass.
func (e *encoder) writeDottedTable(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && !entry.Inline() && !child.dotted {
continue
}
leafPath := append(append([]string{}, path...), entry.Key())
if child := entry.Table(); child != nil && !entry.Inline() {
if err := e.writeDottedTable(child, leafPath); err != nil {
return err return err
} }
continue continue
} }
if err := e.writeDocumentEntry(entry, path); err != nil { e.writeComments(entry.Comments())
if err := e.writeKeyPath(leafPath); err != nil {
return err return err
} }
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
} }
return nil return nil
} }
@@ -112,31 +237,52 @@ func (e *encoder) writeTableEntries(t *Table, path []string) error {
// writeDocumentEntry writes one "key = value" line of a document, with the // writeDocumentEntry writes one "key = value" line of a document, with the
// comments the key carried. A value that is itself an inline table renders // comments the key carried. A value that is itself an inline table renders
// inline from its node, in the written order. // inline from its node, in the written order.
func (e *encoder) writeDocumentEntry(entry *Entry, path []string) error { func (e *encoder) writeDocumentEntry(entry *Entry) error {
e.writeComments(entry.Comments()) e.writeComments(entry.Comments())
if err := e.writeKey(entry.Key()); err != nil { if err := e.writeKey(entry.Key()); err != nil {
return err return err
} }
e.buf.WriteString(" = ") e.buf.WriteString(" = ")
if child := entry.Table(); child != nil { if err := e.writeEntryValueNodes(entry); err != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err return err
} }
if tr := entry.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n') e.buf.WriteByte('\n')
return nil return nil
} }
if err := e.writeValue(entry.Value()); err != nil {
// writeEntryValueNodes writes the value of a document entry. An inline table
// node keeps the written key order even inside a value array, where the
// ordinary value writer would sort the keys.
func (e *encoder) writeEntryValueNodes(entry *Entry) error {
if child := entry.Table(); child != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err
}
} else if arr, ok := entry.Value().([]any); ok {
elems := entry.Elements()
e.buf.WriteByte('[')
for i, item := range arr {
if i > 0 {
e.buf.WriteString(", ")
}
if i < len(elems) && elems[i] != nil {
if err := e.writeInlineTableNode(elems[i]); err != nil {
return err
}
continue
}
if err := e.writeValue(item); err != nil {
return err
}
}
e.buf.WriteByte(']')
} else if err := e.writeValue(entry.Value()); err != nil {
return err return err
} }
if tr := entry.Trailing(); tr != "" { if tr := entry.Trailing(); tr != "" {
e.buf.WriteString(" # ") e.buf.WriteString(" # ")
e.buf.WriteString(tr) e.buf.WriteString(tr)
} }
e.buf.WriteByte('\n')
return nil return nil
} }
+93 -135
View File
@@ -132,6 +132,10 @@ type encoder struct {
// their indentation. // their indentation.
inlineDepth int inlineDepth int
// valueDepth is the nesting level of boxed containers the value writer
// walks, the bound a cyclic map or slice hits instead of the stack.
valueDepth int
// limit is the column at which an inline table is broken; only a // limit is the column at which an inline table is broken; only a
// measuring encoder raises it. // measuring encoder raises it.
limit int limit int
@@ -950,9 +954,11 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, path encPath, for
return nil return nil
} }
// writeArrayValue writes a value array from its reflect value, its elements // writeArrayValue writes a value array from its reflect value. Only the
// written one by one, each falling back to the boxed path only where the // plain scalar kinds reach it: addArrayValue's direct path takes nothing but
// boxed rules rewrite it. // arrays whose elements are scalar kinds without methods, so one writer per
// element needs no boxing branch and no depth walk of its own; the slices
// and maps the boxed rules re-write travel the normaliseValue route.
func (e *encoder) writeArrayValue(v reflect.Value, depth int) error { func (e *encoder) writeArrayValue(v reflect.Value, depth int) error {
if atDepthLimit(depth) { if atDepthLimit(depth) {
return errDepthLimit() return errDepthLimit()
@@ -962,7 +968,7 @@ func (e *encoder) writeArrayValue(v reflect.Value, depth int) error {
if i > 0 { if i > 0 {
e.buf.WriteString(", ") e.buf.WriteString(", ")
} }
if err := e.writeArrayElem(v.Index(i), depth+1); err != nil { if err := e.writeScalarValue(v.Index(i)); err != nil {
return err return err
} }
} }
@@ -970,129 +976,6 @@ func (e *encoder) writeArrayValue(v reflect.Value, depth int) error {
return nil return nil
} }
// writeArrayElem writes one element of a value array. The kinds the boxed
// rules rewrite are handed to normaliseValue and the boxed writer; the rest
// write directly, including nested arrays and inline tables.
func (e *encoder) writeArrayElem(v reflect.Value, depth int) error {
if v.Kind() == reflect.Interface {
if v.IsNil() {
return fmt.Errorf("interpres: cannot encode nil value")
}
v = v.Elem()
}
switch v.Kind() {
case reflect.Slice, reflect.Array:
return e.writeArrayValue(v, depth)
case reflect.Map:
return e.writeInlineMapFromReflect(v, depth)
}
if _, isMarshaler := marshalerOf(v); isMarshaler {
return e.writeNormalisedElem(v, depth)
}
if _, isText, err := textValue(v); err != nil || isText {
if err != nil {
return err
}
return e.writeNormalisedElem(v, depth)
}
switch v.Kind() {
case reflect.String, reflect.Bool, reflect.Int, reflect.Int8, reflect.Int16,
reflect.Int32, reflect.Int64, reflect.Uint, reflect.Uint8, reflect.Uint16,
reflect.Uint32, reflect.Uint64, reflect.Float32, reflect.Float64:
if t := v.Type(); t != durationType && t != numberType {
return e.writeScalarValue(v)
}
}
return e.writeNormalisedElem(v, depth)
}
// writeNormalisedElem normalises one element through the boxed rules and
// writes the result.
func (e *encoder) writeNormalisedElem(v reflect.Value, depth int) error {
val, err := normaliseValueAt(v, depth)
if err != nil {
return err
}
return e.writeValue(val)
}
// writeInlineMapFromReflect renders a map from its reflect value as an inline
// table, the shape the boxed path gives the table elements of a value array:
// sorted keys, single line when it fits, across lines when it does not.
func (e *encoder) writeInlineMapFromReflect(v reflect.Value, depth int) error {
if e.limit >= noInlineBreak {
return e.writeInlineMapFlatReflect(v, depth)
}
flat := e.flat()
err := flat.writeInlineMapFlatReflect(v, depth)
if err != nil {
flat.release()
return err
}
fits := e.column()+flat.buf.Len() <= e.limit
if fits {
e.buf.Write(flat.buf.Bytes())
}
flat.release()
if fits {
return nil
}
return e.writeInlineMapMultilineReflect(v, depth)
}
// writeInlineMapFlatReflect renders the single-line form.
func (e *encoder) writeInlineMapFlatReflect(v reflect.Value, depth int) error {
if v.Type().Key().Kind() != reflect.String {
return fmt.Errorf("map key must be string, got %s", v.Type().Key())
}
keys := make([]string, 0, v.Len())
for _, k := range v.MapKeys() {
keys = append(keys, k.String())
}
slices.Sort(keys)
e.buf.WriteByte('{')
for i, k := range keys {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeKey(k); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeArrayElem(v.MapIndex(reflect.ValueOf(k)), depth+1); err != nil {
return err
}
}
e.buf.WriteByte('}')
return nil
}
// writeInlineMapMultilineReflect renders the across-lines form.
func (e *encoder) writeInlineMapMultilineReflect(v reflect.Value, depth int) error {
keys := make([]string, 0, v.Len())
for _, k := range v.MapKeys() {
keys = append(keys, k.String())
}
slices.Sort(keys)
e.buf.WriteString("{\n")
e.inlineDepth++
for _, k := range keys {
e.writeInlineIndent()
if err := e.writeKey(k); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeArrayElem(v.MapIndex(reflect.ValueOf(k)), depth+1); err != nil {
return err
}
e.buf.WriteString(",\n")
}
e.inlineDepth--
e.writeInlineIndent()
e.buf.WriteByte('}')
return nil
}
// marshalerOf finds the Marshaler a value carries: on the value itself, or on // marshalerOf finds the Marshaler a value carries: on the value itself, or on
// its address, so a pointer-receiver MarshalTOML is found on an addressable // its address, so a pointer-receiver MarshalTOML is found on an addressable
// struct field or slice element, exactly as textMarshalerOf finds MarshalText. // struct field or slice element, exactly as textMarshalerOf finds MarshalText.
@@ -1364,7 +1247,13 @@ func textMarshalerOf(v reflect.Value) (encoding.TextMarshaler, bool) {
return nil, false return nil, false
} }
// isTableElementType reports whether a slice of t is an array of tables. A
// pointer element is looked through, so []*T behaves as []T: the empty-slice
// decision and the header form agree on the same element type.
func isTableElementType(t reflect.Type) bool { func isTableElementType(t reflect.Type) bool {
for t.Kind() == reflect.Pointer {
t = t.Elem()
}
switch t.Kind() { switch t.Kind() {
case reflect.Struct: case reflect.Struct:
return !isScalarStruct(t) && !isTextMarshalerType(t) return !isScalarStruct(t) && !isTextMarshalerType(t)
@@ -1395,12 +1284,24 @@ func (e *encoder) writeBlankLine() {
} }
func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error { func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
// The walk that built the document checked the context on its own
// cadence; the emission of a large document is long enough to need the
// same checks, or a cancellation that lands after the walk would wait a
// full document out.
if err := e.checkCtx(); err != nil {
return err
}
if e.opts.layout == LayoutKindGrouped { if e.opts.layout == LayoutKindGrouped {
// Scalars first, then inline sub-tables as value lines, then the // Scalars first, then inline sub-tables as value lines, then the
// remaining tables as headers, then arrays of tables. Each pass walks // remaining tables as headers, then arrays of tables. Each pass walks
// the entries in place; grouping copies of them cost the encoder a // the entries in place; grouping copies of them cost the encoder a
// third of its allocations for nothing. // third of its allocations for nothing.
for i := range doc.entries { for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
kv := &doc.entries[i] kv := &doc.entries[i]
if kv.kind != entryScalar { if kv.kind != entryScalar {
continue continue
@@ -1413,6 +1314,11 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
// header of this document: a line written after a [header] would be // header of this document: a line written after a [header] would be
// read back as part of that table. // read back as part of that table.
for i := range doc.entries { for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
t := &doc.entries[i] t := &doc.entries[i]
if t.kind != entryTable { if t.kind != entryTable {
continue continue
@@ -1424,6 +1330,11 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
t.emitted = inlined t.emitted = inlined
} }
for i := range doc.entries { for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
t := &doc.entries[i] t := &doc.entries[i]
if t.kind != entryTable || t.emitted { if t.kind != entryTable || t.emitted {
continue continue
@@ -1441,12 +1352,22 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
} }
} }
for i := range doc.entries { for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
a := &doc.entries[i] a := &doc.entries[i]
if a.kind != entryArray { if a.kind != entryArray {
continue continue
} }
path := append(append([]string{}, prefix...), a.key) path := append(append([]string{}, prefix...), a.key)
for j, sub := range a.docs { for j, sub := range a.docs {
if j%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
e.writeBlankLine() e.writeBlankLine()
if j == 0 { if j == 0 {
e.writeComments(a.comments) e.writeComments(a.comments)
@@ -1469,7 +1390,12 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
// own section content; the emitter still writes sub-documents as separate // own section content; the emitter still writes sub-documents as separate
// nested blocks, so a "" sub-keyed scalar following a header for the same // nested blocks, so a "" sub-keyed scalar following a header for the same
// section is impossible in practice (struct fields are visited in order). // section is impossible in practice (struct fields are visited in order).
for _, ent := range doc.entries { for i, ent := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
switch ent.kind { switch ent.kind {
case entryScalar: case entryScalar:
if err := e.writeKV(&ent); err != nil { if err := e.writeKV(&ent); err != nil {
@@ -1699,9 +1625,15 @@ func (e *encoder) writeValue(val any) error {
case float64: case float64:
return e.writeFloat(v) return e.writeFloat(v)
case time.Time: case time.Time:
if err := wholeMinuteOffset(v); err != nil {
return err
}
e.buf.WriteString(offsetString(v)) e.buf.WriteString(offsetString(v))
return nil return nil
case OffsetDateTime: case OffsetDateTime:
if err := wholeMinuteOffset(v.Time); err != nil {
return err
}
e.buf.WriteString(v.String()) e.buf.WriteString(v.String())
return nil return nil
case LocalDateTime: case LocalDateTime:
@@ -1714,6 +1646,39 @@ func (e *encoder) writeValue(val any) error {
e.buf.WriteString(v.String()) e.buf.WriteString(v.String())
return nil return nil
case []any: case []any:
if err := e.enterValueDepth(); err != nil {
return err
}
err := e.writeValueArray(v)
e.valueDepth--
return err
case map[string]any:
if err := e.enterValueDepth(); err != nil {
return err
}
err := e.writeInlineMap(v)
e.valueDepth--
return err
case nil:
return fmt.Errorf("interpres: cannot encode nil value")
default:
return fmt.Errorf("interpres: cannot encode %T", val)
}
}
// enterValueDepth counts one level of boxed container nesting. The
// reflection walk has its own bound, but a tree built by hand and written
// through the boxed path carries no walk, so the writer bounds itself.
func (e *encoder) enterValueDepth() error {
e.valueDepth++
if e.valueDepth > maxEncodeDepth {
return errDepthLimit()
}
return nil
}
// writeValueArray writes a boxed value array, one writeValue per element.
func (e *encoder) writeValueArray(v []any) error {
e.buf.WriteByte('[') e.buf.WriteByte('[')
for i, item := range v { for i, item := range v {
if i > 0 { if i > 0 {
@@ -1725,13 +1690,6 @@ func (e *encoder) writeValue(val any) error {
} }
e.buf.WriteByte(']') e.buf.WriteByte(']')
return nil return nil
case map[string]any:
return e.writeInlineMap(v)
case nil:
return fmt.Errorf("interpres: cannot encode nil value")
default:
return fmt.Errorf("interpres: cannot encode %T", val)
}
} }
// writeInlineMap renders m as a TOML inline table, on one line when it fits // writeInlineMap renders m as a TOML inline table, on one line when it fits
+113 -1
View File
@@ -67,7 +67,7 @@ func TestMarshalFloatSpecials(t *testing.T) {
} }
} }
func TestMarshalFloatNormalizesNegativeZero(t *testing.T) { func TestMarshalFloatNormalisesNegativeZero(t *testing.T) {
// The output contract normalises negative zero to "0.0". // The output contract normalises negative zero to "0.0".
type Cfg struct { type Cfg struct {
Z float64 `toml:"z"` Z float64 `toml:"z"`
@@ -2284,3 +2284,115 @@ func TestEmitFieldComments(t *testing.T) {
} }
}) })
} }
// errWriter fails every write with a fixed error.
type errWriter struct{ err error }
func (w errWriter) Write([]byte) (int, error) { return 0, w.err }
// TestMarshalWrite covers the streaming entry: the happy path with options
// and a failing writer.
func TestMarshalWrite(t *testing.T) {
var buf bytes.Buffer
err := MarshalWrite(&buf, map[string]any{"b": 2, "a": 1})
if err != nil {
t.Fatalf("MarshalWrite: %v", err)
}
// A map carries no order, so the writer uses the sorted one.
if buf.String() != "a = 1\nb = 2\n" {
t.Errorf("output = %q", buf.String())
}
writeErr := errors.New("boom")
if err := MarshalWrite(errWriter{writeErr}, map[string]any{"a": 1}); !errors.Is(err, writeErr) {
t.Errorf("err = %v, want the write error wrapped", err)
}
}
// TestMarshalRejectsUnsupportedKinds pins the clear error a field of a kind
// TOML cannot carry raises, through the struct walk.
func TestMarshalRejectsUnsupportedKinds(t *testing.T) {
tests := []struct {
name string
value any
}{
{"func", struct {
F func() `toml:"f"`
}{}},
{"chan", struct {
C chan int `toml:"c"`
}{}},
{"complex", struct {
Z complex128 `toml:"z"`
}{}},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
_, err := Marshal(tt.value)
if err == nil {
t.Fatalf("Marshal accepted %#v", tt.value)
}
if !strings.Contains(err.Error(), "cannot encode") {
t.Errorf("err = %v, want the cannot-encode complaint", err)
}
})
}
}
// TestMarshalOmitsEmptyPointerTableSlice pins that an empty slice of pointer
// tables is omitted, the rule its non-pointer form already follows.
func TestMarshalOmitsEmptyPointerTableSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
out, err := Marshal(struct {
Items []*item `toml:"items"`
}{})
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if len(out) != 0 {
t.Errorf("output = %q, want the empty array of tables omitted", out)
}
}
// TestMarshalRejectsNonWholeMinuteOffset pins that a zone offset carrying
// seconds is refused instead of silently losing them.
func TestMarshalRejectsNonWholeMinuteOffset(t *testing.T) {
z := time.FixedZone("", 57*60+44)
_, err := Marshal(struct {
Stamp time.Time `toml:"stamp"`
}{Stamp: time.Date(1890, 1, 1, 12, 0, 0, 0, z)})
if err == nil || !strings.Contains(err.Error(), "not a whole number of minutes") {
t.Errorf("err = %v, want the whole-minute offset complaint", err)
}
_, err = Marshal(struct {
Stamp OffsetDateTime `toml:"stamp"`
}{Stamp: OffsetDateTime{time.Date(1890, 1, 1, 12, 0, 0, 0, z)}})
if err == nil || !strings.Contains(err.Error(), "not a whole number of minutes") {
t.Errorf("err = %v, want the whole-minute offset complaint for the wrapper", err)
}
}
// cancelOnMarshal cancels the context the encode runs under, the moment its
// method is called, so the emission that follows is already past the walk's
// own checks.
type cancelOnMarshal struct {
cancel context.CancelFunc
}
func (c cancelOnMarshal) MarshalTOML() (any, error) {
c.cancel()
return int64(1), nil
}
// TestMarshalContextCancelsDuringEmission pins that a context cancelled
// between the walk and the emission stops the encode instead of writing the
// whole document out.
func TestMarshalContextCancelsDuringEmission(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
value := map[string]any{"k": cancelOnMarshal{cancel}}
if _, err := MarshalContext(ctx, value); !errors.Is(err, context.Canceled) {
t.Errorf("err = %v, want the cancellation", err)
}
}
+8 -7
View File
@@ -3,8 +3,8 @@
// Command basic demonstrates decoding and encoding a TOML document with // Command basic demonstrates decoding and encoding a TOML document with
// interpres. It exercises struct mapping, arrays of tables, Marshaler // interpres. It exercises struct mapping, arrays of tables, Marshaler
// customisation, the Decoder's strict mode, and the Encoder's policy // customisation, and the Encoder's policy options, covering every feature a
// options, covering every feature a regular user would reach for. // regular user would reach for.
package main package main
import ( import (
@@ -18,8 +18,7 @@ import (
) )
// document is a small but realistic configuration: it has scalars, a // document is a small but realistic configuration: it has scalars, a
// sub-table, an array of tables, and a date-time. We pick a 32-bit port so // sub-table, an array of tables, and a date-time.
// the demonstration also covers overflow-safe integer conversion.
const document = ` const document = `
title = "interpres demo" title = "interpres demo"
launched = 2024-11-04T09:00:00Z launched = 2024-11-04T09:00:00Z
@@ -62,9 +61,11 @@ type User struct {
Admin bool `toml:"admin"` Admin bool `toml:"admin"`
} }
// Port is a typed alias that controls how its value appears in TOML. The // Port is a typed string alias that carries a Marshaler. The MarshalTOML
// MarshalTOML hook returns a string, so a Port field is rendered as // hook returns the string unchanged, so a Port field is rendered as the
// "host:port" instead of the raw integer. // string it holds, a string the encoding would print the same way without
// the hook; the demonstration that a Marshaler reshapes a value is
// Endpoint's below.
type Port string type Port string
func (p Port) MarshalTOML() (any, error) { func (p Port) MarshalTOML() (any, error) {
+72 -50
View File
@@ -16,8 +16,10 @@
// doc, err := interpres.Parse(data) // doc, err := interpres.Parse(data)
// tree := doc.Map() // tree := doc.Map()
// //
// A Decoder allows strict decoding that rejects keys without a matching // Strict decoding that rejects keys without a matching struct field is an
// struct field, mirroring the RejectUnknownFields option of encoding/json/v2. // option, mirroring the RejectUnknownMembers option of encoding/json/v2:
//
// err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
package interpres package interpres
import ( import (
@@ -212,7 +214,7 @@ func ParseFile(path string) (*Document, error) {
// Valid reports whether data is a valid TOML document: nil when the parser // Valid reports whether data is a valid TOML document: nil when the parser
// accepts it, and the parse error when it does not. It is the library call // accepts it, and the parse error when it does not. It is the library call
// the -validate mode of interpres-decode is built on, and it reads nothing // the --validate mode of interpres-decode is built on, and it reads nothing
// but the bytes it is given. // but the bytes it is given.
func Valid(data []byte) error { func Valid(data []byte) error {
_, err := ParseMapContext(context.Background(), data) _, err := ParseMapContext(context.Background(), data)
@@ -256,9 +258,6 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
return tree, &Document{root: p.doc, footer: p.footer}, nil return tree, &Document{root: p.doc, footer: p.footer}, nil
} }
// Unmarshal parses a TOML document and stores the result in the value pointed
// to by v. v is typically a pointer to a struct or to a map[string]any.
//
// Unmarshal parses a TOML document and stores the result in the value pointed // Unmarshal parses a TOML document and stores the result in the value pointed
// to by v. v is typically a pointer to a struct or to a map[string]any. // to by v. v is typically a pointer to a struct or to a map[string]any.
// Options tune the call; with none, unknown keys are ignored, numbers are // Options tune the call; with none, unknown keys are ignored, numbers are
@@ -266,7 +265,9 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
// //
// Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a // Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a
// case-insensitive match on the field name when no tag is present. A tag of // case-insensitive match on the field name when no tag is present. A tag of
// "-" skips the field. // "-" skips the field. Two document keys that differ only in case and both
// match one field resolve deterministically: the lexicographically greater
// one wins, the same key winning every run.
// //
// A destination implementing Unmarshaler receives the parsed value as it is, // A destination implementing Unmarshaler receives the parsed value as it is,
// a TOML string fills a destination implementing encoding.TextUnmarshaler, and // a TOML string fills a destination implementing encoding.TextUnmarshaler, and
@@ -281,12 +282,12 @@ func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error {
// ParseAs decodes a TOML document into T in one call, the generic shorthand // ParseAs decodes a TOML document into T in one call, the generic shorthand
// for Unmarshal with a destination variable: // for Unmarshal with a destination variable:
// //
// cfg, err := interpres.ParseAs[Config](data) // cfg, err := interpres.ParseAs[Config](data, interpres.RejectUnknownFields(true))
// //
// The zero T comes back with the error. // The options are Unmarshal's. The zero T comes back with the error.
func ParseAs[T any](data []byte) (T, error) { func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error) {
var v T var v T
err := Unmarshal(data, &v) err := Unmarshal(data, &v, opts...)
return v, err return v, err
} }
@@ -315,10 +316,19 @@ func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...Unmarshal
// UnmarshalRead reads the document from r and decodes it into v, the // UnmarshalRead reads the document from r and decodes it into v, the
// streaming-shaped entry the json/v2 vocabulary uses. The reader is // streaming-shaped entry the json/v2 vocabulary uses. The reader is
// consumed in full, because the parser scans its source in place; the // consumed in full, because the parser scans its source in place; with
// options and the behaviour are Unmarshal's. // MaxInputSize set, reading stops one byte past the limit so the size the
// option bounds is the memory held, not what a reader is drained into first.
// The options and the behaviour are Unmarshal's.
func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error { func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error {
data, err := io.ReadAll(r) s := settingsFor(opts)
var data []byte
var err error
if s.maxInputSize > 0 {
data, err = io.ReadAll(io.LimitReader(r, int64(s.maxInputSize)+1))
} else {
data, err = io.ReadAll(r)
}
if err != nil { if err != nil {
return fmt.Errorf("interpres: read: %w", err) return fmt.Errorf("interpres: read: %w", err)
} }
@@ -338,18 +348,19 @@ func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error {
// tree path, so every option means the same thing on every document. // tree path, so every option means the same thing on every document.
type UnmarshalOption func(*decodeSettings) type UnmarshalOption func(*decodeSettings)
// decodeSettings is the option carrier of one decode call. // decodeSettings is the option carrier of one decode call. The context is
// not one: it arrives as its own argument, because every entry point names it
// explicitly.
type decodeSettings struct { type decodeSettings struct {
disallowUnknown bool disallowUnknown bool
useNumber bool useNumber bool
maxDepth int maxDepth int
maxInputSize int maxInputSize int
localLoc *time.Location localLoc *time.Location
ctx context.Context
} }
func settingsFor(opts []UnmarshalOption) *decodeSettings { func settingsFor(opts []UnmarshalOption) *decodeSettings {
s := &decodeSettings{ctx: context.Background()} s := &decodeSettings{}
for _, opt := range opts { for _, opt := range opts {
opt(s) opt(s)
} }
@@ -452,8 +463,8 @@ type Marshaler interface {
// argument is whatever the parser produced for that key: one of string, // argument is whatever the parser produced for that key: one of string,
// bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime, // bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime,
// []any, or map[string]any. A tree built by hand may carry a plain time.Time // []any, or map[string]any. A tree built by hand may carry a plain time.Time
// where the parser would put an OffsetDateTime, and a Decoder configured with // where the parser would put an OffsetDateTime, and NumbersAsLiterals a
// UseNumber a Number. // Number.
// //
// UnmarshalTOML may parse, inspect, or transform the value however it likes, // UnmarshalTOML may parse, inspect, or transform the value however it likes,
// then store the result by mutating its receiver through the standard // then store the result by mutating its receiver through the standard
@@ -461,7 +472,7 @@ type Marshaler interface {
// reflect.Value.Set or by reassigning fields through a pointer the receiver // reflect.Value.Set or by reassigning fields through a pointer the receiver
// holds). // holds).
// //
// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the // UnmarshalTOML is invoked from Unmarshal and its siblings when the
// destination type implements the interface. The decoder does not need to // destination type implements the interface. The decoder does not need to
// consult the concrete return value; whatever the receiver stores is kept. // consult the concrete return value; whatever the receiver stores is kept.
// //
@@ -482,16 +493,21 @@ type UnmarshalerContext interface {
} }
// Marshal returns the TOML encoding of v. The output is valid TOML 1.1. // Marshal returns the TOML encoding of v. The output is valid TOML 1.1.
// Options tune the emission; with none, the layout groups entries by kind,
// empty arrays emit and sub-tables take the header form.
// //
// Marshal traverses v using reflection and applies the following rules: // Marshal traverses v using reflection and applies the following rules:
// //
// - The top-level value must be a struct or a map[string]V. Pointers are // - The top-level value must be a struct, a map[string]V or an OrderedMap
// followed; a nil top-level pointer is an error. // (or a non-nil pointer to one). A Document writes itself back, and a nil
// one is an error.
// - Struct fields are matched by `toml:"name"` tag (case-insensitive // - Struct fields are matched by `toml:"name"` tag (case-insensitive
// fallback to field name; `-` skips). The tag options `omitzero` (skip // fallback to field name; `-` skips). The tag option `omitzero` skips a
// the zero value of the field's type) and `omitempty` (skip an empty // field holding the zero value of its type (a type with an IsZero method
// slice, array, or map) drop a field from the output on encode; the // decides through it), and `omitempty` skips a value that is empty in
// decoder ignores them. Anonymous (embedded) fields without a tag are // the encoding/json sense: an empty string, a zero number, false, a nil
// pointer or interface, and an empty slice, array or map. The decoder
// ignores both options. Anonymous (embedded) fields without a tag are
// inlined. // inlined.
// - Maps use sorted keys for deterministic output. // - Maps use sorted keys for deterministic output.
// - Slices and arrays of structs or maps become TOML arrays of tables; a // - Slices and arrays of structs or maps become TOML arrays of tables; a
@@ -502,10 +518,12 @@ type UnmarshalerContext interface {
// inline table. // inline table.
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time // - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
// and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/ // and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/
// LocalTime (local variants). A date-time writes its seconds only when the value carries // LocalTime (local variants). A date-time writes its seconds only when
// them, and drops the trailing zeros of a fractional second. // the value carries them, and drops the trailing zeros of a fractional
// - A table element of a value array, and a sub-table inlined by // second. A zone offset that is not a whole number of minutes is refused,
// Encoder.InlineTables, is written as an inline table, across lines when it // because TOML has no form that carries its seconds.
// - A table element of a value array, and a sub-table the InlineTables
// option inlines, is written as an inline table, across lines when it
// does not fit one. // does not fit one.
// - Values implementing Marshaler are encoded by calling MarshalTOML and // - Values implementing Marshaler are encoded by calling MarshalTOML and
// using its result. // using its result.
@@ -516,14 +534,10 @@ type UnmarshalerContext interface {
// //
// Marshal rejects a value that nests deeper than 10000 levels with an error // Marshal rejects a value that nests deeper than 10000 levels with an error
// naming the limit, so cyclic data is reported instead of running the stack // naming the limit, so cyclic data is reported instead of running the stack
// out. The output is not guaranteed to be byte-identical to // out. The output is not guaranteed to be byte-identical to the input that
// the input that produced v: comments, whitespace, key order (for maps), // produced v: comments, whitespace, key order (for maps), string quoting
// string quoting style, and the choice between `[table]` headers and inline // style, and the choice between `[table]` headers and inline tables are not
// tables are not preserved. // preserved.
//
// Marshal returns the TOML encoding of v. Options tune the emission; with
// none, the layout groups entries by kind, empty arrays emit and sub-tables
// take the header form.
// //
// Marshal is equivalent to MarshalContext with context.Background. // Marshal is equivalent to MarshalContext with context.Background.
func Marshal(v any, opts ...MarshalOption) ([]byte, error) { func Marshal(v any, opts ...MarshalOption) ([]byte, error) {
@@ -548,12 +562,13 @@ type Statement struct {
} }
// Statements reads a TOML document from r and returns an iterator over its // Statements reads a TOML document from r and returns an iterator over its
// top-level statements in written order: key/value statements, a [table] // top-level statements in written order: key/value statements, including a
// header as one statement carrying its Table node, and an [[array of // value that is an array or an inline table, a [table] header as one
// tables]] as one statement per element, each with the element's node and // statement carrying its Table node, and an [[array of tables]] as one
// its Index. Iteration stops at the first error, which arrives as the second // statement per element, each with the element's node and its Index.
// value, and at a false yield: a caller that breaks after the statement it // Iteration stops at the first error, which arrives as the second value, and
// wanted reads no further ones. // at a false yield: a caller that breaks after the statement it wanted reads
// no further ones.
// //
// The reader is consumed in full before the first statement is yielded, // The reader is consumed in full before the first statement is yielded,
// because the parser scans the source in place; processing the yielded // because the parser scans the source in place; processing the yielded
@@ -572,8 +587,12 @@ func Statements(r io.Reader) iter.Seq2[Statement, error] {
return return
} }
for _, e := range doc.Root().Entries() { for _, e := range doc.Root().Entries() {
if els := e.Elements(); len(els) > 0 { // Only an array of tables yields per element, the branch the
for i, el := range els { // write side takes too: a value array is one statement whatever
// its elements, and an emptied array of tables holds no element
// to yield.
if _, isTables := e.Value().([]map[string]any); isTables && len(e.Elements()) > 0 {
for i, el := range e.Elements() {
if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: el, Index: i}, nil) { if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: el, Index: i}, nil) {
return return
} }
@@ -648,14 +667,14 @@ const (
// interpres.InlineTables(60)) // interpres.InlineTables(60))
type MarshalOption func(*encodeSettings) type MarshalOption func(*encodeSettings)
// encodeSettings is the option carrier of one encode call. // encodeSettings is the option carrier of one encode call. As on the decode
// side, the context arrives as its own argument.
type encodeSettings struct { type encodeSettings struct {
ctx context.Context
cfg encodeConfig cfg encodeConfig
} }
func settingsForEncode(opts []MarshalOption) *encodeSettings { func settingsForEncode(opts []MarshalOption) *encodeSettings {
s := &encodeSettings{ctx: context.Background(), cfg: encodeConfig{layout: LayoutKindGrouped}} s := &encodeSettings{cfg: encodeConfig{layout: LayoutKindGrouped}}
for _, opt := range opts { for _, opt := range opts {
opt(s) opt(s)
} }
@@ -681,6 +700,7 @@ func (s *encodeSettings) marshal(ctx context.Context, v any) ([]byte, error) {
// Layout sets the layout the encoder writes a document's entries in: // Layout sets the layout the encoder writes a document's entries in:
// LayoutKindGrouped, the default, reorders them scalars first, then tables, // LayoutKindGrouped, the default, reorders them scalars first, then tables,
// then arrays of tables; LayoutKindDeclaration preserves declaration order. // then arrays of tables; LayoutKindDeclaration preserves declaration order.
// A value the two constants do not name behaves as LayoutKindGrouped.
func Layout(kind LayoutKind) MarshalOption { func Layout(kind LayoutKind) MarshalOption {
return func(s *encodeSettings) { s.cfg.layout = kind } return func(s *encodeSettings) { s.cfg.layout = kind }
} }
@@ -727,7 +747,9 @@ func InlineTables(threshold int) MarshalOption {
// Go doc comments are not visible to reflection, so the tag is the channel // Go doc comments are not visible to reflection, so the tag is the channel
// that carries the text. Off by default, and a field without a `comment=` // that carries the text. Off by default, and a field without a `comment=`
// option prints none. Multi-line comments carry newlines in the tag, each // option prints none. Multi-line comments carry newlines in the tag, each
// line printed with its own "# " marker. // line printed with its own "# " marker. The tag's options separate with
// commas, so the comment text itself cannot carry one; the first comma ends
// it.
func EmitFieldComments(v bool) MarshalOption { func EmitFieldComments(v bool) MarshalOption {
return func(s *encodeSettings) { s.cfg.emitFieldComments = v } return func(s *encodeSettings) { s.cfg.emitFieldComments = v }
} }
+108
View File
@@ -895,3 +895,111 @@ func TestParseCRLFDocument(t *testing.T) {
t.Errorf("tree = %v", tree) t.Errorf("tree = %v", tree)
} }
} }
// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U:
// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the
// greatest scalar value parses.
func TestParseUnicodeEscapeBoundaries(t *testing.T) {
bad := []struct {
name string
in string
}{
{"high surrogate", `a = "\ud800"`},
{"low surrogate", `a = "\udfff"`},
{"past the greatest scalar", `a = "\U00110000"`},
{"signed short escape", `a = "\u+041"`},
{"negative long escape", `a = "\U-0000001"`},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := Parse([]byte(tt.in))
if err == nil {
t.Fatalf("Parse accepted %q", tt.in)
}
})
}
tree, err := ParseMap([]byte("a = \"\\U0010FFFF\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "􏿿" {
t.Errorf("a = %q", tree["a"])
}
}
// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a
// date-time with a component out of range is rejected as a date-time, not
// left to the number decoder's complaint.
func TestParseRejectsOutOfRangeDateTimes(t *testing.T) {
bad := []struct {
name string
in string
}{
{"hour 24", "a = 1979-05-27T24:00:00Z"},
{"minute 60", "a = 1979-05-27T07:60:00Z"},
{"second 60", "a = 1979-05-27T07:32:60Z"},
{"month 13", "a = 1979-13-27T07:32:00Z"},
{"day 32", "a = 1979-05-32T07:32:00Z"},
{"february the thirtieth", "a = 1979-02-30"},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseMap([]byte(tt.in))
if err == nil {
t.Fatalf("ParseMap accepted %q", tt.in)
}
if !strings.Contains(err.Error(), "invalid date-time") {
t.Errorf("err = %v, want the date-time complaint", err)
}
})
}
}
// TestParseMultilineStringEdges pins the carriage-return and delimiter rules
// of multi-line strings: a bare CR right after the opening delimiter is the
// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the
// content survives.
func TestParseMultilineStringEdges(t *testing.T) {
_, err := ParseMap([]byte("a = \"\"\"\rX\"\"\""))
if err == nil || !strings.Contains(err.Error(), "bare carriage return") {
t.Errorf("err = %v, want the bare-CR error after the delimiter", err)
}
tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "X\r\nY" {
t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"])
}
}
// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes
// before the closing delimiter of a basic multi-line string are content, and
// more than five are the error.
func TestParseMultilineBasicDelimiterRuns(t *testing.T) {
tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != `end"` {
t.Errorf("a = %q", tree["a"])
}
_, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\""))
if err == nil || !strings.Contains(err.Error(), "too many") {
t.Errorf("err = %v, want the too-many-delimiters error", err)
}
}
// TestParseLineEndingBackslashEdges pins the line-ending backslash at the
// very end of the input and before a bare CR.
func TestParseLineEndingBackslashEdges(t *testing.T) {
bad := []string{
"a = \"\"\"x \\\\",
"a = \"\"\"x \\\\\rZ\"\"\"",
}
for _, in := range bad {
if _, err := ParseMap([]byte(in)); err == nil {
t.Errorf("ParseMap accepted %q", in)
}
}
}
+3 -3
View File
@@ -32,7 +32,7 @@ test:
close($c); close($c);
die qq{no total line in coverage.out\n} unless defined $total; die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total; printf qq{Total coverage: %s%%\n}, $total;
exit($total < 82 ? 1 : 0); exit($total < 80 ? 1 : 0);
# The same suite under the race detector. The expensive one. # The same suite under the race detector. The expensive one.
race: race:
@@ -94,13 +94,13 @@ dev:
# Runs the official toml-test compliance suite in both directions, decoder and encoder, against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite. # Runs the official toml-test compliance suite in both directions, decoder and encoder, against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
toml-test: build toml-test: build
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1 toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate. # Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
coverage-html: test coverage-html: test
go tool cover -html=coverage.out -o coverage.html go tool cover -html=coverage.out -o coverage.html
# Cross-compile smoke: the library and the command build for the foreign architectures and the browser and edge runtimes; not a gate, it is a nightly convenience and runs std-lib only. # Cross-compile smoke: the library and the command build for the foreign architectures and the browser and edge runtimes; not a gate, it is a hand-run convenience and runs std-lib only.
cross: cross:
GOARCH=arm64 go build ./... GOARCH=arm64 go build ./...
GOARCH=loong64 go build ./... GOARCH=loong64 go build ./...
+34 -32
View File
@@ -1,4 +1,4 @@
.TH INTERPRES-DECODE 1 "2026-09-21" "interpres 2.0.0" "User Commands" .TH INTERPRES-DECODE 1 "2026-09-22" "interpres 2.0.0" "User Commands"
.SH NAME .SH NAME
interpres-decode \- TOML validator and toml-test harness adapter interpres-decode \- TOML validator and toml-test harness adapter
.SH SYNOPSIS .SH SYNOPSIS
@@ -6,38 +6,38 @@ interpres-decode \- TOML validator and toml-test harness adapter
[\fIFLAGS\fR] [\fIFLAGS\fR]
.br .br
.B interpres-decode .B interpres-decode
.B \-encode .B \-\-encode
.br .br
.B interpres-decode .B interpres-decode
.B \-validate .B \-\-validate
[\fIFILE\fR...] [\fIFILE\fR...]
.br .br
.B interpres-decode .B interpres-decode
.B \-validate .B \-\-validate
[\fIDIRECTORY\fR...] [\fIDIRECTORY\fR...]
.br .br
.B interpres-decode .B interpres-decode
.B \-json .B \-\-json
.br .br
.B interpres-decode .B interpres-decode
.B \-struct .B \-\-struct
.br .br
.B interpres-decode .B interpres-decode
.B \-schema .B \-\-schema
\fITYPE\fR \fITYPE\fR
\fIFILE.go\fR \fIFILE.go\fR
.br .br
.B interpres-decode .B interpres-decode
.B \-version .B \-\-version
.SH DESCRIPTION .SH DESCRIPTION
.B interpres-decode .B interpres-decode
is the toml-test harness adapter in both directions and a TOML validator. is the toml-test harness adapter in both directions and a TOML validator.
Without a mode flag it reads one TOML document from standard input and writes Without a mode flag it reads one TOML document from standard input and writes
the toml-test tagged-JSON representation to standard output. the toml-test tagged-JSON representation to standard output.
.B \-encode .B \-\-encode
reads a tagged-JSON description from standard input and writes the TOML reads a tagged-JSON description from standard input and writes the TOML
document it describes. document it describes.
.B \-validate .B \-\-validate
parses each named file, or standard input when none are named, and prints one parses each named file, or standard input when none are named, and prints one
line per invalid document to standard error; a named directory is walked for line per invalid document to standard error; a named directory is walked for
.B .toml .B .toml
@@ -45,48 +45,49 @@ files, every one validated, and the walk closes with a summary on standard
error naming the counts. The name error naming the counts. The name
.B \- .B \-
means standard input. means standard input.
.B \-json .B \-\-json
prints plain indented JSON instead of the tagged form. prints plain indented JSON instead of the tagged form; it shapes the decoding
.B \-struct output only, so it is rejected together with the mode flags.
.B \-\-struct
prints a Go struct definition inferred from the document on standard input. prints a Go struct definition inferred from the document on standard input.
.B \-schema .B \-\-schema
writes a TOML template for the struct type writes a TOML template for the struct type
\fITYPE\fR \fITYPE\fR
declared in the Go source file declared in the Go source file
\fIFILE.go\fR, \fIFILE.go\fR,
taking the key names, comments and defaults from the fields' tags. taking the key names, comments and defaults from the fields' tags.
.B \-version .B \-\-version
prints the binary's version and exits. prints the binary's version and exits.
.PP .PP
The mode flags The mode flags
.BR \-validate , .BR \-\-validate ,
.BR \-encode , .BR \-\-encode ,
.B \-struct .B \-\-struct
and and
.B \-schema .B \-\-schema
cannot be combined. cannot be combined.
.SH OPTIONS .SH OPTIONS
.TP .TP
.B \-validate .B \-\-validate
Validate the documents instead of emitting tagged JSON. Validate the documents instead of emitting tagged JSON.
.TP .TP
.B \-encode .B \-\-encode
Read tagged JSON from standard input and write TOML instead. Read tagged JSON from standard input and write TOML instead.
.TP .TP
.B \-json .B \-\-json
With the default mode, print plain indented JSON instead of tagged JSON. With the default mode, print plain indented JSON instead of tagged JSON.
.TP .TP
.B \-struct .B \-\-struct
Infer a Go struct definition from the document on standard input and print it. Infer a Go struct definition from the document on standard input and print it.
.TP .TP
.BI \-schema " TYPE" .BI \-\-schema " TYPE"
Write a TOML template for the struct type \fITYPE\fR; the Go source file Write a TOML template for the struct type \fITYPE\fR; the Go source file
follows as the first argument. follows as the first argument.
.TP .TP
.B \-version .B \-\-version
Print the version and exit. Print the version and exit.
.TP .TP
.B \-h .B \-\-help
Print the usage. Print the usage.
.SH EXIT STATUS .SH EXIT STATUS
.TP .TP
@@ -95,11 +96,12 @@ The document parsed and the output was written; in validate mode, every
document parsed. document parsed.
.TP .TP
.B 1 .B 1
Adapter: a parse error. Validate: at least one document is invalid. Adapter: a parse error. Validate: at least one document is invalid. Struct:
the document on standard input failed to parse.
.TP .TP
.B 2 .B 2
A usage error, a read failure, malformed tagged JSON, or a value with no TOML A usage error, a read or write failure, malformed tagged JSON, or a value
representation. with no TOML representation.
.SH EXAMPLES .SH EXAMPLES
Decode a document into tagged JSON: Decode a document into tagged JSON:
.PP .PP
@@ -113,7 +115,7 @@ Validate a directory of configuration, with the summary:
.PP .PP
.nf .nf
.RS .RS
interpres-decode -validate configs/ interpres-decode \-\-validate configs/
.RE .RE
.fi .fi
.PP .PP
@@ -121,7 +123,7 @@ Infer a Go type from a document:
.PP .PP
.nf .nf
.RS .RS
interpres-decode -struct < config.toml > config.go interpres-decode \-\-struct < config.toml > config.go
.RE .RE
.fi .fi
.PP .PP
@@ -129,7 +131,7 @@ Write the template back from the type:
.PP .PP
.nf .nf
.RS .RS
interpres-decode -schema Config config.go interpres-decode \-\-schema Config config.go
.RE .RE
.fi .fi
.SH SEE ALSO .SH SEE ALSO
+72 -21
View File
@@ -48,6 +48,13 @@ type parser struct {
dotted map[string]bool dotted map[string]bool
arrays map[string]bool arrays map[string]bool
// scopeMarks records the definition-map entries added under an array of
// tables, keyed by that array's path, so a new element's reset drops
// exactly what the previous element added. Without it the reset scans
// every map for the prefix, which a document with many elements and many
// definitions outside them turns quadratic.
scopeMarks map[string][]string
currentPath []string currentPath []string
// keys interns key strings: a document that repeats a key across // keys interns key strings: a document that repeats a key across
@@ -196,6 +203,7 @@ func (p *parser) markHeader(pk string) {
p.headers = make(map[string]bool, 4) p.headers = make(map[string]bool, 4)
} }
p.headers[pk] = true p.headers[pk] = true
p.trackScope(pk)
} }
func (p *parser) markFrozen(pk string) { func (p *parser) markFrozen(pk string) {
@@ -203,6 +211,7 @@ func (p *parser) markFrozen(pk string) {
p.frozen = make(map[string]bool, 4) p.frozen = make(map[string]bool, 4)
} }
p.frozen[pk] = true p.frozen[pk] = true
p.trackScope(pk)
} }
func (p *parser) markDotted(pk string) { func (p *parser) markDotted(pk string) {
@@ -210,6 +219,7 @@ func (p *parser) markDotted(pk string) {
p.dotted = make(map[string]bool, 4) p.dotted = make(map[string]bool, 4)
} }
p.dotted[pk] = true p.dotted[pk] = true
p.trackScope(pk)
} }
func (p *parser) markArray(pk string) { func (p *parser) markArray(pk string) {
@@ -217,6 +227,22 @@ func (p *parser) markArray(pk string) {
p.arrays = make(map[string]bool, 2) p.arrays = make(map[string]bool, 2)
} }
p.arrays[pk] = true p.arrays[pk] = true
p.trackScope(pk)
}
// trackScope records a definition entry under every array of tables it falls
// inside, so resetScopeUnder can drop it when a later element opens. An entry
// under no array, such as every definition before the first header, needs no
// record: no reset can ever name it.
func (p *parser) trackScope(pk string) {
for arr := range p.arrays {
if strings.HasPrefix(pk, arr+"\x00") {
if p.scopeMarks == nil {
p.scopeMarks = make(map[string][]string, 2)
}
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
}
}
} }
// internKey returns the shared string for key bytes. The lookup works on the // internKey returns the shared string for key bytes. The lookup works on the
@@ -482,9 +508,14 @@ func (p *parser) parseKeyValue() error {
if p.doc != nil { if p.doc != nil {
node := p.currentNode node := p.currentNode
if len(rest) > 0 { if len(rest) > 0 {
// The tables a dotted key builds hold the position of a line, so
// the write side marks them and gives each leaf back as a dotted
// key rather than a header that would swallow the lines after it.
node = node.addTable(first, dests[0]) node = node.addTable(first, dests[0])
node.dotted = true
for i, k := range rest[:len(rest)-1] { for i, k := range rest[:len(rest)-1] {
node = node.addTable(k, dests[i+1]) node = node.addTable(k, dests[i+1])
node.dotted = true
} }
} }
_, inline := val.(map[string]any) _, inline := val.(map[string]any)
@@ -570,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) {
// resetScopeUnder forgets the definition records nested under key, which // resetScopeUnder forgets the definition records nested under key, which
// belong to the previous element of an array of tables: headers, frozen // belong to the previous element of an array of tables: headers, frozen
// inline tables, dotted-key paths, and nested arrays of tables all start // inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element. // fresh in the new element. The records to drop are the ones the element
// added, which scopeMarks holds; the array's own entry, and everything
// outside it, keep their place.
func (p *parser) resetScopeUnder(key []string) { func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00" pk := pathKey(key)
p.resetMapUnder(p.headers, prefix) for _, k := range p.scopeMarks[pk] {
p.resetMapUnder(p.frozen, prefix) delete(p.headers, k)
p.resetMapUnder(p.dotted, prefix) delete(p.frozen, k)
p.resetMapUnder(p.arrays, prefix) delete(p.dotted, k)
} delete(p.arrays, k)
// resetMapUnder deletes the entries m holds under prefix. An empty or
// unallocated map holds none, so the common case walks nothing.
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
if len(m) == 0 {
return
}
for k := range m {
if strings.HasPrefix(k, prefix) {
delete(m, k)
} }
if p.scopeMarks != nil {
p.scopeMarks[pk] = nil
} }
} }
@@ -708,7 +733,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value") return nil, p.errf("expected a value")
} }
if hasHighByte(tok) && !utf8.ValidString(tok) { if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos) return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
} }
// A date may be followed by a space and a time, forming one date-time. // A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' { if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -719,7 +744,11 @@ func (p *parser) parseAtom() (any, error) {
tok = tok + " " + string(p.src[timeStart:p.pos]) tok = tok + " " + string(p.src[timeStart:p.pos])
} }
} }
if v, ok := parseDateTime(tok); ok { v, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
return v, nil return v, nil
} }
v, err := decodeNumber(tok) v, err := decodeNumber(tok)
@@ -758,6 +787,20 @@ func hasHighByte(s string) bool {
return false return false
} }
// invalidUTF8Offset returns the offset of the first byte in s that does not
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
// that is invalid rather than the end of the token around it.
func invalidUTF8Offset(s string) int {
for i := 0; i < len(s); {
r, size := utf8.DecodeRuneInString(s[i:])
if r == utf8.RuneError && size == 1 {
return i
}
i += size
}
return -1
}
// --- strings --------------------------------------------------------------- // --- strings ---------------------------------------------------------------
func (p *parser) parseBasicString() (string, error) { func (p *parser) parseBasicString() (string, error) {
@@ -895,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) { func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed. // A newline immediately after the opening delimiter is trimmed, and it is
// a newline: a bare CR here is the bare-CR error like anywhere else, not
// a newline to trim.
if !p.eof() && p.peek() == '\r' { if !p.eof() && p.peek() == '\r' {
if next, ok := p.peekAt(1); !ok || next != '\n' {
return "", p.errf("bare carriage return is not allowed in a string")
}
p.pos++ p.pos++
} }
if !p.eof() && p.peek() == '\n' { if !p.eof() && p.peek() == '\n' {
@@ -1047,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) {
} }
hex := string(p.src[p.pos : p.pos+n]) hex := string(p.src[p.pos : p.pos+n])
p.pos += n p.pos += n
v, err := strconv.ParseInt(hex, 16, 64) // ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
// read would let "\U-0000001" through the range checks below only to
// write U+FFFD for a document the grammar rejects.
v, err := strconv.ParseUint(hex, 16, 32)
if err != nil { if err != nil {
return 0, p.errf("invalid unicode escape \\%s", hex) return 0, p.errf("invalid unicode escape \\%s", hex)
} }
+23 -10
View File
@@ -1,4 +1,7 @@
#!/usr/bin/env perl #!/usr/bin/env perl
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
# SPDX-License-Identifier: MIT
# docs-drift compares the toml-test suite counts the documentation names with # docs-drift compares the toml-test suite counts the documentation names with
# the run this repository produces now. A corpus change moves the counts, and # the run this repository produces now. A corpus change moves the counts, and
# README.md and docs/ARCHITECTURE.md quote them; this is the check that keeps # README.md and docs/ARCHITECTURE.md quote them; this is the check that keeps
@@ -6,26 +9,36 @@
use v5.40; use v5.40;
my $out = qx{toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1 2>&1}; my $out = qx{toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1 2>&1};
die "docs-drift: toml-test failed to run; build the adapter first (just build)\n" die "docs-drift: toml-test failed to run; build the adapter first (just build)\n"
if !defined $out || $out =~ /not found|No such file/; if !defined $out || $out =~ /not found|No such file/;
# A red run is a failure to answer, not a drift: the counts it prints describe
# a suite that did not pass, and sending the maintainer to correct counts that
# are correct would be the wrong diagnosis.
die "docs-drift: the toml-test run failed; fix the suite before comparing counts\n"
if $? != 0;
my ($valid) = $out =~ /valid tests:\s+(\d+) passed/; my %live;
my ($invalid) = $out =~ /invalid tests:\s+(\d+) passed/; for my $kind (qw(valid invalid encoder)) {
die "docs-drift: could not read the suite counts from the toml-test output\n" my ($passed) = $out =~ /\b$kind tests:\s+(\d+) passed/;
unless defined $valid && defined $invalid; die "docs-drift: could not read the $kind count from the toml-test output\n"
print "docs-drift: the suite now stands at $valid valid and $invalid invalid cases\n"; unless defined $passed;
my ($failed) = $out =~ /\b$kind tests:\s+\d+ passed, (\d+) failed/;
die "docs-drift: the $kind run has $failed failures; fix the suite first\n"
if defined $failed && $failed != 0;
$live{$kind} = $passed;
}
print "docs-drift: the suite now stands at $live{valid} valid, $live{invalid} invalid and $live{encoder} encoder cases\n";
my $drift = 0; my $drift = 0;
for my $file ('README.md', 'docs/ARCHITECTURE.md') { for my $file ('README.md', 'docs/ARCHITECTURE.md') {
open(my $fh, '<', $file) or die "docs-drift: cannot read $file: $!\n"; open(my $fh, '<', $file) or die "docs-drift: cannot read $file: $!\n";
my $text = do { local $/; <$fh> }; my $text = do { local $/; <$fh> };
close($fh); close($fh);
while ($text =~ /(\d+)\s+(valid|invalid)/g) { while ($text =~ /(\d+)\s+(valid|invalid|encoder)/g) {
my ($quoted, $kind) = ($1, $2); my ($quoted, $kind) = ($1, $2);
my $live = $kind eq 'valid' ? $valid : $invalid; if ($quoted != $live{$kind}) {
if ($quoted != $live) { print "docs-drift: $file quotes $quoted $kind cases, the suite says $live{$kind}\n";
print "docs-drift: $file quotes $quoted $kind cases, the suite says $live\n";
$drift = 1; $drift = 1;
} }
} }
+362 -184
View File
@@ -22,6 +22,14 @@ import (
// document through the tree path, so the observable behaviour is the tree // document through the tree path, so the observable behaviour is the tree
// path's, exactly. A targeted parse either completes with the result the // path's, exactly. A targeted parse either completes with the result the
// tree path would give, or it erases itself. // tree path would give, or it erases itself.
//
// One difference the two paths cannot share: a parse error or a cancellation
// deep in the document leaves the statements before it already written into
// the destination, where the tree path, which parses the whole document
// before it decodes any of it, writes nothing. The value layer shares this
// with encoding/json, whose Unmarshal also leaves a partial destination
// behind a mid-document failure; a destination that must stay untouched on
// error is a destination the caller resets.
var errTargetFallback = errors.New("interpres: targeted decode falls back to the tree path") var errTargetFallback = errors.New("interpres: targeted decode falls back to the tree path")
// targetCache holds whether a destination type may take the targeted parse. // targetCache holds whether a destination type may take the targeted parse.
@@ -33,10 +41,12 @@ var targetCache sync.Map // reflect.Type -> bool
var mapStringAnyType = reflect.TypeFor[map[string]any]() var mapStringAnyType = reflect.TypeFor[map[string]any]()
// typeTargetable reports whether decoding into the struct type t can use the // typeTargetable reports whether decoding into the struct type t can use the
// targeted parse. The one structural ban is untagged embedded maps: their // targeted parse. The structural bans are the shapes whose tree behaviour
// filler-key rule lives in the tree decode, and a targeted document that // the skeleton cannot model: untagged embedded maps, an OrderedMap anywhere a
// meets an unknown table would need a subtree of it. Everything else is safe // table opens, and a custom decode hook on any table the parse would enter
// to attempt, because the value layer is the ordinary decode and every // directly (a struct field, a map field, or the element of a table slice),
// because the tree hands a hook the whole parsed value. Everything else is
// safe to attempt, because the value layer is the ordinary decode and every
// mismatch falls back. // mismatch falls back.
func typeTargetable(t reflect.Type) bool { func typeTargetable(t reflect.Type) bool {
if t == nil || t.Kind() != reflect.Struct || t == orderedMapType { if t == nil || t.Kind() != reflect.Struct || t == orderedMapType {
@@ -59,22 +69,41 @@ func scanTargetable(t reflect.Type, seen map[reflect.Type]bool) bool {
return false return false
} }
for _, loc := range cachedStructSchema(t).byName { for _, loc := range cachedStructSchema(t).byName {
ft := derefType(t.FieldByIndex(loc.index).Type) if !scanTargetableField(derefType(t.FieldByIndex(loc.index).Type), seen) {
return false
}
}
return true
}
// scanTargetableField reports whether one field's type is safe for the
// targeted skeleton to fill directly.
func scanTargetableField(ft reflect.Type, seen map[reflect.Type]bool) bool {
switch ft.Kind() {
case reflect.Struct:
if ft == orderedMapType { if ft == orderedMapType {
return false return false
} }
if ft.Kind() != reflect.Struct || isScalarStruct(ft) { if isScalarStruct(ft) {
continue return true
} }
// A struct field with a custom decode hook receives the whole parsed // A struct the parse enters directly never builds the whole value
// value from the tree decode; the targeted skeleton never builds that // the tree hands a hook, so the hook must win.
// value for a table it enters directly, so the hook must win.
if implementsDecodeHook(ft) || implementsDecodeHook(reflect.PointerTo(ft)) { if implementsDecodeHook(ft) || implementsDecodeHook(reflect.PointerTo(ft)) {
return false return false
} }
if !scanTargetable(ft, seen) { return scanTargetable(ft, seen)
case reflect.Map:
return !implementsDecodeHook(ft) && !implementsDecodeHook(reflect.PointerTo(ft))
case reflect.Slice, reflect.Array:
et := derefType(ft.Elem())
if et == orderedMapType {
return false return false
} }
if et.Kind() == reflect.Struct && !isScalarStruct(et) {
return scanTargetableField(et, seen)
}
return true
} }
return true return true
} }
@@ -110,20 +139,45 @@ func canTargetDecode(v any) bool {
// targetTable is one open table of the targeted parse: the struct (or map) // targetTable is one open table of the targeted parse: the struct (or map)
// value its keys fill, the schema that resolves them (nil for a map or sink // value its keys fill, the schema that resolves them (nil for a map or sink
// destination), the absolute path its errors wrap, and whether it collects // destination), and the absolute path its errors wrap. A sink is the
// the keys no field claims for the strict check. A sink is the destination // destination an unknown subtree gets: its statements parse for the syntax
// an unknown subtree gets: its statements parse for the syntax and definition // and definition contracts, and its values are discarded. The strict and
// contracts, and its values are discarded. // required findings live in the parser's per-address store, not here,
// because the table objects a dotted descent builds are transient while the
// destination is not.
type targetTable struct { type targetTable struct {
rv reflect.Value rv reflect.Value
schema *structSchema schema *structSchema
path []string path []string
sink bool sink bool
keys []string // the keys defined in the table, interned; struct // strict is the strict-decode setting the table was opened with, carried
// tables keep theirs per destination address instead // into the per-address strict state on first sight.
strict bool
// arrayElem marks a sink created as the element of an unknown array of
// tables: a dotted key may not enter it, the tree's own rule for an
// array, while a [sub-table] header may, through the last element.
arrayElem bool
keys []string // the keys defined in a sink, as full path keys
}
// strictState is the strict and required bookkeeping of one struct
// destination, keyed by the value's address.
type strictState struct {
path []string
typ reflect.Type
schema *structSchema
strict bool strict bool
unknown string // strict: the smallest unclaimed key so far unknown string // strict: the smallest unclaimed key so far
resolvedSeen map[string]bool // the schema keys resolved so far, for required resolved map[string]bool
}
// arrayFill tracks how many elements of one fixed-size array the document
// has filled, with what the length mismatch the tree decode reports needs:
// the array's type and its field's path.
type arrayFill struct {
next int
typ reflect.Type
path []string
} }
// targetParser parses a document straight into a struct destination. It // targetParser parses a document straight into a struct destination. It
@@ -139,61 +193,53 @@ type targetParser struct {
rootT *targetTable rootT *targetTable
cur *targetTable cur *targetTable
// arrayNext tracks how many elements of a fixed-size array the document // arrayFills counts the elements of each fixed-size array the document
// has filled, per field address. // has filled, per field address, in the order the arrays were met.
arrayNext map[uintptr]int arrayFills map[uintptr]*arrayFill
fillOrder []*arrayFill
// appendedHere records the slice fields this document's [[headers]] have
// filled: a slice the caller prefilled is replaced by the tree decode,
// not appended to, so the first header over one falls back.
appendedHere map[uintptr]bool
// opened registers every opened table by its path key, sinks included: a // opened registers every opened table by its path key, sinks included: a
// later header or dotted key meets the table the tree already built. // later header or dotted key meets the table the tree already built.
opened map[string]*targetTable opened map[string]*targetTable
// keyAssigned records the fields a key statement assigned directly, per
// table address: an array-of-tables header over such a field is the
// tree's `not an array of tables` error, where a header over a
// header-built array appends.
keyAssignedKeys map[uintptr]map[string]bool
// tableKeysByAddr holds the defined keys of one struct destination, keyed // tableKeysByAddr holds the defined keys of one struct destination, keyed
// by the value's address: the table object a dotted descent builds is // by the value's address: the table object a dotted descent builds is
// transient, the destination is not. // transient, the destination is not.
tableKeysByAddr map[uintptr][]string tableKeysByAddr map[uintptr][]string
}
// markKeyAssigned records that a key statement assigned the field, and // mapKeysByAddr holds the keys the document has defined in one map
// keyAssigned reports that state. Sinks keep no such bookkeeping. // destination, keyed the same way: the destination map the caller
func (tp *targetParser) markKeyAssigned(t *targetTable, key string) { // prefilled is not the parser's state, and a key it holds is not the
if t.sink || t.schema == nil || !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct { // duplicate a key the document repeats is.
return mapKeysByAddr map[uintptr]map[string]bool
}
addr := t.rv.Addr().Pointer()
if tp.keyAssignedKeys == nil {
tp.keyAssignedKeys = make(map[uintptr]map[string]bool, 8)
}
if tp.keyAssignedKeys[addr] == nil {
tp.keyAssignedKeys[addr] = make(map[string]bool, 8)
}
tp.keyAssignedKeys[addr][key] = true
}
func (tp *targetParser) keyAssigned(t *targetTable, key string) bool { // strictByAddr holds each destination's strict and required findings,
if t.sink || t.schema == nil || !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct { // with strictOrder keeping the document order they first appeared in.
return false strictByAddr map[uintptr]*strictState
} strictOrder []uintptr
addr := t.rv.Addr().Pointer()
return tp.keyAssignedKeys[addr][key]
} }
// tableHas reports whether key is already defined in the table. A struct // tableHas reports whether key is already defined in the table. A struct
// table's keys are interned parser strings, so the linear scan compares // table's keys are interned parser strings, so the linear scan compares
// against a handful of short keys, cheaper than hashing a per-table map. // against a handful of short keys, cheaper than hashing a per-table map.
// Struct tables keep their keys by destination address, because the table // Struct tables keep their keys by destination address, and map tables keep
// object a dotted descent builds is transient while the destination is not. // theirs there too, because the table object a dotted descent builds is
// transient while the destination is not, and the destination map's own
// contents are the caller's, not the document's.
func (tp *targetParser) tableHas(t *targetTable, key string) bool { func (tp *targetParser) tableHas(t *targetTable, key string) bool {
switch { switch {
case t.sink: case t.sink:
return slices.Contains(t.keys, key) return slices.Contains(t.keys, key)
case t.schema == nil: case t.schema == nil:
return t.rv.Kind() == reflect.Map && t.rv.MapIndex(reflect.ValueOf(key)).IsValid() if t.rv.Kind() != reflect.Map || !t.rv.CanAddr() {
return false
}
return tp.mapKeysByAddr[t.rv.Addr().Pointer()][key]
default: default:
return slices.Contains(tp.tableKeys(t), key) return slices.Contains(tp.tableKeys(t), key)
} }
@@ -217,6 +263,17 @@ func (tp *targetParser) tableMark(t *targetTable, key string) {
case t.sink: case t.sink:
t.keys = append(t.keys, key) t.keys = append(t.keys, key)
case t.schema == nil: case t.schema == nil:
if !t.rv.CanAddr() || t.rv.Kind() != reflect.Map {
return
}
addr := t.rv.Addr().Pointer()
if tp.mapKeysByAddr == nil {
tp.mapKeysByAddr = make(map[uintptr]map[string]bool, 8)
}
if tp.mapKeysByAddr[addr] == nil {
tp.mapKeysByAddr[addr] = make(map[string]bool, 8)
}
tp.mapKeysByAddr[addr][key] = true
default: default:
if !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct { if !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return return
@@ -229,32 +286,49 @@ func (tp *targetParser) tableMark(t *targetTable, key string) {
} }
} }
// resolvedHas reports whether the resolved schema key has been seen, the // strictState returns the strict and required bookkeeping of the struct
// check a required tag runs: the duplicate bookkeeping tracks the key as the // destination t fills, registering it on first sight so a finding recorded
// document wrote it, the required bookkeeping the key as the schema // on a transient table survives the table.
// resolved it. func (tp *targetParser) strictState(t *targetTable) *strictState {
func (t *targetTable) resolvedHas(key string) bool { addr := t.rv.Addr().Pointer()
return t.resolvedSeen[key] if st, ok := tp.strictByAddr[addr]; ok {
return st
}
st := &strictState{
path: slices.Clone(t.path),
typ: t.rv.Type(),
schema: t.schema,
strict: t.strict,
resolved: make(map[string]bool, 8),
}
tp.strictByAddr[addr] = st
tp.strictOrder = append(tp.strictOrder, addr)
return st
} }
func (t *targetTable) markResolved(key string) { // markResolved records that the document resolved the schema key, the check
// a required tag runs: the duplicate bookkeeping tracks the key as the
// document wrote it, the required bookkeeping the key as the schema resolved
// it.
func (tp *targetParser) markResolved(t *targetTable, key string) {
if t.schema == nil || len(t.schema.required) == 0 { if t.schema == nil || len(t.schema.required) == 0 {
return return
} }
if t.resolvedSeen == nil { if !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
t.resolvedSeen = make(map[string]bool, 8) return
} }
t.resolvedSeen[key] = true tp.strictState(t).resolved[key] = true
} }
// recordStrictUnknown remembers the key no field claims when strict decoding // recordStrictUnknown remembers the key no field claims when strict decoding
// is on: the smallest one is reported, the tree decode's own choice. // is on: the smallest one is reported, the tree decode's own choice.
func (t *targetTable) recordStrictUnknown(key string) { func (tp *targetParser) recordStrictUnknown(t *targetTable, key string) {
if !t.strict { if !t.strict || t.schema == nil || !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return return
} }
if t.unknown == "" || key < t.unknown { st := tp.strictState(t)
t.unknown = key if st.unknown == "" || key < st.unknown {
st.unknown = key
} }
} }
@@ -277,9 +351,10 @@ func parseIntoTargeted(ctx context.Context, data []byte, d *decoder, useNumber b
parser: &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, useNumber: useNumber}, parser: &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, useNumber: useNumber},
d: d, d: d,
root: rv.Elem(), root: rv.Elem(),
arrayNext: make(map[uintptr]int, 4), arrayFills: make(map[uintptr]*arrayFill, 4),
keyAssignedKeys: make(map[uintptr]map[string]bool, 8), appendedHere: make(map[uintptr]bool, 4),
opened: make(map[string]*targetTable, 8), opened: make(map[string]*targetTable, 8),
strictByAddr: make(map[uintptr]*strictState, 8),
} }
tp.rootT = &targetTable{rv: tp.root, schema: schemaRef(tp.root.Type()), strict: d.disallowUnknown} tp.rootT = &targetTable{rv: tp.root, schema: schemaRef(tp.root.Type()), strict: d.disallowUnknown}
tp.tables = append(tp.tables, tp.rootT) tp.tables = append(tp.tables, tp.rootT)
@@ -326,35 +401,57 @@ func (tp *targetParser) run() error {
} }
// reportDeferred raises the decode-stage findings in the tree decode's // reportDeferred raises the decode-stage findings in the tree decode's
// order: the root table first, then the opened tables in document order. The // order. A fixed-size array the document under-filled is the length mismatch
// tree decode reports them after a full parse, so a later parse error always // the tree decode raises, and it comes first. Then the unknown keys, before
// won; here the parse has already completed. // the required ones, because the tree decode meets an unknown key while it
// assigns and checks a table's required keys only once the whole table has
// been; within each class the order is the order the destinations first
// appeared in, the document's own order.
func (tp *targetParser) reportDeferred() error { func (tp *targetParser) reportDeferred() error {
for _, t := range append([]*targetTable{tp.rootT}, tp.tables...) { for _, f := range tp.fillOrder {
if t.sink { if f.next != f.typ.Len() {
return wrapTablePath(f.path, fmt.Errorf("interpres: cannot assign %d elements to %s", f.next, f.typ))
}
}
for _, addr := range tp.strictOrder {
if st := tp.strictByAddr[addr]; st.strict && st.unknown != "" {
return wrapTablePath(st.path, fmt.Errorf("interpres: unknown field %q for %s", st.unknown, st.typ))
}
}
for _, addr := range tp.strictOrder {
st := tp.strictByAddr[addr]
if st.schema == nil {
continue continue
} }
if t.strict && t.unknown != "" { for _, key := range st.schema.required {
return tp.wrapTableErr(t, fmt.Errorf("interpres: unknown field %q for %s", t.unknown, t.rv.Type())) if !st.resolved[key] {
} return wrapTablePath(st.path, fmt.Errorf("interpres: missing required key %q", key))
if t.schema != nil {
for _, key := range t.schema.required {
if !t.resolvedHas(key) {
return tp.wrapTableErr(t, fmt.Errorf("interpres: missing required key %q", key))
}
} }
} }
} }
return nil return nil
} }
// wrapTableErr wraps a table's finding the way the tree decode wraps it: the // wrapTablePath wraps a table's finding the way the tree decode wraps it: the
// root speaks for itself, a nested table gains its path. // root speaks for itself, a nested table gains its path.
func (tp *targetParser) wrapTableErr(t *targetTable, err error) error { func wrapTablePath(path []string, err error) error {
if len(t.path) == 0 { if len(path) == 0 {
return err return err
} }
return &DecodeError{Path: Path(slices.Clone(t.path)), Err: err} return &DecodeError{Path: Path(slices.Clone(path)), Err: err}
}
// arrayFillFor returns the fill record of the fixed-size array fv, keyed by
// its address, created with its field's path on first sight.
func (tp *targetParser) arrayFillFor(fv reflect.Value, path []string) *arrayFill {
addr := fv.Addr().Pointer()
if f, ok := tp.arrayFills[addr]; ok {
return f
}
f := &arrayFill{typ: fv.Type(), path: path}
tp.arrayFills[addr] = f
tp.fillOrder = append(tp.fillOrder, f)
return f
} }
// --- headers --------------------------------------------------------------- // --- headers ---------------------------------------------------------------
@@ -474,11 +571,11 @@ func (tp *targetParser) descendOne(parent *targetTable, key string, abs []string
return nil, errTargetFallback return nil, errTargetFallback
} }
if existing := parent.rv.MapIndex(reflect.ValueOf(key)); existing.IsValid() && !existing.IsNil() { if existing := parent.rv.MapIndex(reflect.ValueOf(key)); existing.IsValid() && !existing.IsNil() {
return &targetTable{rv: existing.Elem(), path: abs}, nil return &targetTable{rv: existing, path: slices.Clone(abs)}, nil
} }
next := reflect.MakeMap(elemT) next := reflect.MakeMap(elemT)
parent.rv.SetMapIndex(reflect.ValueOf(key), next) parent.rv.SetMapIndex(reflect.ValueOf(key), next)
return &targetTable{rv: next, path: abs}, nil return &targetTable{rv: next, path: slices.Clone(abs)}, nil
} }
resolved := key resolved := key
loc, ok := parent.schema.byName[key] loc, ok := parent.schema.byName[key]
@@ -487,18 +584,18 @@ func (tp *targetParser) descendOne(parent *targetTable, key string, abs []string
loc, ok = parent.schema.byName[resolved] loc, ok = parent.schema.byName[resolved]
} }
if !ok { if !ok {
parent.recordStrictUnknown(key) tp.recordStrictUnknown(parent, key)
if opened, ok := tp.opened[pathKey(abs)]; ok { if opened, ok := tp.opened[pathKey(abs)]; ok {
return opened, nil return opened, nil
} }
if tp.tableHas(parent, key) { if tp.tableHas(parent, key) {
return nil, tp.errf("key %q is not a table", key) return nil, tp.errf("key %q is not a table", key)
} }
sink := &targetTable{sink: true, path: abs} sink := &targetTable{sink: true, path: slices.Clone(abs)}
tp.opened[pathKey(abs)] = sink tp.opened[pathKey(abs)] = sink
return sink, nil return sink, nil
} }
parent.markResolved(resolved) tp.markResolved(parent, resolved)
fv, err := fieldByIndex(parent.rv, loc.index) fv, err := fieldByIndex(parent.rv, loc.index)
if err != nil { if err != nil {
return nil, errTargetFallback return nil, errTargetFallback
@@ -509,10 +606,12 @@ func (tp *targetParser) descendOne(parent *targetTable, key string, abs []string
// openValueTable opens a table scope over a placed field value, allocating a // openValueTable opens a table scope over a placed field value, allocating a
// nil pointer on the way. The rules mirror the tree decode's own type // nil pointer on the way. The rules mirror the tree decode's own type
// decisions: a struct enters, a map enters (allocated when nil), an array of // decisions: a struct enters, a map enters (allocated when nil), an array of
// tables enters its last element, and anything else is a type mismatch the // tables enters its last filled element, a slice enters the last element of
// tree decode reports, so it falls back. A scalar field the table keys // an array this document's [[headers]] built (a prefilled slice is a table
// already define is the tree's `key is not a table` error, checked against // the tree decode rejects, so it falls back), and anything else is a type
// parent, the table the key belongs to. // mismatch the tree decode reports, so it falls back too. A scalar field the
// table keys already define is the tree's `key is not a table` error,
// checked against parent, the table the key belongs to.
func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, key string, abs []string, strict bool) (*targetTable, error) { func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, key string, abs []string, strict bool) (*targetTable, error) {
if fv.Kind() == reflect.Pointer { if fv.Kind() == reflect.Pointer {
if fv.IsNil() { if fv.IsNil() {
@@ -528,7 +627,7 @@ func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, ke
if isScalarStruct(fv.Type()) { if isScalarStruct(fv.Type()) {
return nil, errTargetFallback return nil, errTargetFallback
} }
return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: abs, strict: strict}, nil return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: slices.Clone(abs), strict: strict}, nil
case reflect.Map: case reflect.Map:
if fv.Type().Key().Kind() != reflect.String { if fv.Type().Key().Kind() != reflect.String {
return nil, errTargetFallback return nil, errTargetFallback
@@ -536,11 +635,12 @@ func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, ke
if fv.IsNil() { if fv.IsNil() {
fv.Set(reflect.MakeMap(fv.Type())) fv.Set(reflect.MakeMap(fv.Type()))
} }
return &targetTable{rv: fv, path: abs}, nil return &targetTable{rv: fv, path: slices.Clone(abs)}, nil
case reflect.Slice: case reflect.Slice:
if fv.Len() == 0 { if !tp.appendedHere[fv.Addr().Pointer()] {
// No [[header]] ever filled it, so the tree holds a map here and // No [[header]] of this document filled it, so the tree holds a
// its decode raises the type mismatch. // map here, whose decode raises the type mismatch; a prefilled
// slice is the tree's replacement case, not a table to enter.
return nil, errTargetFallback return nil, errTargetFallback
} }
et := derefType(fv.Type().Elem()) et := derefType(fv.Type().Elem())
@@ -548,6 +648,22 @@ func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, ke
return nil, errTargetFallback return nil, errTargetFallback
} }
return &targetTable{rv: fv.Index(fv.Len() - 1), schema: schemaRef(et), path: elementPath(abs, key, fv.Len()-1), strict: strict}, nil return &targetTable{rv: fv.Index(fv.Len() - 1), schema: schemaRef(et), path: elementPath(abs, key, fv.Len()-1), strict: strict}, nil
case reflect.Array:
fill := tp.arrayFillFor(fv, append(slices.Clone(parent.path), key))
if fill.next == 0 {
return nil, errTargetFallback
}
et := derefType(fv.Type().Elem())
if et.Kind() == reflect.Struct {
if isScalarStruct(et) {
return nil, errTargetFallback
}
return &targetTable{rv: fv.Index(fill.next - 1), schema: schemaRef(et), path: elementPath(abs, key, fill.next-1), strict: strict}, nil
}
if et.Kind() == reflect.Map && et.Key().Kind() == reflect.String {
return &targetTable{rv: fv.Index(fill.next - 1), path: elementPath(abs, key, fill.next-1)}, nil
}
return nil, errTargetFallback
} }
if tp.tableHas(parent, key) { if tp.tableHas(parent, key) {
return nil, tp.errf("key %q is not a table", key) return nil, tp.errf("key %q is not a table", key)
@@ -578,12 +694,30 @@ func (tp *targetParser) appendArrayTable(key []string) error {
if tp.parser.dotted[pk] || tp.parser.headers[pk] { if tp.parser.dotted[pk] || tp.parser.headers[pk] {
return tp.errf("key %q is not an array of tables", leaf) return tp.errf("key %q is not an array of tables", leaf)
} }
// A leaf the document already defined as a value or a table is the tree
// parser's own parse error, and a leaf an earlier [[header]] defined
// opens a new element; the arrays map, read before this header marks it,
// is what tells the two apart. A sink parent holds no destination state
// worth consulting.
if !parent.sink && !tp.parser.arrays[pk] && tp.tableHas(parent, leaf) {
return tp.errf("key %q is not an array of tables", leaf)
}
// A new element starts a fresh scope, exactly as the tree parser's own
// header does: sub-headers, inline freezes and nested arrays from the
// previous element no longer apply.
tp.parser.resetScopeUnder(key)
tp.parser.markArray(pk) tp.parser.markArray(pk)
elem, err := tp.appendElement(parent, leaf, key) elem, err := tp.appendElement(parent, leaf, key)
if err != nil { if err != nil {
return err return err
} }
if elem.sink {
// A sink the element scope reuses ([[a.b]] over an unknown a, the
// parent sink) starts the new element with no keys, the way a known
// array's element does.
elem.keys = nil
}
tp.tableMark(parent, leaf) tp.tableMark(parent, leaf)
if !elem.sink { if !elem.sink {
tp.tables = append(tp.tables, elem) tp.tables = append(tp.tables, elem)
@@ -593,9 +727,9 @@ func (tp *targetParser) appendArrayTable(key []string) error {
} }
// appendElement appends one element to the array the leaf names in parent // appendElement appends one element to the array the leaf names in parent
// and returns its table. A leaf no field claims sinks; a field whose array // and returns its table. A leaf no field claims sinks, a fresh namespace per
// element kind cannot be a table falls back, the tree decode owning the type // element; a field whose array element kind cannot be a table falls back,
// error. // the tree decode owning the type error.
func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []string) (*targetTable, error) { func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []string) (*targetTable, error) {
if parent.sink { if parent.sink {
return parent, nil return parent, nil
@@ -608,10 +742,15 @@ func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []st
gk := reflect.ValueOf(leaf) gk := reflect.ValueOf(leaf)
var arr reflect.Value var arr reflect.Value
if existing := parent.rv.MapIndex(gk); existing.IsValid() && !existing.IsNil() { if existing := parent.rv.MapIndex(gk); existing.IsValid() && !existing.IsNil() {
if existing.Elem().Kind() != reflect.Slice { // A map[string]any destination boxes its arrays in the
// interface; a typed map hands the slice itself.
if existing.Kind() == reflect.Interface {
existing = existing.Elem()
}
if existing.Kind() != reflect.Slice {
return nil, tp.errf("key %q is not an array of tables", leaf) return nil, tp.errf("key %q is not an array of tables", leaf)
} }
arr = existing.Elem() arr = existing
} }
var elem reflect.Value var elem reflect.Value
switch { switch {
@@ -645,79 +784,103 @@ func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []st
loc, ok = parent.schema.byName[resolved] loc, ok = parent.schema.byName[resolved]
} }
if !ok { if !ok {
parent.recordStrictUnknown(leaf) tp.recordStrictUnknown(parent, leaf)
if opened, ok := tp.opened[pathKey(parent.path)]; ok { // Every element is a fresh namespace, the way a known array's is,
return opened, nil // registered under the header's path so a [sub-table] header reaches
} // the last element, the tree's rule for a header under an array of
if tp.tableHas(parent, leaf) { // tables; a dotted key skips it, the tree's rule for an array.
return nil, tp.errf("key %q is not an array of tables", leaf) sink := &targetTable{sink: true, arrayElem: true, path: slices.Clone(key)}
} tp.opened[pathKey(key)] = sink
sink := &targetTable{sink: true, path: parent.path}
tp.opened[pathKey(parent.path)] = sink
return sink, nil return sink, nil
} }
parent.markResolved(resolved) tp.markResolved(parent, resolved)
if tp.keyAssigned(parent, resolved) {
// A key statement already assigned the field its own value; the tree
// holds a value array there and its header append is the
// `not an array of tables` parse error.
return nil, tp.errf("key %q is not an array of tables", leaf)
}
fv, err := fieldByIndex(parent.rv, loc.index) fv, err := fieldByIndex(parent.rv, loc.index)
if err != nil { if err != nil {
return nil, errTargetFallback return nil, errTargetFallback
} }
fieldPath := append(slices.Clone(parent.path), leaf)
if fv.Kind() == reflect.Array { if fv.Kind() == reflect.Array {
// A fixed-size array fills position by position; one element too many // A fixed-size array fills position by position; one element too many
// is the length mismatch the tree decode reports. // is the length mismatch the tree decode reports, and one too few is
// the same mismatch, checked when the parse completes.
fill := tp.arrayFillFor(fv, fieldPath)
et := derefType(fv.Type().Elem()) et := derefType(fv.Type().Elem())
switch et.Kind() { switch {
case reflect.Struct: case et.Kind() == reflect.Struct && !isScalarStruct(et):
if isScalarStruct(et) { if fill.next >= fv.Len() {
return nil, errTargetFallback return nil, errTargetFallback
} }
addr := fv.Addr().Pointer() n := fill.next
n := tp.arrayNext[addr] fill.next = n + 1
if n >= fv.Len() { return &targetTable{rv: fv.Index(n), schema: schemaRef(et), path: elementPath(parent.path, leaf, n), strict: parent.strict}, nil
case et.Kind() == reflect.Map && et.Key().Kind() == reflect.String:
if fill.next >= fv.Len() {
return nil, errTargetFallback return nil, errTargetFallback
} }
tp.arrayNext[addr] = n + 1 n := fill.next
return &targetTable{rv: fv.Index(n), schema: schemaRef(et), path: parent.path, strict: parent.strict}, nil fill.next = n + 1
case reflect.Map:
if et.Key().Kind() != reflect.String {
return nil, errTargetFallback
}
addr := fv.Addr().Pointer()
n := tp.arrayNext[addr]
if n >= fv.Len() {
return nil, errTargetFallback
}
tp.arrayNext[addr] = n + 1
elem := reflect.MakeMap(et) elem := reflect.MakeMap(et)
fv.Index(n).Set(elem) fv.Index(n).Set(elem)
return &targetTable{rv: elem, path: parent.path}, nil return &targetTable{rv: elem, path: elementPath(parent.path, leaf, n)}, nil
} }
return nil, errTargetFallback return nil, errTargetFallback
} }
if fv.Kind() != reflect.Slice { if fv.Kind() != reflect.Slice {
if tp.tableHas(parent, leaf) {
return nil, tp.errf("key %q is not an array of tables", leaf) return nil, tp.errf("key %q is not an array of tables", leaf)
} }
// The tree builds an array here without asking the destination, and
// its decode answers with the type mismatch; the fallback keeps the
// message the tree path gives.
return nil, errTargetFallback
}
addr := fv.Addr().Pointer()
if !tp.appendedHere[addr] {
if fv.Len() > 0 {
// A slice the caller prefilled is replaced by the tree decode,
// not appended to; the fallback runs the document the tree's way.
return nil, errTargetFallback
}
tp.appendedHere[addr] = true
}
et := derefType(fv.Type().Elem()) et := derefType(fv.Type().Elem())
switch et.Kind() { switch et.Kind() {
case reflect.Struct: case reflect.Struct:
if isScalarStruct(et) { if isScalarStruct(et) {
return nil, errTargetFallback return nil, errTargetFallback
} }
grown := reflect.Append(fv, reflect.New(et).Elem()) // A pointer element is appended as the allocated pointer and filled
// through its pointee, so []*T takes the same path []T does. The
// element the table fills is the slice's own: the value a New built
// stands apart from the backing array.
var el, appended reflect.Value
if fv.Type().Elem().Kind() == reflect.Pointer {
p := reflect.New(et)
el, appended = p.Elem(), p
} else {
el = reflect.New(et).Elem()
appended = el
}
grown := reflect.Append(fv, appended)
fv.Set(grown) fv.Set(grown)
return &targetTable{rv: grown.Index(grown.Len() - 1), schema: schemaRef(et), path: elementPath(parent.path, leaf, grown.Len()-1), strict: parent.strict}, nil if fv.Type().Elem().Kind() != reflect.Pointer {
el = grown.Index(grown.Len() - 1)
}
return &targetTable{rv: el, schema: schemaRef(et), path: elementPath(parent.path, leaf, grown.Len()-1), strict: parent.strict}, nil
case reflect.Map: case reflect.Map:
if et.Key().Kind() != reflect.String { if et.Key().Kind() != reflect.String {
return nil, errTargetFallback return nil, errTargetFallback
} }
grown := reflect.Append(fv, reflect.MakeMap(et)) m := reflect.MakeMap(et)
var appended reflect.Value = m
if fv.Type().Elem().Kind() == reflect.Pointer {
p := reflect.New(et)
p.Elem().Set(m)
appended = p
}
grown := reflect.Append(fv, appended)
fv.Set(grown) fv.Set(grown)
return &targetTable{rv: grown.Index(grown.Len() - 1), path: elementPath(parent.path, leaf, grown.Len()-1)}, nil return &targetTable{rv: m, path: elementPath(parent.path, leaf, grown.Len()-1)}, nil
} }
return nil, errTargetFallback return nil, errTargetFallback
} }
@@ -780,22 +943,18 @@ func (tp *targetParser) parseKeyStatement() error {
if verr != nil { if verr != nil {
return verr return verr
} }
dupKey := first full := append([]string{first}, rest...)
if len(dest.path) > 0 || len(rest) > 0 { if len(dest.path) > 0 {
full := make([]string, 0, len(dest.path)+len(rest)+1) full = append(slices.Clone(dest.path), full...)
full = append(full, dest.path...)
full = append(full, first)
full = append(full, rest...)
dupKey = pathKey(full)
} }
if tp.tableHas(leafTable, dupKey) { if tp.tableHas(leafTable, pathKey(full)) {
return p.errf("duplicate key %q", leaf) return p.errf("duplicate key %q", leaf)
} }
tp.tableMark(leafTable, dupKey) tp.tableMark(leafTable, pathKey(full))
if m, isMap := val.(map[string]any); isMap { if m, isMap := val.(map[string]any); isMap {
full := make([]string, 0, len(dest.path)+len(leaf)+1) // An inline table freezes the whole path the statement wrote,
full = append(full, dest.path...) // intermediate segments included, so no later header or dotted
full = append(full, leaf) // key can extend it at any depth.
p.freezeInline(full, m) p.freezeInline(full, m)
} }
return nil return nil
@@ -804,27 +963,29 @@ func (tp *targetParser) parseKeyStatement() error {
return p.errf("duplicate key %q", leaf) return p.errf("duplicate key %q", leaf)
} }
tp.tableMark(leafTable, leaf) tp.tableMark(leafTable, leaf)
if dst.Kind() == reflect.Slice || dst.Kind() == reflect.Array {
// Only a field a later [[header]] could append to needs the
// key-assigned record; everything else never checks it.
et := derefType(dst.Type().Elem())
if et.Kind() == reflect.Struct && !isScalarStruct(et) || et.Kind() == reflect.Map {
tp.markKeyAssigned(leafTable, leaf)
}
}
val, err := tp.parseValueInto(dst) val, err := tp.parseValueInto(dst)
if err != nil { if err != nil {
if errors.Is(err, errTargetFallback) {
return err
}
if _, isSyntax := errors.AsType[*SyntaxError](err); !isSyntax {
// A hook's own failure, which the fallback must not rerun: it
// returns wrapped the way the tree decode wraps a field's.
return newDecodeError(leaf, err)
}
return err return err
} }
if mapDst.IsValid() { if mapDst.IsValid() {
mapDst.SetMapIndex(reflect.ValueOf(leaf), dst) mapDst.SetMapIndex(reflect.ValueOf(leaf), dst)
} }
if m, isMap := val.(map[string]any); isMap { if m, isMap := val.(map[string]any); isMap {
// An inline table freezes its paths; the abs slice is built for it // An inline table freezes the whole path the statement wrote,
// alone, after the parse proved one is needed. // intermediate segments included; the slice is built for it alone,
abs := make([]string, 0, len(dest.path)+len(leaf)+1) // after the parse proved one is needed.
abs := make([]string, 0, len(dest.path)+len(rest)+1)
abs = append(abs, dest.path...) abs = append(abs, dest.path...)
abs = append(abs, leaf) abs = append(abs, first)
abs = append(abs, rest...)
p.freezeInline(abs, m) p.freezeInline(abs, m)
} }
return nil return nil
@@ -847,7 +1008,10 @@ func (tp *targetParser) parseValueInto(dst reflect.Value) (any, error) {
return nil, err return nil, err
} }
if err := tp.d.assign(v, dst); err != nil { if err := tp.d.assign(v, dst); err != nil {
return nil, errTargetFallback // The hook has run; falling back would run it a second time on
// tree path, so its error returns as the tree path's own, for
// the caller to wrap the way the tree decode wraps a field's.
return nil, err
} }
return v, nil return v, nil
} }
@@ -890,7 +1054,16 @@ func (tp *targetParser) parseValueInto(dst reflect.Value) (any, error) {
} }
} }
tok := tp.scanNumberToken() tok := tp.scanNumberToken()
if dtv, dtok := parseDateTime(tok); dtok { if hasHighByte(tok) && invalidUTF8Offset(tok) >= 0 {
// The token route the targeted parse takes validates UTF-8 the
// way the tree scanner does, on the byte that does not decode.
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
}
dtv, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
if err := tp.d.assign(dtv, dst); err != nil { if err := tp.d.assign(dtv, dst); err != nil {
return nil, errTargetFallback return nil, errTargetFallback
} }
@@ -1091,10 +1264,10 @@ func (tp *targetParser) descendDotted(dest *targetTable, first string, rest []st
loc, found = tbl.schema.byName[resolved] loc, found = tbl.schema.byName[resolved]
} }
if !found { if !found {
tbl.recordStrictUnknown(leaf) tp.recordStrictUnknown(tbl, leaf)
return reflect.Value{}, reflect.Value{}, leafTable, false, nil return reflect.Value{}, reflect.Value{}, leafTable, false, nil
} }
tbl.markResolved(resolved) tp.markResolved(tbl, resolved)
fv, ferr := fieldByIndex(tbl.rv, loc.index) fv, ferr := fieldByIndex(tbl.rv, loc.index)
if ferr != nil { if ferr != nil {
return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback
@@ -1118,15 +1291,14 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
return nil, errTargetFallback return nil, errTargetFallback
} }
if existing := tbl.rv.MapIndex(reflect.ValueOf(seg)); existing.IsValid() && !existing.IsNil() { if existing := tbl.rv.MapIndex(reflect.ValueOf(seg)); existing.IsValid() && !existing.IsNil() {
ev := existing.Elem() if existing.Kind() != reflect.Map {
if ev.Kind() != reflect.Map {
return nil, tp.errf("key %q is not a table", seg) return nil, tp.errf("key %q is not a table", seg)
} }
return &targetTable{rv: ev, path: segAbs}, nil return &targetTable{rv: existing, path: slices.Clone(segAbs)}, nil
} }
next := reflect.MakeMap(elemT) next := reflect.MakeMap(elemT)
tbl.rv.SetMapIndex(reflect.ValueOf(seg), next) tbl.rv.SetMapIndex(reflect.ValueOf(seg), next)
return &targetTable{rv: next, path: segAbs}, nil return &targetTable{rv: next, path: slices.Clone(segAbs)}, nil
} }
resolved := seg resolved := seg
loc, ok := tbl.schema.byName[seg] loc, ok := tbl.schema.byName[seg]
@@ -1135,17 +1307,21 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
loc, ok = tbl.schema.byName[resolved] loc, ok = tbl.schema.byName[resolved]
} }
if !ok { if !ok {
tbl.recordStrictUnknown(seg) tp.recordStrictUnknown(tbl, seg)
if opened, ok := tp.opened[pathKey(segAbs)]; ok { // A sink an array element created is entered by a [sub-table]
// header, through the last element, but never by a dotted key: the
// tree's descendKey rejects an array where tableAt follows it.
if opened, ok := tp.opened[pathKey(segAbs)]; ok && !opened.arrayElem {
return opened, nil return opened, nil
} }
if tp.tableHas(tbl, seg) { if tp.tableHas(tbl, seg) {
return nil, tp.errf("key %q is not a table", seg) return nil, tp.errf("key %q is not a table", seg)
} }
sink := &targetTable{sink: true, path: segAbs} sink := &targetTable{sink: true, path: slices.Clone(segAbs)}
tp.opened[pathKey(segAbs)] = sink tp.opened[pathKey(segAbs)] = sink
return sink, nil return sink, nil
} }
tp.markResolved(tbl, resolved)
fv, ferr := fieldByIndex(tbl.rv, loc.index) fv, ferr := fieldByIndex(tbl.rv, loc.index)
if ferr != nil { if ferr != nil {
return nil, errTargetFallback return nil, errTargetFallback
@@ -1167,7 +1343,7 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
} }
return nil, errTargetFallback return nil, errTargetFallback
} }
return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: segAbs, strict: tbl.strict}, nil return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: slices.Clone(segAbs), strict: tbl.strict}, nil
case reflect.Map: case reflect.Map:
if fv.Type().Key().Kind() != reflect.String { if fv.Type().Key().Kind() != reflect.String {
return nil, errTargetFallback return nil, errTargetFallback
@@ -1175,7 +1351,7 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
if fv.IsNil() { if fv.IsNil() {
fv.Set(reflect.MakeMap(fv.Type())) fv.Set(reflect.MakeMap(fv.Type()))
} }
return &targetTable{rv: fv, path: segAbs}, nil return &targetTable{rv: fv, path: slices.Clone(segAbs)}, nil
} }
if tp.tableHas(tbl, seg) { if tp.tableHas(tbl, seg) {
return nil, tp.errf("key %q is not a table", seg) return nil, tp.errf("key %q is not a table", seg)
@@ -1193,7 +1369,9 @@ func (tp *targetParser) leafInTable(dest *targetTable, key string) (dst reflect.
if dest.rv.Kind() != reflect.Map { if dest.rv.Kind() != reflect.Map {
return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback
} }
if dest.rv.MapIndex(reflect.ValueOf(key)).IsValid() { if tp.tableHas(dest, key) {
// The duplicate check reads the keys the document defined, not
// the destination map's own contents, which are the caller's.
return reflect.Value{}, reflect.Value{}, leafTable, false, tp.errf("duplicate key %q", key) return reflect.Value{}, reflect.Value{}, leafTable, false, tp.errf("duplicate key %q", key)
} }
elem := reflect.New(dest.rv.Type().Elem()).Elem() elem := reflect.New(dest.rv.Type().Elem()).Elem()
@@ -1206,10 +1384,10 @@ func (tp *targetParser) leafInTable(dest *targetTable, key string) (dst reflect.
loc, ok = dest.schema.byName[resolved] loc, ok = dest.schema.byName[resolved]
} }
if !ok { if !ok {
dest.recordStrictUnknown(key) tp.recordStrictUnknown(dest, key)
return reflect.Value{}, reflect.Value{}, leafTable, false, nil return reflect.Value{}, reflect.Value{}, leafTable, false, nil
} }
dest.markResolved(resolved) tp.markResolved(dest, resolved)
fv, ferr := fieldByIndex(dest.rv, loc.index) fv, ferr := fieldByIndex(dest.rv, loc.index)
if ferr != nil { if ferr != nil {
return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback
+350
View File
@@ -4,9 +4,12 @@
package interpres package interpres
import ( import (
"errors"
"maps"
"net" "net"
"reflect" "reflect"
"strings" "strings"
"sync/atomic"
"testing" "testing"
"time" "time"
) )
@@ -522,3 +525,350 @@ func TestTargetedMapTableShapes(t *testing.T) {
}) })
} }
} }
// TestTargetedNestedMapDescents pins the descents into a map of maps that
// meet entries the document built earlier: a dotted key twice through the
// same sub-table, a header into a dotted-built sub-table, and a typed array
// under a map key. Each shape once panicked on a reflect Elem of a map.
func TestTargetedNestedMapDescents(t *testing.T) {
t.Run("dotted key through one sub-table twice", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\nm.a.c = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["c"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("header under a dotted-built sub-table", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\n[m.a.deep]\nx = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["deep"].(map[string]any)["x"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("typed array under a map key", func(t *testing.T) {
var cfg struct {
M map[string][]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("[[m.arr]]\nx = 1\n\n[[m.arr]]\ny = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.M["arr"]) != 2 || cfg.M["arr"][1]["y"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
}
// TestTargetedPointerElementSlice pins that an array of tables over a slice
// of pointer elements fills the pointed-to structs.
func TestTargetedPointerElementSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items []*item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n\n[[items]]\nn = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 2 || cfg.Items[0] == nil || cfg.Items[1].N != 2 {
t.Errorf("items = %#v", cfg.Items)
}
}
// TestTargetedArrayScopeResets pins that a new element of an array of tables
// starts a fresh definition scope, the contract the changelog documents.
func TestTargetedArrayScopeResets(t *testing.T) {
doc := "[[a]]\nb.c = 1\n\n[[a]]\n\n[a.b]\nx = 1\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
if refErr != nil {
t.Fatalf("tree decode: %v", refErr)
}
if err := Unmarshal([]byte(doc), &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
}
// TestTargetedUnknownArrayElements pins that every element of an unknown
// array of tables is a fresh namespace, and a sub-table header reaches the
// last element the way the tree parser's does.
func TestTargetedUnknownArrayElements(t *testing.T) {
doc := "[[zz]]\nk = 1\n\n[[zz]]\nk = 2\n\n[zz.sub]\nx = 3\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
tgtErr := Unmarshal([]byte(doc), &tgt)
if (refErr == nil) != (tgtErr == nil) {
t.Fatalf("error presence disagrees: tree %v, targeted %v", refErr, tgtErr)
}
if refErr != nil {
return
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
// A dotted key may not enter the array: the tree's own rule.
var dotted targetCfg
dErr := Unmarshal([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
refDotted := treeDecodeInto([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
if (dErr == nil) != (refDotted == nil) {
t.Errorf("dotted into an array: targeted %v, tree %v", dErr, refDotted)
}
}
// TestTargetedFixedArrayUnderFill pins that a fixed-size array the document
// under-fills is the length mismatch the tree decode raises, with the
// field's path.
func TestTargetedFixedArrayUnderFill(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items [2]item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n"), &cfg)
if err == nil {
t.Fatal("unmarshal accepted an under-filled array")
}
want := `interpres: items: cannot assign 1 elements to [2]interpres.item`
if err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedPrefilledSliceReplaced pins that a prefilled slice is replaced
// by the document's elements on both paths, not appended to.
func TestTargetedPrefilledSliceReplaced(t *testing.T) {
type item struct {
N int `toml:"n"`
}
doc := []byte("[[items]]\nn = 1\n")
var ref struct {
Items []item `toml:"items"`
}
ref.Items = []item{{N: 9}}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
var tgt struct {
Items []item `toml:"items"`
}
tgt.Items = []item{{N: 9}}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if len(tgt.Items) != 1 || tgt.Items[0].N != 1 {
t.Errorf("items = %#v, want the prefilled element replaced", tgt.Items)
}
}
// TestTargetedHeaderOverValueArrayKeepsCase pins that a value array assigned
// under a differently cased key than the field's name still blocks the
// array-of-tables header over it, the tree parse error.
func TestTargetedHeaderOverValueArrayKeepsCase(t *testing.T) {
var cfg struct {
Arr []targetNested `toml:"arr"`
}
err := Unmarshal([]byte("Arr = [{x = 1}]\n[[Arr]]\nx = 2\n"), &cfg)
if err == nil || err.Error() != `interpres: line 2: key "Arr" is not an array of tables` {
t.Errorf("err = %v, want the parse error over the assigned field", err)
}
}
// TestTargetedDottedInlineFreezePath pins that an inline table assigned by a
// dotted key freezes the whole path the statement wrote: a later header
// under that path is the extension error, and a key outside it stays free.
func TestTargetedDottedInlineFreezePath(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("m.a.b = {x = 1}\nb.y = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
err = Unmarshal([]byte("m.a.b = {x = 1}\n[m.a.b]\ny = 2\n"), &cfg)
want := `interpres: line 2: cannot extend inline table "m.a.b"`
if err == nil || err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedStrictThroughDottedKeys pins that strict and required findings
// survive the transient tables a dotted descent builds.
func TestTargetedStrictThroughDottedKeys(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("tab.zz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the strict failure through the dotted key", err)
}
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the path through the dotted key", err)
}
}
// TestTargetedRequiredThroughDottedKeys pins that a required tag is honoured
// when the table is reached only through dotted keys.
func TestTargetedRequiredThroughDottedKeys(t *testing.T) {
type nested struct {
X int `toml:"x,required"`
Y int `toml:"y"`
}
var cfg struct {
Tab nested `toml:"tab"`
}
err := Unmarshal([]byte("tab.y = 1\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), `missing required key "x"`) {
t.Errorf("err = %v, want the missing required key through the dotted key", err)
}
}
// TestTargetedOrderedMapSliceFallsBack pins that a slice of OrderedMap
// elements takes the tree path, whose fill keeps the written order.
func TestTargetedOrderedMapSliceFallsBack(t *testing.T) {
var cfg struct {
Items []OrderedMap `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nk = \"v\"\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 1 || cfg.Items[0].Keys()[0] != "k" {
t.Errorf("items = %#v, want the element filled in written order", cfg.Items)
}
}
// hookMap is a named map type whose decode hook counts its calls.
type hookMap map[string]any
var hookMapCalls atomic.Int32
func (h *hookMap) UnmarshalTOML(data any) error {
hookMapCalls.Add(1)
m, _ := data.(map[string]any)
if *h == nil {
*h = hookMap{}
}
maps.Copy((*h), m)
return nil
}
// TestTargetedMapFieldHookGetsWholeTable pins that a named map field with a
// decode hook receives the whole parsed table, even in its header form.
func TestTargetedMapFieldHookGetsWholeTable(t *testing.T) {
type cfg struct {
M hookMap `toml:"m"`
}
var c cfg
hookMapCalls.Store(0)
err := Unmarshal([]byte("[m]\na = 1\nb = 2\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if hookMapCalls.Load() != 1 {
t.Errorf("hook calls = %d, want exactly one with the whole table", hookMapCalls.Load())
}
if c.M["a"] != int64(1) || c.M["b"] != int64(2) {
t.Errorf("m = %#v", c.M)
}
}
// errHook fails every decode with a fixed error and counts its calls.
type errHook struct{ calls *int }
func (e *errHook) UnmarshalTOML(any) error {
if e.calls != nil {
*e.calls++
}
return errors.New("boom")
}
// TestTargetedHookErrorRunsOnce pins that a failing hook's error is the
// tree path's own, wrapped with the key, and that the hook is not run a
// second time by a fallback.
func TestTargetedHookErrorRunsOnce(t *testing.T) {
calls := 0
cfg := struct {
F errHook `toml:"f"`
}{F: errHook{calls: &calls}}
err := Unmarshal([]byte("f = 1\n"), &cfg)
if err == nil || err.Error() != "interpres: f: unmarshal: boom" {
t.Errorf("err = %v, want the wrapped hook failure", err)
}
if calls != 1 {
t.Errorf("hook calls = %d, want one", calls)
}
}
// TestTargetedUnknownBeforeRequired pins the report order the tree decode
// produces: an unknown key wins over a missing required one.
func TestTargetedUnknownBeforeRequired(t *testing.T) {
type inner struct {
X int `toml:"x,required"`
}
var cfg struct {
Tab inner `toml:"tab"`
}
err := Unmarshal([]byte("[tab]\nzz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the unknown key reported before the required one", err)
}
}
// TestTargetedStrictPathStableAcrossHeaders pins that the path a strict
// finding wraps does not alias the parser's key buffer: the table that owns
// the unknown key keeps its name after a later header.
func TestTargetedStrictPathStableAcrossHeaders(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("[tab]\nzz = 1\n\n[lims]\nx = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the finding on tab, not the later header", err)
}
}
// TestTargetedPrefilledMapFieldMergesUnderHeader pins that a prefilled map
// field merges the document's header-form table into it on both paths, the
// rule the root map has always followed.
func TestTargetedPrefilledMapFieldMergesUnderHeader(t *testing.T) {
doc := []byte("[lims]\nnew = 3\n")
var ref, tgt targetCfg
ref.Lims = map[string]any{"keep": "yes"}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
tgt.Lims = map[string]any{"keep": "yes"}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if tgt.Lims["keep"] != "yes" || tgt.Lims["new"] != int64(3) {
t.Errorf("lims = %#v, want the merge", tgt.Lims)
}
}
// TestTargetedNumberTokenValidatesUTF8 pins that the token route the
// targeted parse takes reports invalid UTF-8 with the scanner's own message
// and position.
func TestTargetedNumberTokenValidatesUTF8(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("num = 12\xff\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), "invalid UTF-8 in value at byte offset 8") {
t.Errorf("err = %v, want the UTF-8 complaint on the invalid byte", err)
}
}