17 Commits
Author SHA1 Message Date
petrbalvin e3dda5e115 fix: take the release-check version from the recipe argument
Test / test (push) Successful in 1m46s
Release / gates (push) Successful in 1m36s
Release / release (push) Successful in 39s
2026-09-22 21:49:05 +02:00
petrbalvin e35b4bb90d fix: take the release-check version from the recipe argument
Test / test (push) Canceled after 44s
2026-09-22 21:48:00 +02:00
petrbalvin 75ade89f34 chore: prepare release v2.0.0
Test / test (push) Successful in 2m3s
2026-09-22 21:45:25 +02:00
petrbalvin 41786c5bc6 chore: drop the internal reference from the justfile header 2026-09-22 21:45:25 +02:00
petrbalvin 010a7b2a1e ci: pin the actions by version tag and cap the test timeout at 10m 2026-09-22 21:45:25 +02:00
petrbalvin eb6ac1ab9d ci: drop the schedule triggers, race and fuzz run on dispatch alone
Test / test (push) Successful in 1m54s
Assisted-by: GLM 5.3
2026-09-22 21:28:30 +02:00
petrbalvin b0f40739c7 chore: keep the coverage floor at the documented 80 percent
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin fba53440c7 ci: state where the nightly race and fuzz sweeps run
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 332cd01d44 chore: check the suite colour before comparing documented counts
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 2fa075de00 docs: align every document with the reviewed behaviour
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 4900367970 fix(cmd): long-form flags, honest counts and safer inference
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin b7f39435e1 fix(document): rebuild set nodes and write documents back round-trip
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 0ded34da3c fix(encode): pointer table arrays, whole-minute offsets and emission checks
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin b45f4d65da fix(decode): keep the targeted parse on the tree path's contract
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin f7e427ae3f fix(decode): allocate embedded pointer maps and settle case collisions
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin e677e34508 fix(api): one statement per value array, parseas options and bounded reads
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 2efdb2d059 fix(parse): reject the lenient grammar edges and name out-of-range date-times
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
35 changed files with 2791 additions and 805 deletions
+7 -11
View File
@@ -1,20 +1,16 @@
# Fuzz smoke, Go. A nightly time-boxed run of the fuzz targets, and a hand
# dispatch when a change asks for it.
# Fuzz smoke, Go. Dispatched by hand when a change asks for it.
#
# Fuzzing is exploration, so it never belongs to the push pipeline; a 30 second
# smoke per target, run nightly, is the compromise that catches a new crash
# within a day without holding the shared box. The targets run the seeds and
# whatever the corpus has gathered; a failure leaves its crashing input in
# testdata/fuzz, which the ordinary suite then reproduces on every push.
# smoke per target on a hand dispatch checks a change without holding the
# shared box. The targets run the seeds and whatever the corpus has gathered; a
# failure leaves its crashing input in testdata/fuzz, which the ordinary suite
# then reproduces on every push.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Fuzz
on:
workflow_dispatch:
schedule:
# Nightly at 03:30 UTC, after the race sweep has had the box first.
- cron: "30 3 * * *"
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
@@ -26,9 +22,9 @@ jobs:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: actions/checkout@v7
- uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
+8 -12
View File
@@ -1,20 +1,16 @@
# Race, Go. Dispatched by hand, run nightly on a schedule, and run as part of
# the release gates.
# Race, Go. Dispatched by hand.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is an explicit decision rather than a routine. The schedule is the
# nightly sweep: a failing race on development is known by morning without anyone
# remembering to dispatch it.
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once
# per task; here it is an explicit decision rather than a routine, a hand dispatch
# when a change asks for one. Development carries its race gate on every push through
# that local gate.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
on:
workflow_dispatch:
schedule:
# Nightly at 03:00 UTC, the quietest hours of the shared box.
- cron: "0 3 * * *"
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
@@ -26,9 +22,9 @@ jobs:
runs-on: fedora
timeout-minutes: 45
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: actions/checkout@v7
- uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
@@ -38,4 +34,4 @@ jobs:
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 30m ./...
run: go test -race -count=1 -timeout 10m ./...
+6 -6
View File
@@ -4,8 +4,8 @@
# carries the CHANGELOG section as its body and nothing else. The gates still run first,
# in their own job and once, minus the race detector: race never runs on a push path or a
# tag, and the local gate raced this tree before the tag was cut. The write permission
# sits on the release job alone, and the version contract these steps implement is in the
# `release` skill.
# sits on the release job alone, and the version the binary reports is the one the
# toolchain records from the tag, with nothing injected.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
@@ -31,9 +31,9 @@ jobs:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: actions/checkout@v7
- uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
@@ -77,7 +77,7 @@ jobs:
- name: Tests
# Keep the pattern equal to `packages` in the project's justfile.
run: go test -count=1 -timeout 30m -coverprofile=coverage.out ./...
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./...
- name: Coverage floor
run: |
@@ -103,7 +103,7 @@ jobs:
contents: read
releases: write
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: actions/checkout@v7
- name: Install Perl
# The runner images are minimal and Perl is not guaranteed. The install is a
+6 -5
View File
@@ -1,8 +1,9 @@
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so race runs once inside
# the release pipeline instead. The box is one core and 2 GB beside Gitea, so
# runner box cannot afford the race detector on every push. Race has its own
# pipeline, dispatched by hand, and the local `just gates` runs it once per
# task. The box is one core and 2 GB beside Gitea, so
# parallelism is bounded on purpose and everything runs in one job. Extra jobs would
# duplicate the checkout, the Go setup and the dependency download three times without
# buying any parallelism.
@@ -43,9 +44,9 @@ jobs:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- uses: actions/checkout@v7
- uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6
- uses: actions/setup-go@v6
with:
# The module is the source of truth for the version, so it cannot drift.
go-version-file: go.mod
@@ -79,7 +80,7 @@ jobs:
- name: Tests
# Scope the pattern to the packages that hold the logic when a thin cmd/ drags the
# total under the floor, and keep it equal to `packages` in the project's justfile.
run: go test -count=1 -timeout 30m -coverprofile=coverage.out ./...
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./...
- name: Coverage floor
run: |
+1 -1
View File
@@ -1,7 +1,7 @@
.idea/
.zcode/
# Build artifacts
# Build artefacts
bin/
*.test
*.out
+34 -13
View File
@@ -9,6 +9,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
-
## [2.0.0] - 2026-09-22
### Added
- `encoding.TextMarshaler` and `encoding.TextUnmarshaler` are honoured by
default, with no option to switch them off. A type that implements them is
encoded as a TOML string and decoded from one: `net.IP` becomes
@@ -20,7 +26,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- `time.Duration` is encoded in its canonical Go form as a TOML string,
`1h30m0s`, because TOML has no duration type; the decoder reads that string
back and still accepts a bare integer as the nanosecond count.
- `interpres-decode -encode`, the adapter's other direction: it reads the
- `interpres-decode --encode`, the adapter's other direction: it reads the
toml-test tagged JSON from stdin and writes the TOML document it describes.
The compliance suite now runs the encoder as well as the decoder, 214
encoder cases against the tagged JSON of the valid corpus.
@@ -47,8 +53,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
the plain type takes an offset date-time as it always did; code that asserts
the tree's type, and `UnmarshalTOML` implementations that expect a
`time.Time`, need the new type.
- `MaxNestingDepth(depth)` and `MaxInputSize(size)` options bound the parse a
`Decode` performs, and every parse carries a nesting limit in any case
- `MaxNestingDepth(depth)` and `MaxInputSize(size)` options bound the parse an
`Unmarshal` performs, and every parse carries a nesting limit in any case
(10000 levels, which no hand-written document approaches): a document that
nests arrays or inline tables deeper used to run the stack out and is now
rejected with a `SyntaxError` naming the limit.
@@ -57,7 +63,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
header as one statement with its node, an `[[array of tables]]` as one
statement per element. A caller that breaks after the statement it wanted
reads no further ones. `examples/statements` shows the walk.
- `ParseAs[T](data)`, the generic one-line decode, and `NewSchema[T]()`,
- `ParseAs[T](data, opts...)`, the generic one-line decode, and `NewSchema[T]()`,
which precompiles the struct schema and the interface flags for a hot path
before the first document arrives.
- `EmitFieldComments(true)` prints the comment a field's `toml` tag
@@ -101,26 +107,26 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
is an error wrapped with the key path.
- `MarshalAppend(buf, v)` appends the TOML encoding of v to buf and returns
the extended buffer, the shape `json.MarshalAppend` has.
- `interpres-decode -version` prints the binary's version, the module version
- `interpres-decode --version` prints the binary's version, the module version
the toolchain recorded, so a release-built binary names its own tag.
- `interpres-decode -json` prints plain indented JSON instead of the tagged
- `interpres-decode --json` prints plain indented JSON instead of the tagged
form, the shape for people and diffs, with the date-time wrappers in their
TOML form.
- `interpres-decode -validate` walks a named directory for `.toml` files and
- `interpres-decode --validate` walks a named directory for `.toml` files and
closes the sweep with a summary naming how many documents were checked and
how many were invalid; single files stay quiet on success as before.
- `interpres-decode -struct` infers a Go struct definition from a document:
- `interpres-decode --struct` infers a Go struct definition from a document:
one field per key in written order, nested tables as nested struct types,
an array of tables as a slice. The printed type compiles and decodes the
document it came from.
- `interpres-decode -schema TYPE file.go` writes a TOML template for the
- `interpres-decode --schema TYPE file.go` writes a TOML template for the
named struct type of a Go source, the `comment=` tag option printed as a
comment and the `default=` option as the value. It is the inverse of
`-struct`.
`--struct`.
- `ParseFile(path)` reads the file and parses it into a `Document`, with the
file name at the front of every error it returns, read failure and parse
failure alike. `Valid(data)` reports whether a document parses, nil on
success and the parse error on failure, the library call the `-validate`
success and the parse error on failure, the library call the `--validate`
mode of interpres-decode is built on.
- `SyntaxError` carries the byte `Offset` the scan stopped at and the 1-based
`Column` on the line, beside the line it always had, and `SourceLine(src)`
@@ -155,7 +161,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
interface, or a nil or empty slice, array or map. In 1.x the option covered
only the collections.
- The `toml` tag gained the `inline` option: a struct or map field tagged
`toml:"retry,inline"` emits as `retry = {…}` instead of a header section,
`toml:"retry,inline"` emits as `retry = {…}` instead of a header section,
whatever its size, a named embedded struct included. Forcing it on an array
of tables is an error, because the inline form would re-parse as a value
array and change the value's Go type.
@@ -239,6 +245,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- A top-level value the encoder could not normalise reported its path with
a leading dot, `interpres: .port: ...`; the message now reads
`interpres: port: ...`, the shape `EncodeError.Path` already used.
- A token shaped like a date-time with a component out of range, such as an
hour of 24 or a February the 30th, fell through to the number decoder and
failed with the number complaint `invalid character "-" in number`; it now
fails as the date-time it visibly is, `invalid date-time "..."`.
- A `time.Time` or `OffsetDateTime` whose zone offset is not a whole number
of minutes wrote only the minutes, silently shifting the instant by the
seconds dropped; the encoder now refuses such an offset, which TOML has no
form for, instead of corrupting the value.
- An empty array of tables over pointer elements, `[]*T{}`, emitted as
`key = []` while its value form was omitted; it is omitted too now, the
rule TOML forces, because an empty `[[a]]` has no valid form.
- Two lenient grammar edges are closed: a sign in a `\u` or `\U` escape,
which is not a hex digit, is rejected instead of evaluating, and a bare
carriage return right after a multi-line string's opening delimiter is
the bare-CR error instead of a newline trimmed silently.
### Migration from 1.x
@@ -313,7 +334,7 @@ except where noted above.
comments and trailing commas. The compliance suite runs in TOML 1.1 mode:
214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document
parses exactly as before.
- `interpres-decode -validate [file ...]`: a validate mode beside the
- `interpres-decode --validate [file ...]`: a validate mode beside the
toml-test adapter. It parses each named file, or stdin when none are named,
prints one line per invalid document to stderr, and exits 0 when all are
valid, 1 when one is not, and 2 on a usage or read failure. Install it with
+1
View File
@@ -116,6 +116,7 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
|---|---|---|
| Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite |
| Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs |
| Fuzz | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector |
The local equivalent is `just gates`, which is the same set plus the race
+7 -5
View File
@@ -13,16 +13,17 @@ the entire official [toml-test](https://github.com/toml-lang/toml-test) suite:
`\xHH` escapes; integers in the four radixes with `_` separators; floats with
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds
optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and
`Marshal` for structs and maps, mirroring `encoding/json`.
- **Decoding and encoding**: `Unmarshal` and `Marshal` for structs and maps,
mirroring `encoding/json`; `Parse` and `ParseMap` for the document with its
key order and the plain untyped tree.
- **Strict decoding**: `RejectUnknownFields(true)` rejects keys that
match no destination field, at every struct depth.
- **Custom types**: `Marshaler` and `Unmarshaler` let a type control its own
TOML representation in both directions, and `encoding.TextMarshaler` and
`TextUnmarshaler` are honoured by default, so `net.IP`, `time.Duration` and
user types with text methods need no configuration.
- **Cancellation**: every entry point has a `*Context` sibling that honours a
`context.Context`.
- **Cancellation**: the parse, decode and marshal entries have `*Context`
siblings that honour a `context.Context`, checked while the work runs.
- **Ordered documents**: `Parse` gives a `*Document` that keeps the key order,
tells an inline table from a header one, and carries the comments; `ParseMap`
gives the plain `map[string]any` tree.
@@ -173,7 +174,8 @@ See [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the full workflow, and
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/API.md](docs/API.md): the API reference, decoding and encoding rules
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator,
also shipped as the manual page `man/interpres-decode.1`
## Licence
+89
View File
@@ -0,0 +1,89 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"errors"
"strings"
"testing"
)
// errReader fails every read with a fixed error.
type errReader struct{ err error }
func (r errReader) Read([]byte) (int, error) { return 0, r.err }
// TestUnmarshalRead covers the streaming entry: the happy path with options,
// a failing reader, and MaxInputSize bounding what a reader is drained into.
func TestUnmarshalRead(t *testing.T) {
var got struct {
Name string `toml:"name"`
N int `toml:"n"`
}
err := UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, RejectUnknownFields(true))
if err != nil {
t.Fatalf("UnmarshalRead: %v", err)
}
if got.Name != "x" {
t.Errorf("Name = %q", got.Name)
}
readErr := errors.New("boom")
if err := UnmarshalRead(errReader{readErr}, &got); !errors.Is(err, readErr) {
t.Errorf("err = %v, want the read error wrapped", err)
}
err = UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, MaxInputSize(4))
if err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit", err)
}
// The limit bounds the read itself: a reader that would supply far more
// than the limit is not drained into memory first.
big := strings.Repeat("x", 1<<20)
if err := UnmarshalRead(strings.NewReader(big), &got, MaxInputSize(16)); err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit before the read completes", err)
}
}
// TestParseAsWithOptions covers the generic shorthand carrying options.
func TestParseAsWithOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name"`
}
got, err := ParseAs[cfg]([]byte("name = \"x\"\nrogue = 1\n"), RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), "unknown field") {
t.Errorf("err = %v, want the strict failure", err)
}
// The statements before the failure stay written, the contract the
// targeted path documents and encoding/json follows.
if got.Name != "x" {
t.Errorf("Name = %q, want the statement before the failure kept", got.Name)
}
}
// TestStatementsValueArrays pins that a value array is one statement, a
// scalar array and an array of inline tables alike; only an array of tables
// yields per element.
func TestStatementsValueArrays(t *testing.T) {
src := strings.NewReader("port = [8080, 9090]\nmix = [{y = 1, x = 2}]\n[[items]]\nn = 1\n")
var got []Statement
for stmt, err := range Statements(src) {
if err != nil {
t.Fatal(err)
}
got = append(got, stmt)
}
if len(got) != 3 {
t.Fatalf("got %d statements, want 3", len(got))
}
if got[0].Index != -1 || got[0].Table != nil {
t.Errorf("port statement = %+v, want one plain key/value", got[0])
}
if got[1].Index != -1 || got[1].Table != nil {
t.Errorf("mix statement = %+v, want one plain key/value", got[1])
}
if got[2].Index != 0 || got[2].Table == nil {
t.Errorf("items statement = %+v, want the element with its node", got[2])
}
}
+109 -51
View File
@@ -6,9 +6,11 @@ package main
import (
"fmt"
"io"
"strconv"
"strings"
"time"
"unicode"
"unicode/utf8"
"sourcedock.dev/petrbalvin/interpres/v2"
)
@@ -17,63 +19,115 @@ import (
// like the document: one field per key in written order, nested tables as
// nested struct types, an array of tables as a slice, and the field names
// invented from the keys. It is the onboarding aid: the printed type compiles
// and decodes the document it came from.
// and decodes the document it came from. The definition is built whole and
// written with a single call, so a failing standard output surfaces as one
// error instead of being dropped mid-print.
func inferStruct(data []byte, stdout io.Writer) error {
doc, err := interpres.Parse(data)
if err != nil {
return err
}
fmt.Fprintln(stdout, "// Generated by interpres-decode -struct; decode with")
fmt.Fprintln(stdout, "// sourcedock.dev/petrbalvin/interpres/v2.")
fmt.Fprintln(stdout, "type inferred struct {")
if err := writeInferredFields(stdout, doc.Root(), map[string]bool{}); err != nil {
return err
}
fmt.Fprintln(stdout, "}")
return nil
body := &strings.Builder{}
fmt.Fprintln(body, "// Generated by interpres-decode --struct; decode with")
fmt.Fprintln(body, "// sourcedock.dev/petrbalvin/interpres/v2.")
fmt.Fprintln(body, "type inferred struct {")
writeInferredFields(body, tableFields(doc.Root()), map[string]bool{})
fmt.Fprintln(body, "}")
_, err = io.WriteString(stdout, body.String())
return err
}
// writeInferredFields writes one field per entry of the table. invented
// tracks the field names already used at one level, so two keys that clean
// to the same name do not collide.
func writeInferredFields(w io.Writer, t *interpres.Table, invented map[string]bool) error {
// inferredField is one document key with the entry it is inferred from.
type inferredField struct {
key string
entry *interpres.Entry
}
// tableFields lists a table's entries in written order.
func tableFields(t *interpres.Table) []inferredField {
out := make([]inferredField, 0, len(t.Keys()))
for _, key := range t.Keys() {
entry, _ := t.Get(key)
name := goFieldName(key, invented)
// An array of tables carries a node per element; the nodes of a value
// array are nil wherever an element is not a table, so the nils give
// it away.
var tables []*interpres.Table
for _, el := range entry.Elements() {
if el != nil {
tables = append(tables, el)
}
}
if len(tables) > 0 {
// The type comes from the first element.
fmt.Fprintf(w, "\t%s []struct {\n", name)
if err := writeInferredFields(w, tables[0], map[string]bool{}); err != nil {
return err
}
fmt.Fprintf(w, "\t} `toml:%q`\n", key)
continue
}
val := entry.Value()
if child := entry.Table(); child != nil {
fmt.Fprintf(w, "\t%s struct {\n", name)
if err := writeInferredFields(w, child, map[string]bool{}); err != nil {
return err
}
fmt.Fprintf(w, "\t} `toml:%q`\n", key)
continue
}
if items, ok := val.([]any); ok {
fmt.Fprintf(w, "\t%s []%s `toml:%q`\n", name, inferScalarType(items), key)
continue
}
fmt.Fprintf(w, "\t%s %s `toml:%q`\n", name, goTypeOf(val), key)
out = append(out, inferredField{key: key, entry: entry})
}
return nil
return out
}
// mergedTableFields merges the key sets of an array's elements in first-seen
// order. An array's type has to cover every element, and a key may appear
// only in a later one, so the first element alone does not decide the shape;
// each key is inferred from the first element that carries it.
func mergedTableFields(tables []*interpres.Table) []inferredField {
var out []inferredField
seen := map[string]bool{}
for _, t := range tables {
for _, f := range tableFields(t) {
if seen[f.key] {
continue
}
seen[f.key] = true
out = append(out, f)
}
}
return out
}
// writeInferredFields writes one field per entry, in the order given.
// invented tracks the field names already used at one level, so two keys
// that clean to the same name do not collide.
func writeInferredFields(w *strings.Builder, fields []inferredField, invented map[string]bool) {
for _, f := range fields {
writeInferredField(w, f, invented)
}
}
// writeInferredField writes one field for one entry: an array of tables as a
// slice of structs, a child table as a nested struct, and everything else as
// the scalar or slice the decoded value names.
func writeInferredField(w *strings.Builder, f inferredField, invented map[string]bool) {
name := goFieldName(f.key, invented)
// An array of tables carries a node per element; the nodes of a value
// array are nil wherever an element is not a table. Every node present
// is what tells the two apart: [1, {x=1}] stays a value array even
// though one of its elements is a table.
elements := f.entry.Elements()
allTables := len(elements) > 0
for _, el := range elements {
if el == nil {
allTables = false
break
}
}
if allTables {
fmt.Fprintf(w, "\t%s []struct {\n", name)
writeInferredFields(w, mergedTableFields(elements), map[string]bool{})
fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
}
if child := f.entry.Table(); child != nil {
fmt.Fprintf(w, "\t%s struct {\n", name)
writeInferredFields(w, tableFields(child), map[string]bool{})
fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
}
val := f.entry.Value()
if items, ok := val.([]any); ok {
fmt.Fprintf(w, "\t%s []%s %s\n", name, inferScalarType(items), structTag(f.key))
return
}
fmt.Fprintf(w, "\t%s %s %s\n", name, goTypeOf(val), structTag(f.key))
}
// structTag renders the toml tag of one key as a Go string literal. The raw
// backtick literal is the conventional shape, but a key carrying a backtick
// would end that literal early and the printed definition would not compile,
// so such tags are rendered with strconv.Quote instead.
func structTag(key string) string {
tag := `toml:"` + key + `"`
if !strings.ContainsAny(tag, "`\r") {
return "`" + tag + "`"
}
return strconv.Quote(tag)
}
// goTypeOf names the Go type the decoded value asks for.
@@ -106,8 +160,10 @@ func goTypeOf(val any) string {
}
// goFieldName cleans a document key into an exported Go identifier: the
// words the punctuation splits become capitalised runs, a leading digit gains
// an underscore, and a collision with an earlier name gains a counter.
// words the punctuation splits become capitalised runs, a leading digit
// gains a Field prefix, because an underscore would leave the field
// unexported and the decoder would skip it, and a collision with an earlier
// name gains a counter.
func goFieldName(key string, invented map[string]bool) string {
var b strings.Builder
nextUpper := true
@@ -127,8 +183,10 @@ func goFieldName(key string, invented map[string]bool) string {
if name == "" {
name = "Field"
}
if unicode.IsDigit(rune(name[0])) {
name = "_" + name
// The first rune is decoded rather than taken as a byte, because a key
// may open with a digit beyond ASCII.
if first, _ := utf8.DecodeRuneInString(name); unicode.IsDigit(first) {
name = "Field" + name
}
for invented[name] {
name += "2"
+74 -32
View File
@@ -4,16 +4,16 @@
// Command interpres-decode is the toml-test harness adapter and a TOML
// validator. Without flags it reads a TOML document from standard input and
// writes the toml-test "tagged JSON" representation to standard output. With
// -encode it is the reverse: it reads tagged JSON and writes the TOML document
// it describes. With -validate it checks the named documents, or standard
// input when none are named, and exits non-zero on the first invalid one:
// --encode it is the reverse: it reads tagged JSON and writes the TOML document
// it describes. With --validate it checks the named documents, or standard
// input when none are named, and exits non-zero when one is invalid:
//
// interpres-decode -validate config.toml
// interpres-decode -encode < case.json
// interpres-decode --validate config.toml
// interpres-decode --encode < case.json
//
// Run the official suite in both directions against the adapter with:
//
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode -encode'
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode --encode'
package main
import (
@@ -39,11 +39,16 @@ func main() {
}
// Run runs the command line and returns the process exit code: 0 success,
// 1 an invalid document, 2 a usage, reading, encoding, or
// 1 an invalid document, 2 a usage, reading, writing, encoding, or
// unsupported-value error.
func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError)
fs.SetOutput(stderr)
// The flag package's own diagnostics and default usage render flags
// with a single dash, while the command spells every flag in its
// two-dash long form, the form the manpage documents. Its output is
// therefore discarded and the usage below is the only one printed.
fs.SetOutput(io.Discard)
fs.Usage = func() {}
version := fs.Bool("version", false, "print the version and exit")
validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON")
encode := fs.Bool("encode", false, "read tagged JSON from stdin and write TOML instead")
@@ -52,12 +57,18 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
schemaType := fs.String("schema", "", "write a TOML template for the named struct type; the source file follows as the first argument")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
usage(stdout)
return 0
}
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
usage(stderr)
return 2
}
if *version {
fmt.Fprintf(stdout, "interpres-decode %s\n", versionString())
if _, err := fmt.Fprintf(stdout, "interpres-decode %s\n", versionString()); err != nil {
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0
}
modes := 0
@@ -70,13 +81,19 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
modes++
}
if modes > 1 {
fmt.Fprintln(stderr, "interpres-decode: -validate, -encode, -struct and -schema cannot be combined")
fmt.Fprintln(stderr, "interpres-decode: --validate, --encode, --struct and --schema cannot be combined")
return 2
}
// --json shapes the decoding output only, so it is rejected with every
// mode uniformly instead of being silently ignored by some of them.
if *plainJSON && modes > 0 {
fmt.Fprintln(stderr, "interpres-decode: --json shapes the decoder output and cannot be combined with --encode, --struct, --validate or --schema")
return 2
}
if *schemaType != "" {
rest := fs.Args()
if len(rest) != 1 {
fmt.Fprintln(stderr, "interpres-decode: -schema needs the type name and exactly one Go source file")
fmt.Fprintln(stderr, "interpres-decode: --schema needs the type name and exactly one Go source file")
return 2
}
if err := runSchema(*schemaType, rest[0], stdout); err != nil {
@@ -88,12 +105,8 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
if *validate {
return validatePaths(fs.Args(), stdin, stderr)
}
if *encode && *plainJSON {
fmt.Fprintln(stderr, "interpres-decode: -json shapes the decoder output and cannot be combined with -encode")
return 2
}
if fs.NArg() > 0 {
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with -validate")
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with --validate")
return 2
}
if *encode {
@@ -101,19 +114,25 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
}
data, err := io.ReadAll(stdin)
if err != nil {
fmt.Fprintln(stderr, "read stdin:", err)
fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2
}
if *infer {
if err := inferStruct(data, stdout); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 1
// A document that fails to parse keeps the adapter's invalid
// exit; anything else, a failed write among them, is a tool
// failure.
if _, ok := errors.AsType[*interpres.SyntaxError](err); ok {
return 1
}
return 2
}
return 0
}
tree, err := interpres.ParseMap(data)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 1
}
if *plainJSON {
@@ -121,25 +140,48 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
enc.SetEscapeHTML(false)
enc.SetIndent("", " ")
if err := enc.Encode(plainJSONValue(tree)); err != nil {
fmt.Fprintln(stderr, "encode:", err)
fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2
}
return 0
}
tagged, err := tag(tree)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
enc := json.NewEncoder(stdout)
enc.SetEscapeHTML(false)
if err := enc.Encode(tagged); err != nil {
fmt.Fprintln(stderr, "encode:", err)
fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2
}
return 0
}
// usage prints the command line summary, with every flag in its two-dash
// long form: the flag package's default usage printer renders a single dash,
// and the manpage and docs/CLI.md spell the flags the way this text does.
func usage(w io.Writer) {
fmt.Fprint(w, `Usage: interpres-decode [flags]
Without a mode flag the command reads one TOML document from standard input
and writes the toml-test tagged JSON representation to standard output.
--encode read tagged JSON from standard input and write TOML
instead
--help print this usage
--json with the default mode, print plain indented JSON
instead of tagged JSON
--schema TYPE write a TOML template for the named struct type; the
Go source file follows as the first argument
--struct infer a Go struct definition from the document on
standard input and print it
--validate validate the documents instead of emitting tagged JSON
--version print the version and exit
`)
}
// versionString names the version the binary was built at: the module
// version the toolchain recorded, which is the tag when the release pipeline
// builds it, and (devel) for an ordinary build from a working tree.
@@ -226,8 +268,8 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
return 2
}
}
valid := true
checked := 0
invalid := 0
for _, p := range files {
name := p
var data []byte
@@ -244,17 +286,17 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
}
checked++
if _, err := interpres.ParseMap(data); err != nil {
fmt.Fprintf(stderr, "%s: %v\n", name, err)
valid = false
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
invalid++
}
}
// The single-document run stays quiet on success, the contract the
// compliance tooling relies on; a directory walk closes with the
// summary that makes the sweep readable.
if dirs > 0 {
fmt.Fprintf(stderr, "checked %d documents, %d invalid\n", checked, map[bool]int{true: 0, false: 1}[valid])
fmt.Fprintf(stderr, "checked %d documents, %d invalid\n", checked, invalid)
}
if !valid {
if invalid > 0 {
return 1
}
return 0
@@ -265,17 +307,17 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
data, err := io.ReadAll(stdin)
if err != nil {
fmt.Fprintln(stderr, "read stdin:", err)
fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2
}
var desc any
if err := json.Unmarshal(data, &desc); err != nil {
fmt.Fprintln(stderr, "decode JSON:", err)
fmt.Fprintln(stderr, "interpres-decode: decode JSON:", err)
return 2
}
tree, err := untag(desc)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
doc, ok := tree.(map[string]any)
@@ -285,11 +327,11 @@ func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
}
out, err := interpres.Marshal(doc)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
if _, err := stdout.Write(out); err != nil {
fmt.Fprintln(stderr, "write stdout:", err)
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0
+412 -28
View File
@@ -7,6 +7,8 @@ import (
"bytes"
"encoding/json"
"errors"
"go/parser"
"go/token"
"os"
"path/filepath"
"reflect"
@@ -217,7 +219,7 @@ func TestTaggedHelper(t *testing.T) {
func TestValidateStdinAcceptsValidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"ok\"\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 0 {
if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
@@ -228,7 +230,7 @@ func TestValidateStdinAcceptsValidDocument(t *testing.T) {
func TestValidateStdinRejectsInvalidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"unterminated\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 1 {
if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 1 {
t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") {
@@ -250,10 +252,10 @@ func TestValidateFiles(t *testing.T) {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", good}, nil, &stdout, &stderr); code != 0 {
if code := Run([]string{"--validate", good}, nil, &stdout, &stderr); code != 0 {
t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String())
}
if code := Run([]string{"-validate", good, bad}, nil, &stdout, &stderr); code != 1 {
if code := Run([]string{"--validate", good, bad}, nil, &stdout, &stderr); code != 1 {
t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") {
@@ -263,7 +265,7 @@ func TestValidateFiles(t *testing.T) {
func TestValidateMissingFileReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
if code := Run([]string{"--validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
@@ -274,14 +276,14 @@ func TestAdapterModeRejectsPositionalArgument(t *testing.T) {
if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "-validate") {
t.Fatalf("stderr = %q, want it to point at -validate", stderr.String())
if !strings.Contains(stderr.String(), "--validate") {
t.Fatalf("stderr = %q, want it to point at --validate", stderr.String())
}
}
func TestUnknownFlagReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-nope"}, nil, &stdout, &stderr); code != 2 {
if code := Run([]string{"--nope"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
@@ -303,7 +305,7 @@ func TestRunEncoderScalars(t *testing.T) {
}
`
var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
@@ -334,7 +336,7 @@ func TestRunEncoderNested(t *testing.T) {
}
`
var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
@@ -355,7 +357,7 @@ func TestRunEncoderFloatTagDecides(t *testing.T) {
// tag decides the type; the output must stay a float.
var stdout, stderr bytes.Buffer
in := `{"whole": {"type": "float", "value": "1"}, "exp": {"type": "float", "value": "5e+22"}}`
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
@@ -380,7 +382,7 @@ func TestRunEncoderRejectsBadInput(t *testing.T) {
}
for _, c := range cases {
var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(c.in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(c.in), &stdout, &stderr)
if code != 2 {
t.Errorf("%s: Run returned %d, want 2; stderr = %q", c.name, code, stderr.String())
continue
@@ -396,7 +398,7 @@ func TestRunEncoderRejectsBadInput(t *testing.T) {
func TestRunEncoderFlagConflicts(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-encode", "-validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
if code := Run([]string{"--encode", "--validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for the two modes together", code)
}
if !strings.Contains(stderr.String(), "cannot be combined") {
@@ -405,7 +407,7 @@ func TestRunEncoderFlagConflicts(t *testing.T) {
stdout.Reset()
stderr.Reset()
if code := Run([]string{"-encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
if code := Run([]string{"--encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for an argument", code)
}
}
@@ -432,7 +434,7 @@ n = "a"
t.Fatalf("decode returned %d, stderr = %q", code, stderr.String())
}
var out bytes.Buffer
if code := Run([]string{"-encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
if code := Run([]string{"--encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
t.Fatalf("encode returned %d, stderr = %q", code, stderr.String())
}
want, err := interpres.ParseMap([]byte(doc))
@@ -450,7 +452,7 @@ n = "a"
func TestRunVersion(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"-version"}, strings.NewReader(""), &stdout, &stderr)
code := Run([]string{"--version"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
@@ -463,7 +465,7 @@ func TestRunVersion(t *testing.T) {
func TestRunPlainJSON(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nwhen = 1979-05-27T07:32:00-07:00\nitems = [1, 2]\n")
code := Run([]string{"-json"}, in, &stdout, &stderr)
code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
@@ -481,18 +483,31 @@ func TestRunPlainJSON(t *testing.T) {
func TestValidateDirectorySummary(t *testing.T) {
dir := t.TempDir()
os.WriteFile(filepath.Join(dir, "good.toml"), []byte("a = 1\n"), 0o644)
os.WriteFile(filepath.Join(dir, "bad.toml"), []byte("a =\n"), 0o644)
if err := os.WriteFile(filepath.Join(dir, "good.toml"), []byte("a = 1\n"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "bad.toml"), []byte("a =\n"), 0o644); err != nil {
t.Fatal(err)
}
sub := filepath.Join(dir, "nested")
os.Mkdir(sub, 0o755)
os.WriteFile(filepath.Join(sub, "deep.toml"), []byte("b = true\n"), 0o644)
if err := os.Mkdir(sub, 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(sub, "deep.toml"), []byte("b = true\n"), 0o644); err != nil {
t.Fatal(err)
}
// A second invalid document, so the summary's invalid count is
// exercised beyond the single failure the boolean tracked.
if err := os.WriteFile(filepath.Join(sub, "worse.toml"), []byte("c =\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"-validate", dir}, strings.NewReader(""), &stdout, &stderr)
code := Run([]string{"--validate", dir}, strings.NewReader(""), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 for a directory with an invalid file", code)
t.Fatalf("Run returned %d, want 1 for a directory with invalid files", code)
}
if !strings.Contains(stderr.String(), "checked 3 documents, 1 invalid") {
if !strings.Contains(stderr.String(), "checked 4 documents, 2 invalid") {
t.Errorf("stderr = %q, want the summary", stderr.String())
}
}
@@ -500,7 +515,7 @@ func TestValidateDirectorySummary(t *testing.T) {
func TestInferStruct(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nport = 5432\ntags = [\"a\"]\n\n[server]\nname = \"edge\"\n\n[[items]]\nn = 1\n")
code := Run([]string{"-struct"}, in, &stdout, &stderr)
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
@@ -545,7 +560,7 @@ type Item struct {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"-schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
@@ -572,7 +587,7 @@ func TestRunPlainJSONShapes(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("when = 1979-05-27T07:32:00-07:00\nd = 1979-05-27\nt = 07:32:00\nwall = 1979-05-27T07:32:00\n" +
"items = [1, \"two\"]\n\n[[tables]]\nx = true\n")
code := Run([]string{"-json"}, in, &stdout, &stderr)
code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
@@ -594,7 +609,7 @@ func TestRunPlainJSONShapes(t *testing.T) {
func TestInferStructScalarShapes(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("f = 1.5\nb = true\nd = 1979-05-27\nldt = 1979-05-27T07:32:00\nlt = 07:32:00\nnums = [1, 2, 3]\nmixed = [1, \"a\"]\nempty = []\n")
code := Run([]string{"-struct"}, in, &stdout, &stderr)
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
@@ -614,3 +629,372 @@ func TestInferStructScalarShapes(t *testing.T) {
}
}
}
func TestRunHelpPrintsLongFlags(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--help"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, flag := range []string{"--encode", "--help", "--json", "--schema", "--struct", "--validate", "--version"} {
if !strings.Contains(out, flag) {
t.Errorf("usage output missing %q:\n%s", flag, out)
}
}
// Every flag line of the list names its flag in the two-dash long form
// only, so no line opens with a single dash.
for line := range strings.SplitSeq(strings.TrimRight(out, "\n"), "\n") {
if after, ok := strings.CutPrefix(line, " -"); ok && !strings.HasPrefix(after, "-") {
t.Errorf("usage line %q lists a flag with one dash", line)
}
}
}
func TestGoFieldName(t *testing.T) {
cases := []struct{ key, want string }{
{"host", "Host"},
{"ab", "Ab"},
{"http-host", "HttpHost"},
{"3d", "Field3d"},
{"", "Field"},
}
for _, c := range cases {
if got := goFieldName(c.key, map[string]bool{}); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestGoFieldNameCollision(t *testing.T) {
// Two keys that clean to the same name must not collide; the counter
// keeps the fields apart and both stay exported.
invented := map[string]bool{}
cases := []struct{ key, want string }{
{"a-b", "AB"},
{"a b", "AB2"},
{"a_b", "AB22"},
}
for _, c := range cases {
if got := goFieldName(c.key, invented); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestInferStructDigitLeadingKey(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("3d = true\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Field3d bool") {
t.Errorf("output missing the exported Field3d field:\n%s", stdout.String())
}
}
func TestInferStructBacktickKey(t *testing.T) {
// A backtick in the key would end a raw string literal early, so the
// tag has to be rendered as an interpreted literal instead.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("\"a`b\" = 1\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "AB int64 \"toml:\\\"a`b\\\"\"") {
t.Errorf("output missing the quoted tag:\n%s", out)
}
// The printed definition has to compile; parsing it as Go is the
// syntax half of that proof.
if _, err := parser.ParseFile(token.NewFileSet(), "inferred.go", "package p\n\n"+out, 0); err != nil {
t.Errorf("the printed definition does not parse: %v\n%s", err, out)
}
}
func TestInferStructMergesArrayElements(t *testing.T) {
// The second element carries a key the first lacks, so the slice type
// has to be inferred from both.
var stdout, stderr bytes.Buffer
in := strings.NewReader("[[items]]\nn = 1\n\n[[items]]\nextra = \"late\"\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"Items []struct {",
"N int64 `toml:\"n\"`",
"Extra string `toml:\"extra\"`",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestInferStructMixedArrayStaysValueArray(t *testing.T) {
// One table element does not make the array an array of tables; a
// struct slice would not decode the scalar element.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("arr = [1, {x = 1}]\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "Arr []any") {
t.Errorf("output = %q, want a value array typed []any", out)
}
if strings.Contains(out, "[]struct") {
t.Errorf("output = %q, a mixed array must not become a struct slice", out)
}
}
func TestRunStructParseError(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a =\n"), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 {
t.Errorf("stdout should be empty on parse error, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "line 1") {
t.Errorf("stderr = %q, want the library's line number", stderr.String())
}
}
func TestRunStructWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a = 1\n"), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunSchemaNeedsTypeAndExactlyOneFile(t *testing.T) {
for _, args := range [][]string{{"--schema", "Config"}, {"--schema", "Config", "a.go", "b.go"}} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--schema") {
t.Errorf("Run(%v) stderr = %q, want it to name --schema", args, stderr.String())
}
}
}
func TestRunSchemaUnparsableSource(t *testing.T) {
src := filepath.Join(t.TempDir(), "broken.go")
if err := os.WriteFile(src, []byte("this is not Go\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "broken.go") {
t.Errorf("stderr = %q, want it to name the source file", stderr.String())
}
}
func TestRunSchemaUnknownType(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Missing", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), `no struct type "Missing"`) {
t.Errorf("stderr = %q, want it to name the missing type", stderr.String())
}
}
func TestRunSchemaRecursiveType(t *testing.T) {
// A self-referential struct has no finite template; the generator has
// to name the recursion instead of exhausting the stack.
src := filepath.Join(t.TempDir(), "node.go")
body := "package cfg\n\ntype Node struct {\n\tNext *Node `toml:\"next\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Node", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
for _, want := range []string{"recursive", "Node"} {
if !strings.Contains(stderr.String(), want) {
t.Errorf("stderr = %q, want it to mention %q", stderr.String(), want)
}
}
}
func TestRunSchemaMultiNameField(t *testing.T) {
// A field list may name several fields of one type; each name is one
// TOML key.
src := filepath.Join(t.TempDir(), "range.go")
if err := os.WriteFile(src, []byte("package cfg\n\ntype Range struct {\n\tMin, Max int\n}\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Range", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{"min = 0", "max = 0"} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunSchemaEmbeddedStructs(t *testing.T) {
// The library inlines only untagged embedded structs; a tagged one
// keeps its own section.
src := filepath.Join(t.TempDir(), "embed.go")
body := `package cfg
type Inner struct {
X int ` + "`toml:\"x\"`" + `
}
type Tagged struct {
Inner ` + "`toml:\"inner\"`" + `
Y int ` + "`toml:\"y\"`" + `
}
type Flat struct {
Inner
Z int ` + "`toml:\"z\"`" + `
}
`
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Tagged", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Tagged: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "y = 0") || !strings.Contains(out, "[inner]") {
t.Errorf("Tagged output = %q, want a y scalar and an [inner] section", out)
}
stdout.Reset()
stderr.Reset()
code = Run([]string{"--schema", "Flat", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Flat: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out = stdout.String()
if !strings.Contains(out, "x = 0") || !strings.Contains(out, "z = 0") {
t.Errorf("Flat output = %q, want x and z flattened as scalars", out)
}
if strings.Contains(out, "[inner]") {
t.Errorf("Flat output = %q, an untagged embedded struct must not become a section", out)
}
}
func TestRunVersionWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--version"}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunSchemaWriteFailure(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunEncodeWriteFailure(t *testing.T) {
var stderr bytes.Buffer
in := `{"a": {"type": "integer", "value": "1"}}`
code := Run([]string{"--encode"}, strings.NewReader(in), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunEmptyInput(t *testing.T) {
t.Run("default", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run(nil, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if strings.TrimSpace(stdout.String()) != "{}" {
t.Errorf("stdout = %q, want an empty table", stdout.String())
}
})
t.Run("encode", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2, empty input is not JSON", code)
}
})
t.Run("struct", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "type inferred struct {\n}") {
t.Errorf("stdout = %q, want an empty struct", stdout.String())
}
})
t.Run("validate", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--validate"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
t.Errorf("validate should be quiet, stdout %q stderr %q", stdout.String(), stderr.String())
}
})
}
func TestRunJSONFlagConflicts(t *testing.T) {
for _, args := range [][]string{
{"--encode", "--json"},
{"--struct", "--json"},
{"--validate", "--json"},
{"--json", "--schema", "Config", "config.go"},
} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--json") {
t.Errorf("Run(%v) stderr = %q, want it to explain the --json conflict", args, stderr.String())
}
}
}
+108 -33
View File
@@ -9,8 +9,10 @@ import (
"go/parser"
"go/token"
"io"
"maps"
"path/filepath"
"reflect"
"slices"
"strconv"
"strings"
)
@@ -34,7 +36,9 @@ func runSchema(typeName, sourcePath string, stdout io.Writer) error {
return fmt.Errorf("no struct type %q in %s", typeName, filepath.Base(sourcePath))
}
body := &strings.Builder{}
writeSchemaFields(body, st, types, "")
if err := writeSchemaFields(body, st, types, "", nil); err != nil {
return err
}
_, err = io.WriteString(stdout, strings.TrimLeft(body.String(), "\n"))
return err
}
@@ -74,9 +78,18 @@ type fieldMeta struct {
// writeSchemaFields writes the fields of one struct level: the scalar lines
// first, then the sections, so the template re-parses with every value under
// the header it belongs to. prefix is the dotted path the nested headers
// carry.
func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]*ast.StructType, prefix string) {
metas := metasOf(st, types)
// carry. path holds the struct types of the levels currently being written,
// so a type that reaches itself is reported as recursion instead of
// exhausting the stack.
func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]*ast.StructType, prefix string, path []*ast.StructType) error {
if slices.Contains(path, st) {
return recursionError(st, types)
}
path = append(path, st)
metas, err := metasOf(st, types)
if err != nil {
return err
}
for _, m := range metas {
if _, elemSt := elementStruct(m.typ, types); elemSt != nil {
continue
@@ -98,7 +111,9 @@ func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]
writeComment(w, m.comment)
fmt.Fprintf(w, "[%s%s]\n", prefix, m.key)
if sub := structOf(m.typ, types); sub != nil {
writeSchemaFields(w, sub, types, prefix+m.key+".")
if err := writeSchemaFields(w, sub, types, prefix+m.key+".", path); err != nil {
return err
}
}
fmt.Fprintln(w)
}
@@ -109,9 +124,12 @@ func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]
}
writeComment(w, m.comment)
fmt.Fprintf(w, "[[%s%s]]\n", prefix, m.key)
writeSchemaFields(w, elemSt, types, "")
if err := writeSchemaFields(w, elemSt, types, "", path); err != nil {
return err
}
fmt.Fprintln(w)
}
return nil
}
// writeComment writes the comment lines above a binding.
@@ -125,46 +143,103 @@ func writeComment(w *strings.Builder, text string) {
}
// metasOf flattens the exported fields of a struct. The key comes from the
// toml tag, or the lower-cased field name; a `-` key drops the field.
func metasOf(st *ast.StructType, types map[string]*ast.StructType) []fieldMeta {
// toml tag, or the lower-cased field name; a `-` key drops the field. An
// embedded struct without a tag name flattens into its parent, the way the
// library inlines it, while a tagged one keeps its own section.
func metasOf(st *ast.StructType, types map[string]*ast.StructType) ([]fieldMeta, error) {
return flattenMetas(st, types, nil)
}
// flattenMetas is metasOf with the chain of struct types currently being
// flattened, which stops a struct that embeds itself, directly or through
// another embedded type.
func flattenMetas(st *ast.StructType, types map[string]*ast.StructType, chain map[*ast.StructType]bool) ([]fieldMeta, error) {
if chain[st] {
return nil, recursionError(st, types)
}
// A copy per branch: the chain is the path being flattened now, not the
// set ever visited, so a type embedded in two siblings is not mistaken
// for recursion.
chain = maps.Clone(chain)
if chain == nil {
chain = map[*ast.StructType]bool{}
}
chain[st] = true
var out []fieldMeta
for _, field := range st.Fields.List {
if len(field.Names) == 0 {
// An untagged embedded struct flattens into the parent.
if ident, ok := baseType(field.Type).(*ast.Ident); ok {
if inner, ok := types[ident.Name]; ok {
out = append(out, metasOf(inner, types)...)
}
}
continue
}
name := field.Names[0].Name
if !ast.IsExported(name) {
continue
}
tagText := ""
if field.Tag != nil {
tagText, _ = strconv.Unquote(field.Tag.Value)
}
toml := reflect.StructTag(tagText).Get("toml")
key, opts := "", ""
if toml != "" {
key, opts, _ = strings.Cut(toml, ",")
key, opts, _ := strings.Cut(toml, ",")
if len(field.Names) == 0 {
if key == "" {
// An untagged embedded struct flattens into its parent.
if ident, ok := baseType(field.Type).(*ast.Ident); ok {
if inner, ok := types[ident.Name]; ok {
metas, err := flattenMetas(inner, types, chain)
if err != nil {
return nil, err
}
out = append(out, metas...)
}
}
continue
}
if key == "-" {
continue
}
// A tagged embedded struct is a section of its own; the tag
// name is the only name it has.
out = append(out, fieldMeta{
key: key,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
continue
}
if key == "-" {
continue
}
if key == "" {
key = strings.ToLower(name)
// A field list may name several fields of one type, `Min, Max int`;
// each name is one TOML key.
for _, name := range field.Names {
if !ast.IsExported(name.Name) {
continue
}
fieldKey := key
if fieldKey == "" {
fieldKey = strings.ToLower(name.Name)
}
out = append(out, fieldMeta{
key: fieldKey,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
}
out = append(out, fieldMeta{
key: key,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
}
return out
return out, nil
}
// recursionError names the struct type that reached itself. Such a type has
// no finite TOML template: every level would nest another copy of the same
// shape.
func recursionError(st *ast.StructType, types map[string]*ast.StructType) error {
return fmt.Errorf("recursive type %s: the struct contains itself, so it has no finite template", typeName(st, types))
}
// typeName names the declared struct type st refers to, and "anonymous
// struct" for a literal one that no declaration names.
func typeName(st *ast.StructType, types map[string]*ast.StructType) string {
for name, t := range types {
if t == st {
return name
}
}
return "anonymous struct"
}
// tagOption returns the text a `name=` option carries in the option part of
+32 -15
View File
@@ -79,7 +79,9 @@ func clockString(t time.Time) string {
// offsetString renders an offset date-time, the fourth TOML kind, in the same
// shape: no zero seconds, no trailing zeros in the fraction, and the offset
// written as "Z" when it is zero.
// written as "Z" when it is zero. A zone offset that is not a whole number of
// minutes loses its seconds to this rendering, which is why Marshal refuses
// such a value rather than writing it.
func offsetString(t time.Time) string {
buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
buf = appendClock(buf, t)
@@ -235,17 +237,21 @@ func normaliseDateTimeToken(tok string, kind dateTimeKind) string {
// parseDateTime classifies and parses a bare token as a TOML date-time value.
// It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or
// LocalTime) and whether the token was a date-time at all.
func parseDateTime(tok string) (any, bool) {
// LocalTime), whether the token was a date-time at all, and an error for a
// token whose shape is a date-time a component of which lies outside its
// range: an hour of 24, a day the month does not hold. Such a token is a
// broken date-time, not some other value, so the error names it instead of
// leaving it to the number decoder's complaint.
func parseDateTime(tok string) (any, bool, error) {
if tok == "" || tok[0] < '0' || tok[0] > '9' {
return nil, false
return nil, false, nil
}
if !strings.ContainsAny(tok, "-:") {
return nil, false
return nil, false, nil
}
kind, seconds := scanDateTimeShape(tok)
if kind == dateTimeNone {
return nil, false
return nil, false, nil
}
norm := normaliseDateTimeToken(tok, kind)
switch kind {
@@ -256,7 +262,7 @@ func parseDateTime(tok string) (any, bool) {
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
// A zero offset carries its own anonymous location from time.Parse,
// while the written form is "Z" either way; normalising to UTC keeps
@@ -264,7 +270,7 @@ func parseDateTime(tok string) (any, bool) {
if _, off := t.Zone(); off == 0 {
t = t.In(time.UTC)
}
return OffsetDateTime{t}, true
return OffsetDateTime{t}, true, nil
case dateTimeLocal:
layout := localClockLayout
if seconds {
@@ -272,15 +278,15 @@ func parseDateTime(tok string) (any, bool) {
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalDateTime{t}, true
return LocalDateTime{t}, true, nil
case dateTimeDate:
t, err := time.Parse(localDateOnlyLayout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalDate{t}, true
return LocalDate{t}, true, nil
case dateTimeClock:
layout := localTimeClockLayout
if seconds {
@@ -288,11 +294,22 @@ func parseDateTime(tok string) (any, bool) {
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalTime{t}, true
return LocalTime{t}, true, nil
}
return nil, false
return nil, false, nil
}
// wholeMinuteOffset reports an error when the zone offset carries seconds, a
// shape no TOML offset can hold: writing only the minutes would silently
// shift the instant on the way back, so the encoder refuses the value rather
// than corrupting it.
func wholeMinuteOffset(t time.Time) error {
if _, off := t.Zone(); off%60 != 0 {
return fmt.Errorf("interpres: date-time offset of %d seconds is not a whole number of minutes, which TOML cannot write", off)
}
return nil
}
// isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a
+21 -4
View File
@@ -7,6 +7,7 @@ import (
"context"
"encoding"
"fmt"
"maps"
"reflect"
"slices"
"strings"
@@ -337,7 +338,8 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
if len(schema.required) > 0 {
seen = make(map[string]bool, len(tbl))
}
for key, val := range tbl {
for _, key := range d.tableKeys(tbl) {
val := tbl[key]
// A key that is already lowercase, which document keys usually are,
// hits the map directly; only a miss pays for the case fold.
resolved := key
@@ -349,12 +351,15 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
if !ok {
if schema.embedMaps != nil {
// Leftover keys land in an untagged embedded map, the inverse
// of the encoder inlining that map's entries.
// of the encoder inlining that map's entries. The assign call
// rather than assignMap itself lets it allocate the embedded
// pointer the field may be, the way any other destination is
// reached.
mv, err := fieldByIndex(dst, schema.embedMaps[0])
if err != nil {
return newDecodeError(key, err)
}
if err := d.assignMap(map[string]any{key: val}, mv); err != nil {
if err := d.assign(map[string]any{key: val}, mv); err != nil {
return newDecodeError(key, err)
}
}
@@ -379,6 +384,18 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
return nil
}
// tableKeys returns the keys of tbl in the order the document wrote them
// when the node index knows it, and in sorted order otherwise, the order a
// hand-built tree or a node-free parse offers. The order settles which of
// two keys that differ only in case wins one field: the same key wins every
// run, instead of whichever a map iteration happened to hand out.
func (d *decoder) tableKeys(tbl map[string]any) []string {
if node := d.nodeOf(tbl); node != nil {
return node.Keys()
}
return slices.Sorted(maps.Keys(tbl))
}
func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
if dst.Type().Key().Kind() != reflect.String {
return fmt.Errorf("interpres: map key must be a string, got %s", dst.Type().Key())
@@ -499,7 +516,7 @@ func (d *decoder) setLocalTimeValue(t time.Time, dst reflect.Value) error {
t.Hour(), t.Minute(), t.Second(), t.Nanosecond(), d.loc)))
return nil
}
return fmt.Errorf("interpres: cannot assign local date-time to time.Time; set Decoder.LocalTimeLocation to choose the zone")
return fmt.Errorf("interpres: cannot assign local date-time to time.Time; set LocalTimeLocation to choose the zone")
}
return fmt.Errorf("interpres: cannot assign local date-time to %s", dst.Type())
}
+82
View File
@@ -1690,3 +1690,85 @@ func TestErrorMessagesGolden(t *testing.T) {
})
}
}
// TestLocalTimeLocationLeavesOffsetsAlone pins that the option's zone is
// used for local date-times only: an offset date-time keeps the offset the
// document wrote.
func TestLocalTimeLocationLeavesOffsetsAlone(t *testing.T) {
var cfg struct {
Stamp time.Time `toml:"stamp"`
}
err := Unmarshal([]byte("stamp = 1979-05-27T07:32:00-07:00\n"), &cfg,
LocalTimeLocation(time.FixedZone("Prague", 2*60*60)))
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if _, off := cfg.Stamp.Zone(); off != -7*60*60 {
t.Errorf("offset = %d, want the document's -07:00", off/3600)
}
}
// TestUnmarshalCaseCollisionIsDeterministic pins that two keys differing
// only in case, both matching one field, resolve the same way on every run
// and on both decode paths.
func TestUnmarshalCaseCollisionIsDeterministic(t *testing.T) {
type cfg struct {
Host string `toml:"host"`
}
in := []byte("Host = \"upper\"\nhost = \"lower\"\n")
// The tree path iterates a map, so pin the winner across many runs.
var want string
for range 50 {
var viaTree cfg
if err := treeDecodeInto(in, &viaTree); err != nil {
t.Fatalf("tree decode: %v", err)
}
if want == "" {
want = viaTree.Host
} else if viaTree.Host != want {
t.Fatalf("tree decode is not deterministic: %q then %q", want, viaTree.Host)
}
}
var targeted cfg
if err := Unmarshal(in, &targeted); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if targeted.Host != want {
t.Errorf("targeted Host = %q, tree %q", targeted.Host, want)
}
}
// TestUnmarshalEmbeddedPointerMap pins that leftover keys reach an embedded
// pointer to a map, allocating it, rather than panicking on the pointer.
func TestUnmarshalEmbeddedPointerMap(t *testing.T) {
type Extra map[string]int
type cfg struct {
*Extra
Name string `toml:"name"`
}
var c cfg
err := Unmarshal([]byte("name = \"x\"\nrogue = 7\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Extra == nil || (*c.Extra)["rogue"] != 7 {
t.Errorf("embedded map = %v, want rogue allocated and filled", c.Extra)
}
}
// TestUnmarshalIgnoresEncodeTagOptions pins that the emission-only tag
// options change nothing on the decode side.
func TestUnmarshalIgnoresEncodeTagOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name,omitempty"`
Port int `toml:"port,omitzero,comment=The port"`
}
var c cfg
err := Unmarshal([]byte("name = \"x\"\nport = 8080\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Name != "x" || c.Port != 8080 {
t.Errorf("cfg = %+v, want both fields filled", c)
}
}
+16 -17
View File
@@ -105,7 +105,7 @@ Encode options:
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
| `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header |
### `func ParseAs[T any](data []byte) (T, error)`
### `func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error)`
The generic shorthand for `Unmarshal` with a destination variable:
@@ -125,7 +125,8 @@ that is not a struct warms nothing.
### `func Statements(r io.Reader) iter.Seq2[Statement, error]`
Iterates the top-level statements of the document r carries, in written
order: key/value statements, a `[table]` header as one statement carrying
order: key/value statements, a value array or an inline table among them as
one statement whatever it holds, a `[table]` header as one statement carrying
its `Table` node, and an `[[array of tables]]` as one statement per element
with the element's node and its `Index`. Iteration stops at the first error
and at a false yield, so a caller looking for one section reads no further.
@@ -223,11 +224,7 @@ introduced:
and no surrounding space, so `# note` is stored as `note` and a bare `#` as
`""`.
A `Document` is not a value to marshal: `Marshal` writes values, so it refuses
one and points at `doc.Map()`. Writing a document back, with its order and its
comments, belongs with the editing API.
### `func Unmarshal(data []byte, v any) error`
### `func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error`
Parses `data` and stores the result in the value pointed to by `v`, typically a
pointer to a struct or to `map[string]any`. Equivalent to
@@ -240,11 +237,11 @@ if err := interpres.Unmarshal(data, &cfg); err != nil {
}
```
### `func UnmarshalContext(ctx context.Context, data []byte, v any) error`
### `func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...UnmarshalOption) error`
The cancellable variant of `Unmarshal`.
### `func Marshal(v any) ([]byte, error)`
### `func Marshal(v any, opts ...MarshalOption) ([]byte, error)`
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
TOML document. The emission rules are in the [Encoding](#encoding) section
@@ -254,12 +251,12 @@ below. Equivalent to `MarshalContext(context.Background(), v)`.
out, err := interpres.Marshal(cfg)
```
### `func MarshalContext(ctx context.Context, v any) ([]byte, error)`
### `func MarshalContext(ctx context.Context, v any, opts ...MarshalOption) ([]byte, error)`
The cancellable variant of `Marshal`. The context is checked before any work
and every 64 fields during the reflection walk.
### `func MarshalAppend(buf []byte, v any) ([]byte, error)`
### `func MarshalAppend(buf []byte, v any, opts ...MarshalOption) ([]byte, error)`
Appends the TOML encoding of `v` to `buf` and returns the extended buffer, the
shape `json.MarshalAppend` has. A failed encoding leaves `buf` untouched.
@@ -525,9 +522,9 @@ through the ordinary assignment rules, so every conversion, hook and error the
path is pinned by a differential fuzz target that decodes every generated
document both ways and compares the results.
A document or destination the direct skeleton cannot model — an unknown table
A document or destination the direct skeleton cannot model (an unknown table
under strictness it must sink, a hook that needs the whole parsed value, an
embedded map filler — falls back to the tree path and reruns, so the
embedded map filler) falls back to the tree path and reruns, so the
observable behaviour is always the tree path's, exactly. Nothing changes for
`Parse`, `ParseMap` or the document API: the tree remains theirs.
@@ -561,8 +558,10 @@ sequenceDiagram
### Input constraints
`Marshal` and `MarshalWrite` accept a `struct`, a `map[string]V`, or a
non-nil pointer to one, where `V` is any value `Marshal` itself understands. A
different top-level value fails:
non-nil pointer to one, where `V` is any value `Marshal` itself understands.
An `OrderedMap` and a `Document` are accepted as themselves: the first in its
written key order, the second written back as it stands. A different
top-level value fails:
| Input | Error |
|---|---|
@@ -736,7 +735,7 @@ across a round-trip.
A nil slice is always omitted. An empty (length 0) array of tables is always
omitted, because TOML forbids an empty `[[a]]`. Other empty arrays emit as
`key = []` by default; `OmitEmptyArrays()` skips them as well, so
`key = []` by default; `OmitEmptyArrays(true)` skips them as well, so
`[]string{}` is treated like a nil slice.
### Long strings
@@ -863,7 +862,7 @@ encoding/json/v2 made current, with the differences TOML asks for:
| `encoding.TextMarshaler`, `TextUnmarshaler` | honoured, the same | a type that renders itself as text becomes a TOML string, both ways |
| `*json.UnmarshalTypeError` | `*DecodeError` | the path is segments with a `String()` renderer, not a dotted string |
| `*json.SyntaxError` | `*SyntaxError` | the TOML error adds the byte `Offset` and the `Column` to the line |
| context support | `*Context` variants of every entry point | encoding/json has none |
| context support | `*Context` variants of the parse, decode and marshal entries | encoding/json has none |
## Types
+26 -11
View File
@@ -5,16 +5,17 @@ source tree; nothing is aspirational.
## Overview
interpres is one public library package, one command, and one example. The
interpres is one public library package, one command, and two examples. The
library implements the whole of TOML 1.1, decoding and encoding, in the
standard library alone; the command wraps the parser and the encoder for the
toml-test compliance harness, against which it stands at 214 valid, 467 invalid
and 214 encoder cases with zero failures; the example demonstrates the API.
and 214 encoder cases with zero failures; the examples demonstrate the API: one the document round trip, one the statement iterator.
```mermaid
flowchart TD
CLI[cmd/interpres-decode<br/>toml-test adapter] --> API
EX[examples/basic<br/>usage demo] --> API
EX2[examples/statements<br/>statement iterator demo] --> API
subgraph Lib [package interpres]
API[interpres.go<br/>public API and types]
API --> P[parser.go<br/>recursive-descent parser]
@@ -35,9 +36,10 @@ strict validation.
| Path | Responsibility |
|---|---|
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, the option constructors, `Marshaler`, `Unmarshaler`, `SyntaxError`, the local date-time types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `-encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, the option constructors, `Marshaler`, `Unmarshaler`, `SyntaxError`, the error and option types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `--encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
| `examples/basic` | A runnable tour of the API. Documentation in executable form, not part of the library. |
| `examples/statements` | The `Statements` iterator over a document, the shape a configuration tool reads. Documentation in executable form. |
Inside the library package, one file owns one concern:
@@ -46,18 +48,28 @@ Inside the library package, one file owns one concern:
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree, records the nodes a [Document](API.md#documents) is built from, and enforces the structural rules of TOML 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
| `document.go` | The parsed-document types: `Document`, `Table` and `Entry`, which carry the key order, whether a table was written inline, and the comments. The values they expose are the parser's own tree, not a copy. |
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `datetime.go` | The four date-time types (`OffsetDateTime` and the three local wrappers) and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `orderedmap.go` | `OrderedMap`, the table that keeps its key order, and the node index the decoder reads the written order from. |
| `target.go` | The targeted parse: the struct skeleton resolved against the document while it scans, no intermediate tree. Falls back to the tree path for every shape it does not model. |
| `docwrite.go` | The write side of the document pipeline: `UnmarshalDocument` and the writer that renders a `Document` back with its order and comments. |
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
| `encode.go` | The reverse walk: builds an intermediate `tomlDoc` per table (which is what preserves declaration order and enables the group-by-kind partition) and then emits it as TOML. |
The boundary that matters: `parser.go` produces only untyped trees
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); `decode.go` and
`encode.go` are the only files that touch `reflect`; the command never touches
either, it consumes `Parse` alone.
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); the reflection work
lives in `decode.go`, `encode.go`, `target.go` and `orderedmap.go`; the command
consumes `ParseMap`, `Parse` and `Marshal`, and owns no parsing or emission
logic of its own.
## Data flow
Decoding is parse, then one reflection walk. `SyntaxError` values are produced
Decoding has two paths. The direct one parses straight into a struct
destination: `target.go` resolves the table skeleton against the struct
schema while the document scans, and values assign through the ordinary
decoder rules, so no intermediate tree exists; that is the hot path every
`Unmarshal` into a struct takes. A document or destination the direct
skeleton cannot model falls back to the tree path: parse the whole document,
then one reflection walk over the tree. `SyntaxError` values are produced
inside `parser.go` and returned as-is; conversion errors are produced inside
`decode.go` and wrapped with the key path as they unwind.
@@ -66,10 +78,13 @@ sequenceDiagram
participant Caller
participant API as interpres.go
participant P as parser.go
participant T as target.go
participant D as decode.go
Caller->>API: Unmarshal(data, v)
API->>P: ParseContext(ctx, data)
P->>P: number and datetime atoms
API->>T: targeted parse into the struct
T->>P: scanner, grammar, atoms
T-->>API: result, error or fallback
API->>P: on fallback, ParseContext(ctx, data)
P-->>API: map tree or *SyntaxError
API->>D: decode(tree, reflect value)
D-->>API: nil or wrapped field error
+2
View File
@@ -15,6 +15,8 @@ The benchmarks live in `bench_test.go`, next to the code they measure:
| `BenchmarkParseLong` | `ParseMap` over a generated document with about 2000 array-of-tables entries |
| `BenchmarkStrictDecodeLong` | `Unmarshal` into a typed document under `RejectUnknownFields`, over the same long document |
| `BenchmarkMarshalLong` | `Marshal` of the tree `ParseMap` produced from the long document |
| `BenchmarkStrictDecodeTree` | the tree-path reference decode of the representative document: parse, then the reflection walk |
| `BenchmarkStrictDecodeTreeLong` | the tree-path reference decode of the long document, the A/B baseline of the targeted parse |
## Running
+45 -40
View File
@@ -3,6 +3,7 @@
The reference below is taken from the program itself. `interpres-decode` is
the toml-test harness adapter in both directions, decoding TOML into tagged
JSON and encoding tagged JSON back into TOML, and it also validates documents.
The same reference ships as the manual page `man/interpres-decode.1`.
Install it with Go itself, no release assets involved:
```sh
@@ -13,50 +14,52 @@ go install sourcedock.dev/petrbalvin/interpres/v2/cmd/interpres-decode@latest
```sh
interpres-decode [flags]
interpres-decode -encode
interpres-decode -validate [file ...]
interpres-decode -validate [directory ...]
interpres-decode -json
interpres-decode -struct
interpres-decode -schema TYPE file.go
interpres-decode -version
interpres-decode --encode
interpres-decode --validate [file ...]
interpres-decode --validate [directory ...]
interpres-decode --json
interpres-decode --struct
interpres-decode --schema TYPE file.go
interpres-decode --version
```
Without `-validate`, `-encode`, `-json` or `-struct` the program is the
decoding half of the toml-test adapter: it takes no arguments, reads one TOML
document from stdin, and writes the toml-test tagged-JSON form to stdout.
Build it locally with `just build`, which compiles it into
Without `--validate`, `--encode`, `--json`, `--struct` or `--schema` the
program is the decoding half of the toml-test adapter: it takes no arguments,
reads one TOML document from stdin, and writes the toml-test tagged-JSON form
to stdout. Build it locally with `just build`, which compiles it into
`bin/interpres-decode`, or run it straight from the module directory with
`just run`.
With `-encode` the direction is reversed: the program reads a tagged-JSON
With `--encode` the direction is reversed: the program reads a tagged-JSON
description from stdin and writes the TOML document it describes to stdout,
which is the shape toml-test expects of an encoder command. It takes no
arguments either, and the mode flags cannot be combined.
With `-validate` the program parses each named file instead, or stdin when no
With `--validate` the program parses each named file instead, or stdin when no
file is named, and prints one line per invalid document to stderr. It is
quiet on valid documents, which is the shape a CI step wants. The `-` name
means stdin. A named directory is walked for `.toml` files, every one of them
validated, and the walk closes with a summary on stderr naming how many
documents were checked and how many were invalid.
With `-json` the decoding half prints plain indented JSON instead of the
With `--json` the decoding half prints plain indented JSON instead of the
tagged form, the shape for people and diffs: the values keep their types as
JSON sees them, and the date-time wrappers print in their TOML form.
JSON sees them, and the date-time wrappers print in their TOML form. The flag
shapes the decoding output only, so it is rejected together with the mode
flags.
With `-struct` the program reads a TOML document from stdin and prints a Go
With `--struct` the program reads a TOML document from stdin and prints a Go
struct definition shaped like it: one field per key in written order, nested
tables as nested struct types, and an array of tables as a slice. The
printed type compiles and decodes the document it came from.
With `-schema` the program reads a Go source file and writes a TOML template
With `--schema` the program reads a Go source file and writes a TOML template
for the named struct type: one key per exported field, the `comment=` tag
option printed as a comment above it, and the `default=` option as the value
where one is set. It is the inverse of `-struct`, for config-driven
where one is set. It is the inverse of `--struct`, for config-driven
applications that generate their example configuration from the type.
`-version` prints the binary's version and exits. The release pipeline builds
`--version` prints the binary's version and exits. The release pipeline builds
at the tag, so a released binary prints its own tag; a build from a working
tree prints `(devel)`.
@@ -64,21 +67,21 @@ tree prints `(devel)`.
| Flag | Effect |
|---|---|
| `-validate` | validate the documents instead of emitting tagged JSON; directories are walked for `.toml` files |
| `-encode` | read tagged JSON from stdin and write TOML instead |
| `-json` | with the default mode, print plain indented JSON instead of tagged JSON |
| `-struct` | infer a Go struct definition from the document on stdin and print it |
| `-schema TYPE` | write a TOML template for the struct type TYPE from the Go source file named as the first argument |
| `-version` | print the version and exit |
| `-h` | print the usage |
| `--validate` | validate the documents instead of emitting tagged JSON; directories are walked for `.toml` files |
| `--encode` | read tagged JSON from stdin and write TOML instead |
| `--json` | with the default mode, print plain indented JSON instead of tagged JSON |
| `--struct` | infer a Go struct definition from the document on stdin and print it |
| `--schema TYPE` | write a TOML template for the struct type TYPE from the Go source file named as the first argument |
| `--version` | print the version and exit |
| `--help` | print the usage |
## Exit codes
| Code | Meaning |
|---|---|
| `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed; schema, struct, version: the output was written |
| `1` | adapter: parse error; validate: at least one document is invalid |
| `2` | a usage error, a read failure, malformed tagged JSON, or a value with no TOML representation |
| `1` | adapter: parse error; validate: at least one document is invalid; struct: the document on stdin failed to parse |
| `2` | a usage error, a read or write failure, malformed tagged JSON, or a value with no TOML representation |
## Wire format
@@ -105,7 +108,7 @@ wrapped in an object with a `type` and a `value`:
| local date | `date-local` | `1979-05-27` |
| local time | `time-local` | `07:32:00.999999` |
The `-encode` mode reads exactly this form back. Two properties of it are
The `--encode` mode reads exactly this form back. Two properties of it are
worth knowing. A float whose value has no fraction and no exponent is written
as a bare integer string, `{"type": "float", "value": "1"}`, so there the tag
decides the type and not the literal. And the form cannot tell an array of
@@ -126,10 +129,10 @@ port = 9090
```
The output is the equivalent value tree as one JSON object. Turn a description
back into TOML with `-encode`:
back into TOML with `--encode`:
```sh
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode -encode
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode --encode
```
```toml
@@ -139,14 +142,14 @@ title = "hello"
Validate the TOML files of another repository in CI:
```sh
interpres-decode -validate config.toml deploy/example.toml
interpres-decode --validate config.toml deploy/example.toml
```
An invalid document reports the file and the library's line number:
```sh
$ interpres-decode -validate bad.toml
bad.toml: interpres: line 1: expected a value
$ interpres-decode --validate bad.toml
interpres-decode: bad.toml: interpres: line 1: expected a value
$ echo $?
1
```
@@ -155,22 +158,24 @@ Sweep a whole directory tree of configuration, with the summary the walk
closes on:
```sh
$ interpres-decode -validate configs/
configs/old.toml: interpres: line 3: duplicate key "port"
$ interpres-decode --validate configs/
interpres-decode: configs/old.toml: interpres: line 3: duplicate key "port"
checked 14 documents, 1 invalid
$ echo $?
1
```
See the document a `-struct` template would decode:
See the document a `--struct` template would decode:
```sh
echo 'host = "db"
port = 5432
' | ./bin/interpres-decode -struct
' | ./bin/interpres-decode --struct
```
```go
// Generated by interpres-decode --struct; decode with
// sourcedock.dev/petrbalvin/interpres/v2.
type inferred struct {
Host string `toml:"host"`
Port int64 `toml:"port"`
@@ -182,7 +187,7 @@ the Go source declares fields tagged
`toml:"host,comment=The host to dial,default=example.org"`:
```sh
./bin/interpres-decode -schema Config config.go
./bin/interpres-decode --schema Config config.go
```
```toml
@@ -193,7 +198,7 @@ host = "example.org"
Print the binary's version:
```sh
$ ./bin/interpres-decode -version
$ ./bin/interpres-decode --version
interpres-decode v2.0.0
```
+4
View File
@@ -46,6 +46,9 @@ prints the same list.
| `just install` | builds, then copies the binary into `~/.local/bin` (`BINDIR` overrides) |
| `just uninstall` | removes the installed binary |
| `just clean` | removes `bin/` and `coverage.out` |
| `just cross` | cross-compile smoke of the library and the command for arm64, loong64, riscv64 and the browser and edge runtimes; a hand-run convenience, not a gate |
| `just release-check X.Y.Z` | the release pre-flight: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release |
| `just docs-drift` | compares the toml-test counts the documentation quotes with a live suite run |
## Running a single test
@@ -99,6 +102,7 @@ pipeline.
|---|---|---|
| `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite |
| `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally |
| `fuzz.yml` | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut |
## Releases
+122 -26
View File
@@ -34,15 +34,31 @@ func (d *Document) Root() *Table {
}
// Map returns the value tree, the shape ParseMap gives. It is the tree the
// document was parsed into, not a copy.
func (d *Document) Map() map[string]any { return d.root.values }
// document was parsed into, not a copy. A nil document or one with no root
// holds no values.
func (d *Document) Map() map[string]any {
if d == nil || d.root == nil {
return nil
}
return d.root.values
}
// Footer returns the comment lines that follow the last statement, and every
// line of a document that holds no statement at all.
func (d *Document) Footer() []string { return d.footer }
func (d *Document) Footer() []string {
if d == nil {
return nil
}
return d.footer
}
// SetFooter replaces those lines.
func (d *Document) SetFooter(lines []string) { d.footer = lines }
func (d *Document) SetFooter(lines []string) {
if d == nil {
return
}
d.footer = lines
}
// The document-level convenience forms of the Table edit API; they act on
// the root table.
@@ -87,6 +103,11 @@ type Table struct {
// rather than under a header or as a dotted key.
inline bool
// dotted records that a dotted key introduced the table, `a.b = 1`
// building the a around the leaf: the write side gives such a table back
// as dotted key lines, the form that holds the position of a line.
dotted bool
// comments are the lines above the table's header, trailing is the comment
// on the header's own line. Both are empty for a table a dotted key
// introduced, which has no line of its own.
@@ -98,8 +119,12 @@ func newTable(values map[string]any) *Table {
return &Table{values: values, index: map[string]*Entry{}}
}
// Keys returns the table's keys in the order they were written.
// Keys returns the table's keys in the order they were written. A nil table
// holds none, the answer a document without a root gives through Root.
func (t *Table) Keys() []string {
if t == nil {
return nil
}
keys := make([]string, len(t.entries))
for i, e := range t.entries {
keys[i] = e.key
@@ -109,43 +134,75 @@ func (t *Table) Keys() []string {
// Values returns the table's values, which is the map the value tree holds for
// it.
func (t *Table) Values() map[string]any { return t.values }
func (t *Table) Values() map[string]any {
if t == nil {
return nil
}
return t.values
}
// Entries returns the table's entries in written order.
func (t *Table) Entries() []*Entry { return t.entries }
func (t *Table) Entries() []*Entry {
if t == nil {
return nil
}
return t.entries
}
// Get returns the entry for key, and whether the table has one.
func (t *Table) Get(key string) (*Entry, bool) {
if t == nil {
return nil, false
}
e, ok := t.index[key]
return e, ok
}
// Inline reports whether the table was written as an inline table, `{…}`,
// rather than under a header or introduced by a dotted key.
func (t *Table) Inline() bool { return t.inline }
func (t *Table) Inline() bool { return t != nil && t.inline }
// Comments returns the comment lines above the table's header, or above the
// key that introduced it. Lines carry no leading '#' and no surrounding space.
func (t *Table) Comments() []string { return t.comments }
func (t *Table) Comments() []string {
if t == nil {
return nil
}
return t.comments
}
// SetComments replaces those lines. Each line is written back with a "# " in
// front of it, so a line should not carry one.
func (t *Table) SetComments(lines []string) { t.comments = lines }
func (t *Table) SetComments(lines []string) {
if t == nil {
return
}
t.comments = lines
}
// Trailing returns the comment on the header's own line, without the '#'.
func (t *Table) Trailing() string { return t.trailing }
func (t *Table) Trailing() string {
if t == nil {
return ""
}
return t.trailing
}
// SetTrailing replaces that comment.
func (t *Table) SetTrailing(line string) { t.trailing = line }
func (t *Table) SetTrailing(line string) {
if t == nil {
return
}
t.trailing = line
}
// addValue records a key of the table, in written order.
// addValue records a key of the table, in written order. The caller gives
// the entry a table node or element nodes when the value has that shape; a
// map value left without a node writes as an inline table.
func (t *Table) addValue(key string, val any, inline bool) *Entry {
e := &Entry{table: t, key: key, inline: inline}
t.entries = append(t.entries, e)
t.index[key] = e
if node, ok := val.(map[string]any); ok {
e.child = newTable(node)
}
return e
}
@@ -178,6 +235,9 @@ func (t *Table) addElement(key string, values map[string]any) *Table {
// child returns the node of a table-valued key, or nil.
func (t *Table) child(key string) *Table {
if t == nil {
return nil
}
if e, ok := t.index[key]; ok {
return e.child
}
@@ -239,6 +299,9 @@ func (e *Entry) SetTrailing(line string) { e.trailing = line }
// GetString returns the string the key holds, and whether it holds one.
func (t *Table) GetString(key string) (string, bool) {
if t == nil {
return "", false
}
v, ok := t.values[key]
s, ok := v.(string)
return s, ok
@@ -246,6 +309,9 @@ func (t *Table) GetString(key string) (string, bool) {
// GetInt returns the integer the key holds, and whether it holds one.
func (t *Table) GetInt(key string) (int64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key]
i, ok := v.(int64)
return i, ok
@@ -253,6 +319,9 @@ func (t *Table) GetInt(key string) (int64, bool) {
// GetFloat returns the float the key holds, and whether it holds one.
func (t *Table) GetFloat(key string) (float64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key]
f, ok := v.(float64)
return f, ok
@@ -260,6 +329,9 @@ func (t *Table) GetFloat(key string) (float64, bool) {
// GetBool returns the boolean the key holds, and whether it holds one.
func (t *Table) GetBool(key string) (bool, bool) {
if t == nil {
return false, false
}
v, ok := t.values[key]
b, ok := v.(bool)
return b, ok
@@ -267,6 +339,9 @@ func (t *Table) GetBool(key string) (bool, bool) {
// GetArray returns the value array the key holds, and whether it holds one.
func (t *Table) GetArray(key string) ([]any, bool) {
if t == nil {
return nil, false
}
v, ok := t.values[key]
a, ok := v.([]any)
return a, ok
@@ -282,9 +357,13 @@ func (t *Table) GetTable(key string) (*Table, bool) {
// Set stores value under key. A key the table already has keeps its position
// and its comments; a new one joins the end. A value of map[string]any
// becomes a table node of its own, written under a header like any other
// table; a Go map carries no order, so its keys take sorted order. A value
// table, and replaces the node the key held, which belonged to the value the
// key held; a Go map carries no order, so its keys take sorted order. A value
// of []map[string]any becomes an array-of-tables node.
func (t *Table) Set(key string, value any) {
if t == nil {
return
}
e, ok := t.index[key]
if !ok {
t.values[key] = value
@@ -303,11 +382,10 @@ func (t *Table) Set(key string, value any) {
t.values[key] = value
switch v := value.(type) {
case map[string]any:
if e.child == nil {
e.child = newOrderedTable(v)
} else {
e.child.values = v
}
// The node is rebuilt rather than patched: the entries and the index
// belong to the table the key held, and writing the new value
// through them would leave the old table's keys in the output.
e.child = newOrderedTable(v)
e.elements = nil
case []map[string]any:
e.child = nil
@@ -322,15 +400,30 @@ func (t *Table) Set(key string, value any) {
}
// newOrderedTable builds a table node for a value the caller set, its keys
// entered as entries in sorted order, the order Marshal writes maps in.
// entered in sorted order, the order Marshal writes maps in.
func newOrderedTable(m map[string]any) *Table {
return orderedTable(m, 0)
}
// orderedTable is newOrderedTable's recursion. The depth bound is the value
// encoder's: a cyclic map stopped here is written by the value writer, which
// reports it instead of running the stack out.
func orderedTable(m map[string]any, depth int) *Table {
t := newTable(m)
for _, k := range slices.Sorted(maps.Keys(m)) {
v := m[k]
_, isMap := v.(map[string]any)
e := t.addValue(k, v, false)
if isMap {
e.child = newOrderedTable(v.(map[string]any))
if depth >= maxEncodeDepth {
continue
}
switch val := v.(type) {
case map[string]any:
e.child = orderedTable(val, depth+1)
case []map[string]any:
e.elements = make([]*Table, len(val))
for i, item := range val {
e.elements[i] = orderedTable(item, depth+1)
}
}
}
return t
@@ -338,6 +431,9 @@ func newOrderedTable(m map[string]any) *Table {
// Delete removes key and everything it holds.
func (t *Table) Delete(key string) {
if t == nil {
return
}
if _, ok := t.values[key]; !ok {
return
}
+141
View File
@@ -4,6 +4,7 @@
package interpres
import (
"reflect"
"slices"
"strings"
"testing"
@@ -416,3 +417,143 @@ func TestDocumentEditPipeline(t *testing.T) {
}
})
}
// TestMarshalDocumentRoundTrips pins that a parsed document written back
// re-parses to the same tree: arrays of tables keep exactly one header per
// element, dotted keys hold their line position without swallowing the keys
// after them, inline tables inside value arrays keep their written order,
// and comments travel with their statements.
func TestMarshalDocumentRoundTrips(t *testing.T) {
tests := []struct {
name string
src string
}{
{"array of tables", "[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n"},
{"array of tables with comments", "# about items\n[[items]] # first\nname = \"a\"\n"},
{"dotted key before a later key", "a.b = 1\nc = 2\n"},
{"dotted keys grouped", "a.b = 1\na.c = 2\nd = 3\n"},
{"dotted key with a nested leaf", "a.b.c = 1\nz = 2\n"},
{"header section after a dotted key", "a.b = 1\n\n[a.x]\ny = 2\n"},
{"inline tables in a value array keep order", "arr = [{y = 1, x = 2}, {second = true, first = false}]\n"},
{"nested array of tables", "[[items]]\nn = 1\n\n[items.sub]\nk = \"v\"\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
doc, err := Parse([]byte(tt.src))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
reparsed, err := Parse(out)
if err != nil {
t.Fatalf("re-parse of %q: %v", out, err)
}
if !reflect.DeepEqual(doc.Map(), reparsed.Map()) {
t.Errorf("round trip changed the tree:\nin: %#v\nout: %#v", doc.Map(), reparsed.Map())
}
if got, want := reparsed.Root().Keys(), doc.Root().Keys(); !slices.Equal(got, want) {
t.Errorf("root keys = %v, want %v", got, want)
}
})
}
}
// TestMarshalDocumentArrayComments pins where the comments of an array of
// tables land: above and beside the [[header]] itself.
func TestMarshalDocumentArrayComments(t *testing.T) {
doc, err := Parse([]byte("# element one\n[[items]] # trailing\nname = \"a\"\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "# element one\n[[items]] # trailing\nname = \"a\"\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetReplacesTableNode pins that Set over a key holding a table
// rebuilds the node, so the new map's keys are the ones written.
func TestTableSetReplacesTableNode(t *testing.T) {
doc, err := Parse([]byte("[cache]\nz = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("cache", map[string]any{"a": int64(2)})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "[cache]\na = 2\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetNestedArraysOfTables pins that a value set through the edit API
// carries its arrays of tables into the header form.
func TestTableSetNestedArraysOfTables(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("t", map[string]any{"items": []map[string]any{{"n": int64(1)}, {"n": int64(2)}}})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if !strings.Contains(string(out), "[[t.items]]") {
t.Errorf("output = %q, want the array of tables under a header", out)
}
}
// TestTableSetCyclicMapErrors pins that a cyclic map set through the edit API
// reaches the depth limit instead of the stack.
func TestTableSetCyclicMapErrors(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
m := map[string]any{}
m["self"] = m
doc.Set("cyclic", m)
if _, err := Marshal(doc); err == nil || !strings.Contains(err.Error(), "nests deeper") {
t.Errorf("err = %v, want the depth-limit complaint", err)
}
}
// TestDocumentNilSafety pins that the nil document answers its readers
// instead of panicking, the contract Root already carries.
func TestDocumentNilSafety(t *testing.T) {
var doc *Document
if doc.Map() != nil {
t.Errorf("Map = %v", doc.Map())
}
if doc.Footer() != nil {
t.Errorf("Footer = %v", doc.Footer())
}
doc.SetFooter([]string{"x"})
if e, ok := doc.Get("k"); e != nil || ok {
t.Errorf("Get = %v, %v", e, ok)
}
if _, ok := doc.GetString("k"); ok {
t.Error("GetString on a nil document reports a value")
}
if _, ok := doc.GetTable("k"); ok {
t.Error("GetTable on a nil document reports a value")
}
doc.Set("k", 1)
doc.Delete("k")
if keys := doc.Root().Keys(); keys != nil {
t.Errorf("Keys = %v", keys)
}
if doc.Root().Entries() != nil {
t.Errorf("Entries = %v", doc.Root().Entries())
}
}
+183 -37
View File
@@ -43,8 +43,8 @@ func (e *encoder) writeDocument(doc *Document) error {
}
// writeDocumentFooter writes the comment lines that follow the last
// statement, each separated from it by a blank line, the shape the parser
// reads them back from.
// statement. The parser collects them wherever they sit after it, so the
// writer needs no blank line of its own to have them read back.
func (e *encoder) writeDocumentFooter(footer []string) {
for _, line := range footer {
e.buf.WriteString("# ")
@@ -53,8 +53,9 @@ func (e *encoder) writeDocumentFooter(footer []string) {
}
}
// writeTableEntries writes one table's entries in written order at the given
// header path, nil for the document root, whose keys need no header.
// writeTableEntries writes one table at the given header path, nil for the
// document root, whose keys need no header: the blank line, the comments,
// the header line with its trailing comment, then the body.
func (e *encoder) writeTableEntries(t *Table, path []string) error {
if t == nil {
return nil
@@ -66,77 +67,222 @@ func (e *encoder) writeTableEntries(t *Table, path []string) error {
if err := e.writeKeyPath(path); err != nil {
return err
}
e.buf.WriteString("]\n")
e.buf.WriteString("]")
if tr := t.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
e.buf.WriteByte('\n')
}
e.buf.WriteByte('\n')
}
return e.writeTableBody(t, path)
}
// writeTableBody writes one table's entries: the value lines first, in
// written order, then the header sections. In a valid document every line at
// one level precedes the headers below it, so the split reorders nothing;
// what it prevents is a table a dotted key introduced, which the parse nests
// as a sub-table at the position of a line, from swallowing the lines that
// follow it into its header.
func (e *encoder) writeTableBody(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if !e.isLineEntry(entry) {
continue
}
if err := e.writeLineEntry(entry, path); err != nil {
return err
}
}
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if _, isTables := entry.Value().([]map[string]any); isTables && len(entry.Elements()) > 0 {
for i, el := range entry.Elements() {
elemPath := append(append([]string{}, path...), entry.Key())
e.writeBlankLine()
if i == 0 {
e.writeComments(entry.Comments())
}
e.buf.WriteString("[[")
if err := e.writeKeyPath(elemPath); err != nil {
return err
}
e.buf.WriteString("]]\n")
if err := e.writeTableEntries(el, elemPath); err != nil {
return err
}
}
continue
}
if child := entry.Table(); child != nil && !entry.Inline() {
headerPath := append(append([]string{}, path...), entry.Key())
if err := e.writeTableEntries(child, headerPath); err != nil {
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
// A dotted table writes as lines above; its own header-form
// sub-tables are sections the document placed after those lines,
// so the section pass reaches through the dotted entry.
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if err := e.writeDocumentEntry(entry, path); err != nil {
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeDottedSections writes the header-form sub-tables of a dotted table:
// the sections the document placed after the dotted lines, reached through
// the dotted entry itself.
func (e *encoder) writeDottedSections(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeSectionEntry writes one entry the line pass left behind: a table or
// an array of tables under its header, at the path this level carries.
func (e *encoder) writeSectionEntry(entry *Entry, path []string) error {
if _, isTables := entry.Value().([]map[string]any); isTables {
// An array of tables keeps its header form, one element per header
// with the element's own comments above it; the body that follows is
// the element's, with no header of its own to repeat.
elemPath := append(append([]string{}, path...), entry.Key())
for i, el := range entry.Elements() {
e.writeBlankLine()
if i == 0 {
e.writeComments(entry.Comments())
}
e.writeComments(el.Comments())
e.buf.WriteString("[[")
if err := e.writeKeyPath(elemPath); err != nil {
return err
}
e.buf.WriteString("]]")
if tr := el.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
if err := e.writeTableBody(el, elemPath); err != nil {
return err
}
}
return nil
}
headerPath := append(append([]string{}, path...), entry.Key())
return e.writeTableEntries(entry.Table(), headerPath)
}
// isLineEntry reports whether an entry writes as one or more "key = value"
// lines at its own level: a value, an inline table, or a table a dotted key
// introduced, which goes back as dotted keys. An emptied array of tables
// counts as one only so the line pass can drop it, the omission the value
// encoder applies to an empty array of tables too.
func (e *encoder) isLineEntry(entry *Entry) bool {
if child := entry.Table(); child != nil {
return entry.Inline() || child.dotted
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return len(entry.Elements()) == 0
}
return true
}
// writeLineEntry writes one entry as lines at this level, and drops an
// emptied array of tables, which has no TOML form.
func (e *encoder) writeLineEntry(entry *Entry, path []string) error {
if child := entry.Table(); child != nil && !entry.Inline() {
return e.writeDottedTable(child, append(append([]string{}, path...), entry.Key()))
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return nil
}
return e.writeDocumentEntry(entry)
}
// writeDottedTable writes a table a dotted key introduced as one dotted line
// per leaf, in written order: `a.b = 1`. A sub-table the document added
// under a header stays a section and is left to the section pass.
func (e *encoder) writeDottedTable(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && !entry.Inline() && !child.dotted {
continue
}
leafPath := append(append([]string{}, path...), entry.Key())
if child := entry.Table(); child != nil && !entry.Inline() {
if err := e.writeDottedTable(child, leafPath); err != nil {
return err
}
continue
}
e.writeComments(entry.Comments())
if err := e.writeKeyPath(leafPath); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
}
return nil
}
// writeDocumentEntry writes one "key = value" line of a document, with the
// comments the key carried. A value that is itself an inline table renders
// inline from its node, in the written order.
func (e *encoder) writeDocumentEntry(entry *Entry, path []string) error {
func (e *encoder) writeDocumentEntry(entry *Entry) error {
e.writeComments(entry.Comments())
if err := e.writeKey(entry.Key()); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
return nil
}
// writeEntryValueNodes writes the value of a document entry. An inline table
// node keeps the written key order even inside a value array, where the
// ordinary value writer would sort the keys.
func (e *encoder) writeEntryValueNodes(entry *Entry) error {
if child := entry.Table(); child != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err
}
if tr := entry.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
} else if arr, ok := entry.Value().([]any); ok {
elems := entry.Elements()
e.buf.WriteByte('[')
for i, item := range arr {
if i > 0 {
e.buf.WriteString(", ")
}
if i < len(elems) && elems[i] != nil {
if err := e.writeInlineTableNode(elems[i]); err != nil {
return err
}
continue
}
if err := e.writeValue(item); err != nil {
return err
}
}
e.buf.WriteByte('\n')
return nil
}
if err := e.writeValue(entry.Value()); err != nil {
e.buf.WriteByte(']')
} else if err := e.writeValue(entry.Value()); err != nil {
return err
}
if tr := entry.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
return nil
}
+97 -139
View File
@@ -132,6 +132,10 @@ type encoder struct {
// their indentation.
inlineDepth int
// valueDepth is the nesting level of boxed containers the value writer
// walks, the bound a cyclic map or slice hits instead of the stack.
valueDepth int
// limit is the column at which an inline table is broken; only a
// measuring encoder raises it.
limit int
@@ -950,9 +954,11 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, path encPath, for
return nil
}
// writeArrayValue writes a value array from its reflect value, its elements
// written one by one, each falling back to the boxed path only where the
// boxed rules rewrite it.
// writeArrayValue writes a value array from its reflect value. Only the
// plain scalar kinds reach it: addArrayValue's direct path takes nothing but
// arrays whose elements are scalar kinds without methods, so one writer per
// element needs no boxing branch and no depth walk of its own; the slices
// and maps the boxed rules re-write travel the normaliseValue route.
func (e *encoder) writeArrayValue(v reflect.Value, depth int) error {
if atDepthLimit(depth) {
return errDepthLimit()
@@ -962,7 +968,7 @@ func (e *encoder) writeArrayValue(v reflect.Value, depth int) error {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeArrayElem(v.Index(i), depth+1); err != nil {
if err := e.writeScalarValue(v.Index(i)); err != nil {
return err
}
}
@@ -970,129 +976,6 @@ func (e *encoder) writeArrayValue(v reflect.Value, depth int) error {
return nil
}
// writeArrayElem writes one element of a value array. The kinds the boxed
// rules rewrite are handed to normaliseValue and the boxed writer; the rest
// write directly, including nested arrays and inline tables.
func (e *encoder) writeArrayElem(v reflect.Value, depth int) error {
if v.Kind() == reflect.Interface {
if v.IsNil() {
return fmt.Errorf("interpres: cannot encode nil value")
}
v = v.Elem()
}
switch v.Kind() {
case reflect.Slice, reflect.Array:
return e.writeArrayValue(v, depth)
case reflect.Map:
return e.writeInlineMapFromReflect(v, depth)
}
if _, isMarshaler := marshalerOf(v); isMarshaler {
return e.writeNormalisedElem(v, depth)
}
if _, isText, err := textValue(v); err != nil || isText {
if err != nil {
return err
}
return e.writeNormalisedElem(v, depth)
}
switch v.Kind() {
case reflect.String, reflect.Bool, reflect.Int, reflect.Int8, reflect.Int16,
reflect.Int32, reflect.Int64, reflect.Uint, reflect.Uint8, reflect.Uint16,
reflect.Uint32, reflect.Uint64, reflect.Float32, reflect.Float64:
if t := v.Type(); t != durationType && t != numberType {
return e.writeScalarValue(v)
}
}
return e.writeNormalisedElem(v, depth)
}
// writeNormalisedElem normalises one element through the boxed rules and
// writes the result.
func (e *encoder) writeNormalisedElem(v reflect.Value, depth int) error {
val, err := normaliseValueAt(v, depth)
if err != nil {
return err
}
return e.writeValue(val)
}
// writeInlineMapFromReflect renders a map from its reflect value as an inline
// table, the shape the boxed path gives the table elements of a value array:
// sorted keys, single line when it fits, across lines when it does not.
func (e *encoder) writeInlineMapFromReflect(v reflect.Value, depth int) error {
if e.limit >= noInlineBreak {
return e.writeInlineMapFlatReflect(v, depth)
}
flat := e.flat()
err := flat.writeInlineMapFlatReflect(v, depth)
if err != nil {
flat.release()
return err
}
fits := e.column()+flat.buf.Len() <= e.limit
if fits {
e.buf.Write(flat.buf.Bytes())
}
flat.release()
if fits {
return nil
}
return e.writeInlineMapMultilineReflect(v, depth)
}
// writeInlineMapFlatReflect renders the single-line form.
func (e *encoder) writeInlineMapFlatReflect(v reflect.Value, depth int) error {
if v.Type().Key().Kind() != reflect.String {
return fmt.Errorf("map key must be string, got %s", v.Type().Key())
}
keys := make([]string, 0, v.Len())
for _, k := range v.MapKeys() {
keys = append(keys, k.String())
}
slices.Sort(keys)
e.buf.WriteByte('{')
for i, k := range keys {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeKey(k); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeArrayElem(v.MapIndex(reflect.ValueOf(k)), depth+1); err != nil {
return err
}
}
e.buf.WriteByte('}')
return nil
}
// writeInlineMapMultilineReflect renders the across-lines form.
func (e *encoder) writeInlineMapMultilineReflect(v reflect.Value, depth int) error {
keys := make([]string, 0, v.Len())
for _, k := range v.MapKeys() {
keys = append(keys, k.String())
}
slices.Sort(keys)
e.buf.WriteString("{\n")
e.inlineDepth++
for _, k := range keys {
e.writeInlineIndent()
if err := e.writeKey(k); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeArrayElem(v.MapIndex(reflect.ValueOf(k)), depth+1); err != nil {
return err
}
e.buf.WriteString(",\n")
}
e.inlineDepth--
e.writeInlineIndent()
e.buf.WriteByte('}')
return nil
}
// marshalerOf finds the Marshaler a value carries: on the value itself, or on
// its address, so a pointer-receiver MarshalTOML is found on an addressable
// struct field or slice element, exactly as textMarshalerOf finds MarshalText.
@@ -1364,7 +1247,13 @@ func textMarshalerOf(v reflect.Value) (encoding.TextMarshaler, bool) {
return nil, false
}
// isTableElementType reports whether a slice of t is an array of tables. A
// pointer element is looked through, so []*T behaves as []T: the empty-slice
// decision and the header form agree on the same element type.
func isTableElementType(t reflect.Type) bool {
for t.Kind() == reflect.Pointer {
t = t.Elem()
}
switch t.Kind() {
case reflect.Struct:
return !isScalarStruct(t) && !isTextMarshalerType(t)
@@ -1395,12 +1284,24 @@ func (e *encoder) writeBlankLine() {
}
func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
// The walk that built the document checked the context on its own
// cadence; the emission of a large document is long enough to need the
// same checks, or a cancellation that lands after the walk would wait a
// full document out.
if err := e.checkCtx(); err != nil {
return err
}
if e.opts.layout == LayoutKindGrouped {
// Scalars first, then inline sub-tables as value lines, then the
// remaining tables as headers, then arrays of tables. Each pass walks
// the entries in place; grouping copies of them cost the encoder a
// third of its allocations for nothing.
for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
kv := &doc.entries[i]
if kv.kind != entryScalar {
continue
@@ -1413,6 +1314,11 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
// header of this document: a line written after a [header] would be
// read back as part of that table.
for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
t := &doc.entries[i]
if t.kind != entryTable {
continue
@@ -1424,6 +1330,11 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
t.emitted = inlined
}
for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
t := &doc.entries[i]
if t.kind != entryTable || t.emitted {
continue
@@ -1441,12 +1352,22 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
}
}
for i := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
a := &doc.entries[i]
if a.kind != entryArray {
continue
}
path := append(append([]string{}, prefix...), a.key)
for j, sub := range a.docs {
if j%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
e.writeBlankLine()
if j == 0 {
e.writeComments(a.comments)
@@ -1469,7 +1390,12 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
// own section content; the emitter still writes sub-documents as separate
// nested blocks, so a "" sub-keyed scalar following a header for the same
// section is impossible in practice (struct fields are visited in order).
for _, ent := range doc.entries {
for i, ent := range doc.entries {
if i%ctxCheckInterval == 0 {
if err := e.checkCtx(); err != nil {
return err
}
}
switch ent.kind {
case entryScalar:
if err := e.writeKV(&ent); err != nil {
@@ -1699,9 +1625,15 @@ func (e *encoder) writeValue(val any) error {
case float64:
return e.writeFloat(v)
case time.Time:
if err := wholeMinuteOffset(v); err != nil {
return err
}
e.buf.WriteString(offsetString(v))
return nil
case OffsetDateTime:
if err := wholeMinuteOffset(v.Time); err != nil {
return err
}
e.buf.WriteString(v.String())
return nil
case LocalDateTime:
@@ -1714,19 +1646,19 @@ func (e *encoder) writeValue(val any) error {
e.buf.WriteString(v.String())
return nil
case []any:
e.buf.WriteByte('[')
for i, item := range v {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeValue(item); err != nil {
return err
}
if err := e.enterValueDepth(); err != nil {
return err
}
e.buf.WriteByte(']')
return nil
err := e.writeValueArray(v)
e.valueDepth--
return err
case map[string]any:
return e.writeInlineMap(v)
if err := e.enterValueDepth(); err != nil {
return err
}
err := e.writeInlineMap(v)
e.valueDepth--
return err
case nil:
return fmt.Errorf("interpres: cannot encode nil value")
default:
@@ -1734,6 +1666,32 @@ func (e *encoder) writeValue(val any) error {
}
}
// enterValueDepth counts one level of boxed container nesting. The
// reflection walk has its own bound, but a tree built by hand and written
// through the boxed path carries no walk, so the writer bounds itself.
func (e *encoder) enterValueDepth() error {
e.valueDepth++
if e.valueDepth > maxEncodeDepth {
return errDepthLimit()
}
return nil
}
// writeValueArray writes a boxed value array, one writeValue per element.
func (e *encoder) writeValueArray(v []any) error {
e.buf.WriteByte('[')
for i, item := range v {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeValue(item); err != nil {
return err
}
}
e.buf.WriteByte(']')
return nil
}
// writeInlineMap renders m as a TOML inline table, on one line when it fits
// there and across lines when it does not.
func (e *encoder) writeInlineMap(m map[string]any) error {
+113 -1
View File
@@ -67,7 +67,7 @@ func TestMarshalFloatSpecials(t *testing.T) {
}
}
func TestMarshalFloatNormalizesNegativeZero(t *testing.T) {
func TestMarshalFloatNormalisesNegativeZero(t *testing.T) {
// The output contract normalises negative zero to "0.0".
type Cfg struct {
Z float64 `toml:"z"`
@@ -2284,3 +2284,115 @@ func TestEmitFieldComments(t *testing.T) {
}
})
}
// errWriter fails every write with a fixed error.
type errWriter struct{ err error }
func (w errWriter) Write([]byte) (int, error) { return 0, w.err }
// TestMarshalWrite covers the streaming entry: the happy path with options
// and a failing writer.
func TestMarshalWrite(t *testing.T) {
var buf bytes.Buffer
err := MarshalWrite(&buf, map[string]any{"b": 2, "a": 1})
if err != nil {
t.Fatalf("MarshalWrite: %v", err)
}
// A map carries no order, so the writer uses the sorted one.
if buf.String() != "a = 1\nb = 2\n" {
t.Errorf("output = %q", buf.String())
}
writeErr := errors.New("boom")
if err := MarshalWrite(errWriter{writeErr}, map[string]any{"a": 1}); !errors.Is(err, writeErr) {
t.Errorf("err = %v, want the write error wrapped", err)
}
}
// TestMarshalRejectsUnsupportedKinds pins the clear error a field of a kind
// TOML cannot carry raises, through the struct walk.
func TestMarshalRejectsUnsupportedKinds(t *testing.T) {
tests := []struct {
name string
value any
}{
{"func", struct {
F func() `toml:"f"`
}{}},
{"chan", struct {
C chan int `toml:"c"`
}{}},
{"complex", struct {
Z complex128 `toml:"z"`
}{}},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
_, err := Marshal(tt.value)
if err == nil {
t.Fatalf("Marshal accepted %#v", tt.value)
}
if !strings.Contains(err.Error(), "cannot encode") {
t.Errorf("err = %v, want the cannot-encode complaint", err)
}
})
}
}
// TestMarshalOmitsEmptyPointerTableSlice pins that an empty slice of pointer
// tables is omitted, the rule its non-pointer form already follows.
func TestMarshalOmitsEmptyPointerTableSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
out, err := Marshal(struct {
Items []*item `toml:"items"`
}{})
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if len(out) != 0 {
t.Errorf("output = %q, want the empty array of tables omitted", out)
}
}
// TestMarshalRejectsNonWholeMinuteOffset pins that a zone offset carrying
// seconds is refused instead of silently losing them.
func TestMarshalRejectsNonWholeMinuteOffset(t *testing.T) {
z := time.FixedZone("", 57*60+44)
_, err := Marshal(struct {
Stamp time.Time `toml:"stamp"`
}{Stamp: time.Date(1890, 1, 1, 12, 0, 0, 0, z)})
if err == nil || !strings.Contains(err.Error(), "not a whole number of minutes") {
t.Errorf("err = %v, want the whole-minute offset complaint", err)
}
_, err = Marshal(struct {
Stamp OffsetDateTime `toml:"stamp"`
}{Stamp: OffsetDateTime{time.Date(1890, 1, 1, 12, 0, 0, 0, z)}})
if err == nil || !strings.Contains(err.Error(), "not a whole number of minutes") {
t.Errorf("err = %v, want the whole-minute offset complaint for the wrapper", err)
}
}
// cancelOnMarshal cancels the context the encode runs under, the moment its
// method is called, so the emission that follows is already past the walk's
// own checks.
type cancelOnMarshal struct {
cancel context.CancelFunc
}
func (c cancelOnMarshal) MarshalTOML() (any, error) {
c.cancel()
return int64(1), nil
}
// TestMarshalContextCancelsDuringEmission pins that a context cancelled
// between the walk and the emission stops the encode instead of writing the
// whole document out.
func TestMarshalContextCancelsDuringEmission(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
value := map[string]any{"k": cancelOnMarshal{cancel}}
if _, err := MarshalContext(ctx, value); !errors.Is(err, context.Canceled) {
t.Errorf("err = %v, want the cancellation", err)
}
}
+8 -7
View File
@@ -3,8 +3,8 @@
// Command basic demonstrates decoding and encoding a TOML document with
// interpres. It exercises struct mapping, arrays of tables, Marshaler
// customisation, the Decoder's strict mode, and the Encoder's policy
// options, covering every feature a regular user would reach for.
// customisation, and the Encoder's policy options, covering every feature a
// regular user would reach for.
package main
import (
@@ -18,8 +18,7 @@ import (
)
// document is a small but realistic configuration: it has scalars, a
// sub-table, an array of tables, and a date-time. We pick a 32-bit port so
// the demonstration also covers overflow-safe integer conversion.
// sub-table, an array of tables, and a date-time.
const document = `
title = "interpres demo"
launched = 2024-11-04T09:00:00Z
@@ -62,9 +61,11 @@ type User struct {
Admin bool `toml:"admin"`
}
// Port is a typed alias that controls how its value appears in TOML. The
// MarshalTOML hook returns a string, so a Port field is rendered as
// "host:port" instead of the raw integer.
// Port is a typed string alias that carries a Marshaler. The MarshalTOML
// hook returns the string unchanged, so a Port field is rendered as the
// string it holds, a string the encoding would print the same way without
// the hook; the demonstration that a Marshaler reshapes a value is
// Endpoint's below.
type Port string
func (p Port) MarshalTOML() (any, error) {
+72 -50
View File
@@ -16,8 +16,10 @@
// doc, err := interpres.Parse(data)
// tree := doc.Map()
//
// A Decoder allows strict decoding that rejects keys without a matching
// struct field, mirroring the RejectUnknownFields option of encoding/json/v2.
// Strict decoding that rejects keys without a matching struct field is an
// option, mirroring the RejectUnknownMembers option of encoding/json/v2:
//
// err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
package interpres
import (
@@ -212,7 +214,7 @@ func ParseFile(path string) (*Document, error) {
// Valid reports whether data is a valid TOML document: nil when the parser
// accepts it, and the parse error when it does not. It is the library call
// the -validate mode of interpres-decode is built on, and it reads nothing
// the --validate mode of interpres-decode is built on, and it reads nothing
// but the bytes it is given.
func Valid(data []byte) error {
_, err := ParseMapContext(context.Background(), data)
@@ -256,9 +258,6 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
return tree, &Document{root: p.doc, footer: p.footer}, nil
}
// Unmarshal parses a TOML document and stores the result in the value pointed
// to by v. v is typically a pointer to a struct or to a map[string]any.
//
// Unmarshal parses a TOML document and stores the result in the value pointed
// to by v. v is typically a pointer to a struct or to a map[string]any.
// Options tune the call; with none, unknown keys are ignored, numbers are
@@ -266,7 +265,9 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
//
// Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a
// case-insensitive match on the field name when no tag is present. A tag of
// "-" skips the field.
// "-" skips the field. Two document keys that differ only in case and both
// match one field resolve deterministically: the lexicographically greater
// one wins, the same key winning every run.
//
// A destination implementing Unmarshaler receives the parsed value as it is,
// a TOML string fills a destination implementing encoding.TextUnmarshaler, and
@@ -281,12 +282,12 @@ func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error {
// ParseAs decodes a TOML document into T in one call, the generic shorthand
// for Unmarshal with a destination variable:
//
// cfg, err := interpres.ParseAs[Config](data)
// cfg, err := interpres.ParseAs[Config](data, interpres.RejectUnknownFields(true))
//
// The zero T comes back with the error.
func ParseAs[T any](data []byte) (T, error) {
// The options are Unmarshal's. The zero T comes back with the error.
func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error) {
var v T
err := Unmarshal(data, &v)
err := Unmarshal(data, &v, opts...)
return v, err
}
@@ -315,10 +316,19 @@ func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...Unmarshal
// UnmarshalRead reads the document from r and decodes it into v, the
// streaming-shaped entry the json/v2 vocabulary uses. The reader is
// consumed in full, because the parser scans its source in place; the
// options and the behaviour are Unmarshal's.
// consumed in full, because the parser scans its source in place; with
// MaxInputSize set, reading stops one byte past the limit so the size the
// option bounds is the memory held, not what a reader is drained into first.
// The options and the behaviour are Unmarshal's.
func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error {
data, err := io.ReadAll(r)
s := settingsFor(opts)
var data []byte
var err error
if s.maxInputSize > 0 {
data, err = io.ReadAll(io.LimitReader(r, int64(s.maxInputSize)+1))
} else {
data, err = io.ReadAll(r)
}
if err != nil {
return fmt.Errorf("interpres: read: %w", err)
}
@@ -338,18 +348,19 @@ func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error {
// tree path, so every option means the same thing on every document.
type UnmarshalOption func(*decodeSettings)
// decodeSettings is the option carrier of one decode call.
// decodeSettings is the option carrier of one decode call. The context is
// not one: it arrives as its own argument, because every entry point names it
// explicitly.
type decodeSettings struct {
disallowUnknown bool
useNumber bool
maxDepth int
maxInputSize int
localLoc *time.Location
ctx context.Context
}
func settingsFor(opts []UnmarshalOption) *decodeSettings {
s := &decodeSettings{ctx: context.Background()}
s := &decodeSettings{}
for _, opt := range opts {
opt(s)
}
@@ -452,8 +463,8 @@ type Marshaler interface {
// argument is whatever the parser produced for that key: one of string,
// bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime,
// []any, or map[string]any. A tree built by hand may carry a plain time.Time
// where the parser would put an OffsetDateTime, and a Decoder configured with
// UseNumber a Number.
// where the parser would put an OffsetDateTime, and NumbersAsLiterals a
// Number.
//
// UnmarshalTOML may parse, inspect, or transform the value however it likes,
// then store the result by mutating its receiver through the standard
@@ -461,7 +472,7 @@ type Marshaler interface {
// reflect.Value.Set or by reassigning fields through a pointer the receiver
// holds).
//
// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the
// UnmarshalTOML is invoked from Unmarshal and its siblings when the
// destination type implements the interface. The decoder does not need to
// consult the concrete return value; whatever the receiver stores is kept.
//
@@ -482,16 +493,21 @@ type UnmarshalerContext interface {
}
// Marshal returns the TOML encoding of v. The output is valid TOML 1.1.
// Options tune the emission; with none, the layout groups entries by kind,
// empty arrays emit and sub-tables take the header form.
//
// Marshal traverses v using reflection and applies the following rules:
//
// - The top-level value must be a struct or a map[string]V. Pointers are
// followed; a nil top-level pointer is an error.
// - The top-level value must be a struct, a map[string]V or an OrderedMap
// (or a non-nil pointer to one). A Document writes itself back, and a nil
// one is an error.
// - Struct fields are matched by `toml:"name"` tag (case-insensitive
// fallback to field name; `-` skips). The tag options `omitzero` (skip
// the zero value of the field's type) and `omitempty` (skip an empty
// slice, array, or map) drop a field from the output on encode; the
// decoder ignores them. Anonymous (embedded) fields without a tag are
// fallback to field name; `-` skips). The tag option `omitzero` skips a
// field holding the zero value of its type (a type with an IsZero method
// decides through it), and `omitempty` skips a value that is empty in
// the encoding/json sense: an empty string, a zero number, false, a nil
// pointer or interface, and an empty slice, array or map. The decoder
// ignores both options. Anonymous (embedded) fields without a tag are
// inlined.
// - Maps use sorted keys for deterministic output.
// - Slices and arrays of structs or maps become TOML arrays of tables; a
@@ -502,10 +518,12 @@ type UnmarshalerContext interface {
// inline table.
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
// and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/
// LocalTime (local variants). A date-time writes its seconds only when the value carries
// them, and drops the trailing zeros of a fractional second.
// - A table element of a value array, and a sub-table inlined by
// Encoder.InlineTables, is written as an inline table, across lines when it
// LocalTime (local variants). A date-time writes its seconds only when
// the value carries them, and drops the trailing zeros of a fractional
// second. A zone offset that is not a whole number of minutes is refused,
// because TOML has no form that carries its seconds.
// - A table element of a value array, and a sub-table the InlineTables
// option inlines, is written as an inline table, across lines when it
// does not fit one.
// - Values implementing Marshaler are encoded by calling MarshalTOML and
// using its result.
@@ -516,14 +534,10 @@ type UnmarshalerContext interface {
//
// Marshal rejects a value that nests deeper than 10000 levels with an error
// naming the limit, so cyclic data is reported instead of running the stack
// out. The output is not guaranteed to be byte-identical to
// the input that produced v: comments, whitespace, key order (for maps),
// string quoting style, and the choice between `[table]` headers and inline
// tables are not preserved.
//
// Marshal returns the TOML encoding of v. Options tune the emission; with
// none, the layout groups entries by kind, empty arrays emit and sub-tables
// take the header form.
// out. The output is not guaranteed to be byte-identical to the input that
// produced v: comments, whitespace, key order (for maps), string quoting
// style, and the choice between `[table]` headers and inline tables are not
// preserved.
//
// Marshal is equivalent to MarshalContext with context.Background.
func Marshal(v any, opts ...MarshalOption) ([]byte, error) {
@@ -548,12 +562,13 @@ type Statement struct {
}
// Statements reads a TOML document from r and returns an iterator over its
// top-level statements in written order: key/value statements, a [table]
// header as one statement carrying its Table node, and an [[array of
// tables]] as one statement per element, each with the element's node and
// its Index. Iteration stops at the first error, which arrives as the second
// value, and at a false yield: a caller that breaks after the statement it
// wanted reads no further ones.
// top-level statements in written order: key/value statements, including a
// value that is an array or an inline table, a [table] header as one
// statement carrying its Table node, and an [[array of tables]] as one
// statement per element, each with the element's node and its Index.
// Iteration stops at the first error, which arrives as the second value, and
// at a false yield: a caller that breaks after the statement it wanted reads
// no further ones.
//
// The reader is consumed in full before the first statement is yielded,
// because the parser scans the source in place; processing the yielded
@@ -572,8 +587,12 @@ func Statements(r io.Reader) iter.Seq2[Statement, error] {
return
}
for _, e := range doc.Root().Entries() {
if els := e.Elements(); len(els) > 0 {
for i, el := range els {
// Only an array of tables yields per element, the branch the
// write side takes too: a value array is one statement whatever
// its elements, and an emptied array of tables holds no element
// to yield.
if _, isTables := e.Value().([]map[string]any); isTables && len(e.Elements()) > 0 {
for i, el := range e.Elements() {
if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: el, Index: i}, nil) {
return
}
@@ -648,14 +667,14 @@ const (
// interpres.InlineTables(60))
type MarshalOption func(*encodeSettings)
// encodeSettings is the option carrier of one encode call.
// encodeSettings is the option carrier of one encode call. As on the decode
// side, the context arrives as its own argument.
type encodeSettings struct {
ctx context.Context
cfg encodeConfig
}
func settingsForEncode(opts []MarshalOption) *encodeSettings {
s := &encodeSettings{ctx: context.Background(), cfg: encodeConfig{layout: LayoutKindGrouped}}
s := &encodeSettings{cfg: encodeConfig{layout: LayoutKindGrouped}}
for _, opt := range opts {
opt(s)
}
@@ -681,6 +700,7 @@ func (s *encodeSettings) marshal(ctx context.Context, v any) ([]byte, error) {
// Layout sets the layout the encoder writes a document's entries in:
// LayoutKindGrouped, the default, reorders them scalars first, then tables,
// then arrays of tables; LayoutKindDeclaration preserves declaration order.
// A value the two constants do not name behaves as LayoutKindGrouped.
func Layout(kind LayoutKind) MarshalOption {
return func(s *encodeSettings) { s.cfg.layout = kind }
}
@@ -727,7 +747,9 @@ func InlineTables(threshold int) MarshalOption {
// Go doc comments are not visible to reflection, so the tag is the channel
// that carries the text. Off by default, and a field without a `comment=`
// option prints none. Multi-line comments carry newlines in the tag, each
// line printed with its own "# " marker.
// line printed with its own "# " marker. The tag's options separate with
// commas, so the comment text itself cannot carry one; the first comma ends
// it.
func EmitFieldComments(v bool) MarshalOption {
return func(s *encodeSettings) { s.cfg.emitFieldComments = v }
}
+108
View File
@@ -895,3 +895,111 @@ func TestParseCRLFDocument(t *testing.T) {
t.Errorf("tree = %v", tree)
}
}
// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U:
// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the
// greatest scalar value parses.
func TestParseUnicodeEscapeBoundaries(t *testing.T) {
bad := []struct {
name string
in string
}{
{"high surrogate", `a = "\ud800"`},
{"low surrogate", `a = "\udfff"`},
{"past the greatest scalar", `a = "\U00110000"`},
{"signed short escape", `a = "\u+041"`},
{"negative long escape", `a = "\U-0000001"`},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := Parse([]byte(tt.in))
if err == nil {
t.Fatalf("Parse accepted %q", tt.in)
}
})
}
tree, err := ParseMap([]byte("a = \"\\U0010FFFF\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "􏿿" {
t.Errorf("a = %q", tree["a"])
}
}
// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a
// date-time with a component out of range is rejected as a date-time, not
// left to the number decoder's complaint.
func TestParseRejectsOutOfRangeDateTimes(t *testing.T) {
bad := []struct {
name string
in string
}{
{"hour 24", "a = 1979-05-27T24:00:00Z"},
{"minute 60", "a = 1979-05-27T07:60:00Z"},
{"second 60", "a = 1979-05-27T07:32:60Z"},
{"month 13", "a = 1979-13-27T07:32:00Z"},
{"day 32", "a = 1979-05-32T07:32:00Z"},
{"february the thirtieth", "a = 1979-02-30"},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseMap([]byte(tt.in))
if err == nil {
t.Fatalf("ParseMap accepted %q", tt.in)
}
if !strings.Contains(err.Error(), "invalid date-time") {
t.Errorf("err = %v, want the date-time complaint", err)
}
})
}
}
// TestParseMultilineStringEdges pins the carriage-return and delimiter rules
// of multi-line strings: a bare CR right after the opening delimiter is the
// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the
// content survives.
func TestParseMultilineStringEdges(t *testing.T) {
_, err := ParseMap([]byte("a = \"\"\"\rX\"\"\""))
if err == nil || !strings.Contains(err.Error(), "bare carriage return") {
t.Errorf("err = %v, want the bare-CR error after the delimiter", err)
}
tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "X\r\nY" {
t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"])
}
}
// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes
// before the closing delimiter of a basic multi-line string are content, and
// more than five are the error.
func TestParseMultilineBasicDelimiterRuns(t *testing.T) {
tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != `end"` {
t.Errorf("a = %q", tree["a"])
}
_, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\""))
if err == nil || !strings.Contains(err.Error(), "too many") {
t.Errorf("err = %v, want the too-many-delimiters error", err)
}
}
// TestParseLineEndingBackslashEdges pins the line-ending backslash at the
// very end of the input and before a bare CR.
func TestParseLineEndingBackslashEdges(t *testing.T) {
bad := []string{
"a = \"\"\"x \\\\",
"a = \"\"\"x \\\\\rZ\"\"\"",
}
for _, in := range bad {
if _, err := ParseMap([]byte(in)); err == nil {
t.Errorf("ParseMap accepted %q", in)
}
}
}
+9 -7
View File
@@ -1,7 +1,7 @@
# interpres.
#
# Everything below the variable block is the standard recipe set from the `justfile`
# skill, identical in every repository; project values live in the variable block only.
# Everything below the variable block is the standard recipe set, identical in
# every repository; project values live in the variable block only.
binary := "interpres-decode"
package := "./cmd/interpres-decode"
@@ -32,7 +32,7 @@ test:
close($c);
die qq{no total line in coverage.out\n} unless defined $total;
printf qq{Total coverage: %s%%\n}, $total;
exit($total < 82 ? 1 : 0);
exit($total < 80 ? 1 : 0);
# The same suite under the race detector. The expensive one.
race:
@@ -94,13 +94,13 @@ dev:
# Runs the official toml-test compliance suite in both directions, decoder and encoder, against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
toml-test: build
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
coverage-html: test
go tool cover -html=coverage.out -o coverage.html
# Cross-compile smoke: the library and the command build for the foreign architectures and the browser and edge runtimes; not a gate, it is a nightly convenience and runs std-lib only.
# Cross-compile smoke: the library and the command build for the foreign architectures and the browser and edge runtimes; not a gate, it is a hand-run convenience and runs std-lib only.
cross:
GOARCH=arm64 go build ./...
GOARCH=loong64 go build ./...
@@ -115,10 +115,12 @@ cross:
example:
go run ./examples/basic
# The release pre-flight from the release skill, in one command: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release. Not a gate, it is the checklist before a release may even be discussed.
# The release pre-flight, in one command: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release. Not a gate, it is the checklist before a release may even be discussed.
release-check version:
#!/usr/bin/env perl
my ($version) = @ARGV;
# The version arrives through the recipe interpolation: just does not hand
# positional arguments to a shebang script's @ARGV.
my $version = "{{version}}";
$version =~ m{\Av?\d+\.\d+\.\d+\z} or die qq{usage: just release-check X.Y.Z\n};
my $branch = qx{git rev-parse --abbrev-ref HEAD};
chomp $branch;
+34 -32
View File
@@ -1,4 +1,4 @@
.TH INTERPRES-DECODE 1 "2026-09-21" "interpres 2.0.0" "User Commands"
.TH INTERPRES-DECODE 1 "2026-09-22" "interpres 2.0.0" "User Commands"
.SH NAME
interpres-decode \- TOML validator and toml-test harness adapter
.SH SYNOPSIS
@@ -6,38 +6,38 @@ interpres-decode \- TOML validator and toml-test harness adapter
[\fIFLAGS\fR]
.br
.B interpres-decode
.B \-encode
.B \-\-encode
.br
.B interpres-decode
.B \-validate
.B \-\-validate
[\fIFILE\fR...]
.br
.B interpres-decode
.B \-validate
.B \-\-validate
[\fIDIRECTORY\fR...]
.br
.B interpres-decode
.B \-json
.B \-\-json
.br
.B interpres-decode
.B \-struct
.B \-\-struct
.br
.B interpres-decode
.B \-schema
.B \-\-schema
\fITYPE\fR
\fIFILE.go\fR
.br
.B interpres-decode
.B \-version
.B \-\-version
.SH DESCRIPTION
.B interpres-decode
is the toml-test harness adapter in both directions and a TOML validator.
Without a mode flag it reads one TOML document from standard input and writes
the toml-test tagged-JSON representation to standard output.
.B \-encode
.B \-\-encode
reads a tagged-JSON description from standard input and writes the TOML
document it describes.
.B \-validate
.B \-\-validate
parses each named file, or standard input when none are named, and prints one
line per invalid document to standard error; a named directory is walked for
.B .toml
@@ -45,48 +45,49 @@ files, every one validated, and the walk closes with a summary on standard
error naming the counts. The name
.B \-
means standard input.
.B \-json
prints plain indented JSON instead of the tagged form.
.B \-struct
.B \-\-json
prints plain indented JSON instead of the tagged form; it shapes the decoding
output only, so it is rejected together with the mode flags.
.B \-\-struct
prints a Go struct definition inferred from the document on standard input.
.B \-schema
.B \-\-schema
writes a TOML template for the struct type
\fITYPE\fR
declared in the Go source file
\fIFILE.go\fR,
taking the key names, comments and defaults from the fields' tags.
.B \-version
.B \-\-version
prints the binary's version and exits.
.PP
The mode flags
.BR \-validate ,
.BR \-encode ,
.B \-struct
.BR \-\-validate ,
.BR \-\-encode ,
.B \-\-struct
and
.B \-schema
.B \-\-schema
cannot be combined.
.SH OPTIONS
.TP
.B \-validate
.B \-\-validate
Validate the documents instead of emitting tagged JSON.
.TP
.B \-encode
.B \-\-encode
Read tagged JSON from standard input and write TOML instead.
.TP
.B \-json
.B \-\-json
With the default mode, print plain indented JSON instead of tagged JSON.
.TP
.B \-struct
.B \-\-struct
Infer a Go struct definition from the document on standard input and print it.
.TP
.BI \-schema " TYPE"
.BI \-\-schema " TYPE"
Write a TOML template for the struct type \fITYPE\fR; the Go source file
follows as the first argument.
.TP
.B \-version
.B \-\-version
Print the version and exit.
.TP
.B \-h
.B \-\-help
Print the usage.
.SH EXIT STATUS
.TP
@@ -95,11 +96,12 @@ The document parsed and the output was written; in validate mode, every
document parsed.
.TP
.B 1
Adapter: a parse error. Validate: at least one document is invalid.
Adapter: a parse error. Validate: at least one document is invalid. Struct:
the document on standard input failed to parse.
.TP
.B 2
A usage error, a read failure, malformed tagged JSON, or a value with no TOML
representation.
A usage error, a read or write failure, malformed tagged JSON, or a value
with no TOML representation.
.SH EXAMPLES
Decode a document into tagged JSON:
.PP
@@ -113,7 +115,7 @@ Validate a directory of configuration, with the summary:
.PP
.nf
.RS
interpres-decode -validate configs/
interpres-decode \-\-validate configs/
.RE
.fi
.PP
@@ -121,7 +123,7 @@ Infer a Go type from a document:
.PP
.nf
.RS
interpres-decode -struct < config.toml > config.go
interpres-decode \-\-struct < config.toml > config.go
.RE
.fi
.PP
@@ -129,7 +131,7 @@ Write the template back from the type:
.PP
.nf
.RS
interpres-decode -schema Config config.go
interpres-decode \-\-schema Config config.go
.RE
.fi
.SH SEE ALSO
+72 -21
View File
@@ -48,6 +48,13 @@ type parser struct {
dotted map[string]bool
arrays map[string]bool
// scopeMarks records the definition-map entries added under an array of
// tables, keyed by that array's path, so a new element's reset drops
// exactly what the previous element added. Without it the reset scans
// every map for the prefix, which a document with many elements and many
// definitions outside them turns quadratic.
scopeMarks map[string][]string
currentPath []string
// keys interns key strings: a document that repeats a key across
@@ -196,6 +203,7 @@ func (p *parser) markHeader(pk string) {
p.headers = make(map[string]bool, 4)
}
p.headers[pk] = true
p.trackScope(pk)
}
func (p *parser) markFrozen(pk string) {
@@ -203,6 +211,7 @@ func (p *parser) markFrozen(pk string) {
p.frozen = make(map[string]bool, 4)
}
p.frozen[pk] = true
p.trackScope(pk)
}
func (p *parser) markDotted(pk string) {
@@ -210,6 +219,7 @@ func (p *parser) markDotted(pk string) {
p.dotted = make(map[string]bool, 4)
}
p.dotted[pk] = true
p.trackScope(pk)
}
func (p *parser) markArray(pk string) {
@@ -217,6 +227,22 @@ func (p *parser) markArray(pk string) {
p.arrays = make(map[string]bool, 2)
}
p.arrays[pk] = true
p.trackScope(pk)
}
// trackScope records a definition entry under every array of tables it falls
// inside, so resetScopeUnder can drop it when a later element opens. An entry
// under no array, such as every definition before the first header, needs no
// record: no reset can ever name it.
func (p *parser) trackScope(pk string) {
for arr := range p.arrays {
if strings.HasPrefix(pk, arr+"\x00") {
if p.scopeMarks == nil {
p.scopeMarks = make(map[string][]string, 2)
}
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
}
}
}
// internKey returns the shared string for key bytes. The lookup works on the
@@ -482,9 +508,14 @@ func (p *parser) parseKeyValue() error {
if p.doc != nil {
node := p.currentNode
if len(rest) > 0 {
// The tables a dotted key builds hold the position of a line, so
// the write side marks them and gives each leaf back as a dotted
// key rather than a header that would swallow the lines after it.
node = node.addTable(first, dests[0])
node.dotted = true
for i, k := range rest[:len(rest)-1] {
node = node.addTable(k, dests[i+1])
node.dotted = true
}
}
_, inline := val.(map[string]any)
@@ -570,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) {
// resetScopeUnder forgets the definition records nested under key, which
// belong to the previous element of an array of tables: headers, frozen
// inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element.
// fresh in the new element. The records to drop are the ones the element
// added, which scopeMarks holds; the array's own entry, and everything
// outside it, keep their place.
func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00"
p.resetMapUnder(p.headers, prefix)
p.resetMapUnder(p.frozen, prefix)
p.resetMapUnder(p.dotted, prefix)
p.resetMapUnder(p.arrays, prefix)
}
// resetMapUnder deletes the entries m holds under prefix. An empty or
// unallocated map holds none, so the common case walks nothing.
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
if len(m) == 0 {
return
pk := pathKey(key)
for _, k := range p.scopeMarks[pk] {
delete(p.headers, k)
delete(p.frozen, k)
delete(p.dotted, k)
delete(p.arrays, k)
}
for k := range m {
if strings.HasPrefix(k, prefix) {
delete(m, k)
}
if p.scopeMarks != nil {
p.scopeMarks[pk] = nil
}
}
@@ -708,7 +733,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value")
}
if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
}
// A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -719,7 +744,11 @@ func (p *parser) parseAtom() (any, error) {
tok = tok + " " + string(p.src[timeStart:p.pos])
}
}
if v, ok := parseDateTime(tok); ok {
v, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
return v, nil
}
v, err := decodeNumber(tok)
@@ -758,6 +787,20 @@ func hasHighByte(s string) bool {
return false
}
// invalidUTF8Offset returns the offset of the first byte in s that does not
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
// that is invalid rather than the end of the token around it.
func invalidUTF8Offset(s string) int {
for i := 0; i < len(s); {
r, size := utf8.DecodeRuneInString(s[i:])
if r == utf8.RuneError && size == 1 {
return i
}
i += size
}
return -1
}
// --- strings ---------------------------------------------------------------
func (p *parser) parseBasicString() (string, error) {
@@ -895,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed.
// A newline immediately after the opening delimiter is trimmed, and it is
// a newline: a bare CR here is the bare-CR error like anywhere else, not
// a newline to trim.
if !p.eof() && p.peek() == '\r' {
if next, ok := p.peekAt(1); !ok || next != '\n' {
return "", p.errf("bare carriage return is not allowed in a string")
}
p.pos++
}
if !p.eof() && p.peek() == '\n' {
@@ -1047,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) {
}
hex := string(p.src[p.pos : p.pos+n])
p.pos += n
v, err := strconv.ParseInt(hex, 16, 64)
// ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
// read would let "\U-0000001" through the range checks below only to
// write U+FFFD for a document the grammar rejects.
v, err := strconv.ParseUint(hex, 16, 32)
if err != nil {
return 0, p.errf("invalid unicode escape \\%s", hex)
}
+23 -10
View File
@@ -1,4 +1,7 @@
#!/usr/bin/env perl
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
# SPDX-License-Identifier: MIT
# docs-drift compares the toml-test suite counts the documentation names with
# the run this repository produces now. A corpus change moves the counts, and
# README.md and docs/ARCHITECTURE.md quote them; this is the check that keeps
@@ -6,26 +9,36 @@
use v5.40;
my $out = qx{toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1 2>&1};
my $out = qx{toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1 2>&1};
die "docs-drift: toml-test failed to run; build the adapter first (just build)\n"
if !defined $out || $out =~ /not found|No such file/;
# A red run is a failure to answer, not a drift: the counts it prints describe
# a suite that did not pass, and sending the maintainer to correct counts that
# are correct would be the wrong diagnosis.
die "docs-drift: the toml-test run failed; fix the suite before comparing counts\n"
if $? != 0;
my ($valid) = $out =~ /valid tests:\s+(\d+) passed/;
my ($invalid) = $out =~ /invalid tests:\s+(\d+) passed/;
die "docs-drift: could not read the suite counts from the toml-test output\n"
unless defined $valid && defined $invalid;
print "docs-drift: the suite now stands at $valid valid and $invalid invalid cases\n";
my %live;
for my $kind (qw(valid invalid encoder)) {
my ($passed) = $out =~ /\b$kind tests:\s+(\d+) passed/;
die "docs-drift: could not read the $kind count from the toml-test output\n"
unless defined $passed;
my ($failed) = $out =~ /\b$kind tests:\s+\d+ passed, (\d+) failed/;
die "docs-drift: the $kind run has $failed failures; fix the suite first\n"
if defined $failed && $failed != 0;
$live{$kind} = $passed;
}
print "docs-drift: the suite now stands at $live{valid} valid, $live{invalid} invalid and $live{encoder} encoder cases\n";
my $drift = 0;
for my $file ('README.md', 'docs/ARCHITECTURE.md') {
open(my $fh, '<', $file) or die "docs-drift: cannot read $file: $!\n";
my $text = do { local $/; <$fh> };
close($fh);
while ($text =~ /(\d+)\s+(valid|invalid)/g) {
while ($text =~ /(\d+)\s+(valid|invalid|encoder)/g) {
my ($quoted, $kind) = ($1, $2);
my $live = $kind eq 'valid' ? $valid : $invalid;
if ($quoted != $live) {
print "docs-drift: $file quotes $quoted $kind cases, the suite says $live\n";
if ($quoted != $live{$kind}) {
print "docs-drift: $file quotes $quoted $kind cases, the suite says $live{$kind}\n";
$drift = 1;
}
}
+369 -191
View File
@@ -22,6 +22,14 @@ import (
// document through the tree path, so the observable behaviour is the tree
// path's, exactly. A targeted parse either completes with the result the
// tree path would give, or it erases itself.
//
// One difference the two paths cannot share: a parse error or a cancellation
// deep in the document leaves the statements before it already written into
// the destination, where the tree path, which parses the whole document
// before it decodes any of it, writes nothing. The value layer shares this
// with encoding/json, whose Unmarshal also leaves a partial destination
// behind a mid-document failure; a destination that must stay untouched on
// error is a destination the caller resets.
var errTargetFallback = errors.New("interpres: targeted decode falls back to the tree path")
// targetCache holds whether a destination type may take the targeted parse.
@@ -33,10 +41,12 @@ var targetCache sync.Map // reflect.Type -> bool
var mapStringAnyType = reflect.TypeFor[map[string]any]()
// typeTargetable reports whether decoding into the struct type t can use the
// targeted parse. The one structural ban is untagged embedded maps: their
// filler-key rule lives in the tree decode, and a targeted document that
// meets an unknown table would need a subtree of it. Everything else is safe
// to attempt, because the value layer is the ordinary decode and every
// targeted parse. The structural bans are the shapes whose tree behaviour
// the skeleton cannot model: untagged embedded maps, an OrderedMap anywhere a
// table opens, and a custom decode hook on any table the parse would enter
// directly (a struct field, a map field, or the element of a table slice),
// because the tree hands a hook the whole parsed value. Everything else is
// safe to attempt, because the value layer is the ordinary decode and every
// mismatch falls back.
func typeTargetable(t reflect.Type) bool {
if t == nil || t.Kind() != reflect.Struct || t == orderedMapType {
@@ -59,22 +69,41 @@ func scanTargetable(t reflect.Type, seen map[reflect.Type]bool) bool {
return false
}
for _, loc := range cachedStructSchema(t).byName {
ft := derefType(t.FieldByIndex(loc.index).Type)
if !scanTargetableField(derefType(t.FieldByIndex(loc.index).Type), seen) {
return false
}
}
return true
}
// scanTargetableField reports whether one field's type is safe for the
// targeted skeleton to fill directly.
func scanTargetableField(ft reflect.Type, seen map[reflect.Type]bool) bool {
switch ft.Kind() {
case reflect.Struct:
if ft == orderedMapType {
return false
}
if ft.Kind() != reflect.Struct || isScalarStruct(ft) {
continue
if isScalarStruct(ft) {
return true
}
// A struct field with a custom decode hook receives the whole parsed
// value from the tree decode; the targeted skeleton never builds that
// value for a table it enters directly, so the hook must win.
// A struct the parse enters directly never builds the whole value
// the tree hands a hook, so the hook must win.
if implementsDecodeHook(ft) || implementsDecodeHook(reflect.PointerTo(ft)) {
return false
}
if !scanTargetable(ft, seen) {
return scanTargetable(ft, seen)
case reflect.Map:
return !implementsDecodeHook(ft) && !implementsDecodeHook(reflect.PointerTo(ft))
case reflect.Slice, reflect.Array:
et := derefType(ft.Elem())
if et == orderedMapType {
return false
}
if et.Kind() == reflect.Struct && !isScalarStruct(et) {
return scanTargetableField(et, seen)
}
return true
}
return true
}
@@ -110,20 +139,45 @@ func canTargetDecode(v any) bool {
// targetTable is one open table of the targeted parse: the struct (or map)
// value its keys fill, the schema that resolves them (nil for a map or sink
// destination), the absolute path its errors wrap, and whether it collects
// the keys no field claims for the strict check. A sink is the destination
// an unknown subtree gets: its statements parse for the syntax and definition
// contracts, and its values are discarded.
// destination), and the absolute path its errors wrap. A sink is the
// destination an unknown subtree gets: its statements parse for the syntax
// and definition contracts, and its values are discarded. The strict and
// required findings live in the parser's per-address store, not here,
// because the table objects a dotted descent builds are transient while the
// destination is not.
type targetTable struct {
rv reflect.Value
schema *structSchema
path []string
sink bool
keys []string // the keys defined in the table, interned; struct
// tables keep theirs per destination address instead
strict bool
unknown string // strict: the smallest unclaimed key so far
resolvedSeen map[string]bool // the schema keys resolved so far, for required
// strict is the strict-decode setting the table was opened with, carried
// into the per-address strict state on first sight.
strict bool
// arrayElem marks a sink created as the element of an unknown array of
// tables: a dotted key may not enter it, the tree's own rule for an
// array, while a [sub-table] header may, through the last element.
arrayElem bool
keys []string // the keys defined in a sink, as full path keys
}
// strictState is the strict and required bookkeeping of one struct
// destination, keyed by the value's address.
type strictState struct {
path []string
typ reflect.Type
schema *structSchema
strict bool
unknown string // strict: the smallest unclaimed key so far
resolved map[string]bool
}
// arrayFill tracks how many elements of one fixed-size array the document
// has filled, with what the length mismatch the tree decode reports needs:
// the array's type and its field's path.
type arrayFill struct {
next int
typ reflect.Type
path []string
}
// targetParser parses a document straight into a struct destination. It
@@ -139,61 +193,53 @@ type targetParser struct {
rootT *targetTable
cur *targetTable
// arrayNext tracks how many elements of a fixed-size array the document
// has filled, per field address.
arrayNext map[uintptr]int
// arrayFills counts the elements of each fixed-size array the document
// has filled, per field address, in the order the arrays were met.
arrayFills map[uintptr]*arrayFill
fillOrder []*arrayFill
// appendedHere records the slice fields this document's [[headers]] have
// filled: a slice the caller prefilled is replaced by the tree decode,
// not appended to, so the first header over one falls back.
appendedHere map[uintptr]bool
// opened registers every opened table by its path key, sinks included: a
// later header or dotted key meets the table the tree already built.
opened map[string]*targetTable
// keyAssigned records the fields a key statement assigned directly, per
// table address: an array-of-tables header over such a field is the
// tree's `not an array of tables` error, where a header over a
// header-built array appends.
keyAssignedKeys map[uintptr]map[string]bool
// tableKeysByAddr holds the defined keys of one struct destination, keyed
// by the value's address: the table object a dotted descent builds is
// transient, the destination is not.
tableKeysByAddr map[uintptr][]string
}
// markKeyAssigned records that a key statement assigned the field, and
// keyAssigned reports that state. Sinks keep no such bookkeeping.
func (tp *targetParser) markKeyAssigned(t *targetTable, key string) {
if t.sink || t.schema == nil || !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return
}
addr := t.rv.Addr().Pointer()
if tp.keyAssignedKeys == nil {
tp.keyAssignedKeys = make(map[uintptr]map[string]bool, 8)
}
if tp.keyAssignedKeys[addr] == nil {
tp.keyAssignedKeys[addr] = make(map[string]bool, 8)
}
tp.keyAssignedKeys[addr][key] = true
}
// mapKeysByAddr holds the keys the document has defined in one map
// destination, keyed the same way: the destination map the caller
// prefilled is not the parser's state, and a key it holds is not the
// duplicate a key the document repeats is.
mapKeysByAddr map[uintptr]map[string]bool
func (tp *targetParser) keyAssigned(t *targetTable, key string) bool {
if t.sink || t.schema == nil || !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return false
}
addr := t.rv.Addr().Pointer()
return tp.keyAssignedKeys[addr][key]
// strictByAddr holds each destination's strict and required findings,
// with strictOrder keeping the document order they first appeared in.
strictByAddr map[uintptr]*strictState
strictOrder []uintptr
}
// tableHas reports whether key is already defined in the table. A struct
// table's keys are interned parser strings, so the linear scan compares
// against a handful of short keys, cheaper than hashing a per-table map.
// Struct tables keep their keys by destination address, because the table
// object a dotted descent builds is transient while the destination is not.
// Struct tables keep their keys by destination address, and map tables keep
// theirs there too, because the table object a dotted descent builds is
// transient while the destination is not, and the destination map's own
// contents are the caller's, not the document's.
func (tp *targetParser) tableHas(t *targetTable, key string) bool {
switch {
case t.sink:
return slices.Contains(t.keys, key)
case t.schema == nil:
return t.rv.Kind() == reflect.Map && t.rv.MapIndex(reflect.ValueOf(key)).IsValid()
if t.rv.Kind() != reflect.Map || !t.rv.CanAddr() {
return false
}
return tp.mapKeysByAddr[t.rv.Addr().Pointer()][key]
default:
return slices.Contains(tp.tableKeys(t), key)
}
@@ -217,6 +263,17 @@ func (tp *targetParser) tableMark(t *targetTable, key string) {
case t.sink:
t.keys = append(t.keys, key)
case t.schema == nil:
if !t.rv.CanAddr() || t.rv.Kind() != reflect.Map {
return
}
addr := t.rv.Addr().Pointer()
if tp.mapKeysByAddr == nil {
tp.mapKeysByAddr = make(map[uintptr]map[string]bool, 8)
}
if tp.mapKeysByAddr[addr] == nil {
tp.mapKeysByAddr[addr] = make(map[string]bool, 8)
}
tp.mapKeysByAddr[addr][key] = true
default:
if !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return
@@ -229,32 +286,49 @@ func (tp *targetParser) tableMark(t *targetTable, key string) {
}
}
// resolvedHas reports whether the resolved schema key has been seen, the
// check a required tag runs: the duplicate bookkeeping tracks the key as the
// document wrote it, the required bookkeeping the key as the schema
// resolved it.
func (t *targetTable) resolvedHas(key string) bool {
return t.resolvedSeen[key]
// strictState returns the strict and required bookkeeping of the struct
// destination t fills, registering it on first sight so a finding recorded
// on a transient table survives the table.
func (tp *targetParser) strictState(t *targetTable) *strictState {
addr := t.rv.Addr().Pointer()
if st, ok := tp.strictByAddr[addr]; ok {
return st
}
st := &strictState{
path: slices.Clone(t.path),
typ: t.rv.Type(),
schema: t.schema,
strict: t.strict,
resolved: make(map[string]bool, 8),
}
tp.strictByAddr[addr] = st
tp.strictOrder = append(tp.strictOrder, addr)
return st
}
func (t *targetTable) markResolved(key string) {
// markResolved records that the document resolved the schema key, the check
// a required tag runs: the duplicate bookkeeping tracks the key as the
// document wrote it, the required bookkeeping the key as the schema resolved
// it.
func (tp *targetParser) markResolved(t *targetTable, key string) {
if t.schema == nil || len(t.schema.required) == 0 {
return
}
if t.resolvedSeen == nil {
t.resolvedSeen = make(map[string]bool, 8)
if !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return
}
t.resolvedSeen[key] = true
tp.strictState(t).resolved[key] = true
}
// recordStrictUnknown remembers the key no field claims when strict decoding
// is on: the smallest one is reported, the tree decode's own choice.
func (t *targetTable) recordStrictUnknown(key string) {
if !t.strict {
func (tp *targetParser) recordStrictUnknown(t *targetTable, key string) {
if !t.strict || t.schema == nil || !t.rv.CanAddr() || t.rv.Kind() != reflect.Struct {
return
}
if t.unknown == "" || key < t.unknown {
t.unknown = key
st := tp.strictState(t)
if st.unknown == "" || key < st.unknown {
st.unknown = key
}
}
@@ -274,12 +348,13 @@ func parseIntoTargeted(ctx context.Context, data []byte, d *decoder, useNumber b
}
rv := reflect.ValueOf(v)
tp := &targetParser{
parser: &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, useNumber: useNumber},
d: d,
root: rv.Elem(),
arrayNext: make(map[uintptr]int, 4),
keyAssignedKeys: make(map[uintptr]map[string]bool, 8),
opened: make(map[string]*targetTable, 8),
parser: &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, useNumber: useNumber},
d: d,
root: rv.Elem(),
arrayFills: make(map[uintptr]*arrayFill, 4),
appendedHere: make(map[uintptr]bool, 4),
opened: make(map[string]*targetTable, 8),
strictByAddr: make(map[uintptr]*strictState, 8),
}
tp.rootT = &targetTable{rv: tp.root, schema: schemaRef(tp.root.Type()), strict: d.disallowUnknown}
tp.tables = append(tp.tables, tp.rootT)
@@ -326,35 +401,57 @@ func (tp *targetParser) run() error {
}
// reportDeferred raises the decode-stage findings in the tree decode's
// order: the root table first, then the opened tables in document order. The
// tree decode reports them after a full parse, so a later parse error always
// won; here the parse has already completed.
// order. A fixed-size array the document under-filled is the length mismatch
// the tree decode raises, and it comes first. Then the unknown keys, before
// the required ones, because the tree decode meets an unknown key while it
// assigns and checks a table's required keys only once the whole table has
// been; within each class the order is the order the destinations first
// appeared in, the document's own order.
func (tp *targetParser) reportDeferred() error {
for _, t := range append([]*targetTable{tp.rootT}, tp.tables...) {
if t.sink {
for _, f := range tp.fillOrder {
if f.next != f.typ.Len() {
return wrapTablePath(f.path, fmt.Errorf("interpres: cannot assign %d elements to %s", f.next, f.typ))
}
}
for _, addr := range tp.strictOrder {
if st := tp.strictByAddr[addr]; st.strict && st.unknown != "" {
return wrapTablePath(st.path, fmt.Errorf("interpres: unknown field %q for %s", st.unknown, st.typ))
}
}
for _, addr := range tp.strictOrder {
st := tp.strictByAddr[addr]
if st.schema == nil {
continue
}
if t.strict && t.unknown != "" {
return tp.wrapTableErr(t, fmt.Errorf("interpres: unknown field %q for %s", t.unknown, t.rv.Type()))
}
if t.schema != nil {
for _, key := range t.schema.required {
if !t.resolvedHas(key) {
return tp.wrapTableErr(t, fmt.Errorf("interpres: missing required key %q", key))
}
for _, key := range st.schema.required {
if !st.resolved[key] {
return wrapTablePath(st.path, fmt.Errorf("interpres: missing required key %q", key))
}
}
}
return nil
}
// wrapTableErr wraps a table's finding the way the tree decode wraps it: the
// wrapTablePath wraps a table's finding the way the tree decode wraps it: the
// root speaks for itself, a nested table gains its path.
func (tp *targetParser) wrapTableErr(t *targetTable, err error) error {
if len(t.path) == 0 {
func wrapTablePath(path []string, err error) error {
if len(path) == 0 {
return err
}
return &DecodeError{Path: Path(slices.Clone(t.path)), Err: err}
return &DecodeError{Path: Path(slices.Clone(path)), Err: err}
}
// arrayFillFor returns the fill record of the fixed-size array fv, keyed by
// its address, created with its field's path on first sight.
func (tp *targetParser) arrayFillFor(fv reflect.Value, path []string) *arrayFill {
addr := fv.Addr().Pointer()
if f, ok := tp.arrayFills[addr]; ok {
return f
}
f := &arrayFill{typ: fv.Type(), path: path}
tp.arrayFills[addr] = f
tp.fillOrder = append(tp.fillOrder, f)
return f
}
// --- headers ---------------------------------------------------------------
@@ -474,11 +571,11 @@ func (tp *targetParser) descendOne(parent *targetTable, key string, abs []string
return nil, errTargetFallback
}
if existing := parent.rv.MapIndex(reflect.ValueOf(key)); existing.IsValid() && !existing.IsNil() {
return &targetTable{rv: existing.Elem(), path: abs}, nil
return &targetTable{rv: existing, path: slices.Clone(abs)}, nil
}
next := reflect.MakeMap(elemT)
parent.rv.SetMapIndex(reflect.ValueOf(key), next)
return &targetTable{rv: next, path: abs}, nil
return &targetTable{rv: next, path: slices.Clone(abs)}, nil
}
resolved := key
loc, ok := parent.schema.byName[key]
@@ -487,18 +584,18 @@ func (tp *targetParser) descendOne(parent *targetTable, key string, abs []string
loc, ok = parent.schema.byName[resolved]
}
if !ok {
parent.recordStrictUnknown(key)
tp.recordStrictUnknown(parent, key)
if opened, ok := tp.opened[pathKey(abs)]; ok {
return opened, nil
}
if tp.tableHas(parent, key) {
return nil, tp.errf("key %q is not a table", key)
}
sink := &targetTable{sink: true, path: abs}
sink := &targetTable{sink: true, path: slices.Clone(abs)}
tp.opened[pathKey(abs)] = sink
return sink, nil
}
parent.markResolved(resolved)
tp.markResolved(parent, resolved)
fv, err := fieldByIndex(parent.rv, loc.index)
if err != nil {
return nil, errTargetFallback
@@ -509,10 +606,12 @@ func (tp *targetParser) descendOne(parent *targetTable, key string, abs []string
// openValueTable opens a table scope over a placed field value, allocating a
// nil pointer on the way. The rules mirror the tree decode's own type
// decisions: a struct enters, a map enters (allocated when nil), an array of
// tables enters its last element, and anything else is a type mismatch the
// tree decode reports, so it falls back. A scalar field the table keys
// already define is the tree's `key is not a table` error, checked against
// parent, the table the key belongs to.
// tables enters its last filled element, a slice enters the last element of
// an array this document's [[headers]] built (a prefilled slice is a table
// the tree decode rejects, so it falls back), and anything else is a type
// mismatch the tree decode reports, so it falls back too. A scalar field the
// table keys already define is the tree's `key is not a table` error,
// checked against parent, the table the key belongs to.
func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, key string, abs []string, strict bool) (*targetTable, error) {
if fv.Kind() == reflect.Pointer {
if fv.IsNil() {
@@ -528,7 +627,7 @@ func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, ke
if isScalarStruct(fv.Type()) {
return nil, errTargetFallback
}
return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: abs, strict: strict}, nil
return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: slices.Clone(abs), strict: strict}, nil
case reflect.Map:
if fv.Type().Key().Kind() != reflect.String {
return nil, errTargetFallback
@@ -536,11 +635,12 @@ func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, ke
if fv.IsNil() {
fv.Set(reflect.MakeMap(fv.Type()))
}
return &targetTable{rv: fv, path: abs}, nil
return &targetTable{rv: fv, path: slices.Clone(abs)}, nil
case reflect.Slice:
if fv.Len() == 0 {
// No [[header]] ever filled it, so the tree holds a map here and
// its decode raises the type mismatch.
if !tp.appendedHere[fv.Addr().Pointer()] {
// No [[header]] of this document filled it, so the tree holds a
// map here, whose decode raises the type mismatch; a prefilled
// slice is the tree's replacement case, not a table to enter.
return nil, errTargetFallback
}
et := derefType(fv.Type().Elem())
@@ -548,6 +648,22 @@ func (tp *targetParser) openValueTable(fv reflect.Value, parent *targetTable, ke
return nil, errTargetFallback
}
return &targetTable{rv: fv.Index(fv.Len() - 1), schema: schemaRef(et), path: elementPath(abs, key, fv.Len()-1), strict: strict}, nil
case reflect.Array:
fill := tp.arrayFillFor(fv, append(slices.Clone(parent.path), key))
if fill.next == 0 {
return nil, errTargetFallback
}
et := derefType(fv.Type().Elem())
if et.Kind() == reflect.Struct {
if isScalarStruct(et) {
return nil, errTargetFallback
}
return &targetTable{rv: fv.Index(fill.next - 1), schema: schemaRef(et), path: elementPath(abs, key, fill.next-1), strict: strict}, nil
}
if et.Kind() == reflect.Map && et.Key().Kind() == reflect.String {
return &targetTable{rv: fv.Index(fill.next - 1), path: elementPath(abs, key, fill.next-1)}, nil
}
return nil, errTargetFallback
}
if tp.tableHas(parent, key) {
return nil, tp.errf("key %q is not a table", key)
@@ -578,12 +694,30 @@ func (tp *targetParser) appendArrayTable(key []string) error {
if tp.parser.dotted[pk] || tp.parser.headers[pk] {
return tp.errf("key %q is not an array of tables", leaf)
}
// A leaf the document already defined as a value or a table is the tree
// parser's own parse error, and a leaf an earlier [[header]] defined
// opens a new element; the arrays map, read before this header marks it,
// is what tells the two apart. A sink parent holds no destination state
// worth consulting.
if !parent.sink && !tp.parser.arrays[pk] && tp.tableHas(parent, leaf) {
return tp.errf("key %q is not an array of tables", leaf)
}
// A new element starts a fresh scope, exactly as the tree parser's own
// header does: sub-headers, inline freezes and nested arrays from the
// previous element no longer apply.
tp.parser.resetScopeUnder(key)
tp.parser.markArray(pk)
elem, err := tp.appendElement(parent, leaf, key)
if err != nil {
return err
}
if elem.sink {
// A sink the element scope reuses ([[a.b]] over an unknown a, the
// parent sink) starts the new element with no keys, the way a known
// array's element does.
elem.keys = nil
}
tp.tableMark(parent, leaf)
if !elem.sink {
tp.tables = append(tp.tables, elem)
@@ -593,9 +727,9 @@ func (tp *targetParser) appendArrayTable(key []string) error {
}
// appendElement appends one element to the array the leaf names in parent
// and returns its table. A leaf no field claims sinks; a field whose array
// element kind cannot be a table falls back, the tree decode owning the type
// error.
// and returns its table. A leaf no field claims sinks, a fresh namespace per
// element; a field whose array element kind cannot be a table falls back,
// the tree decode owning the type error.
func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []string) (*targetTable, error) {
if parent.sink {
return parent, nil
@@ -608,10 +742,15 @@ func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []st
gk := reflect.ValueOf(leaf)
var arr reflect.Value
if existing := parent.rv.MapIndex(gk); existing.IsValid() && !existing.IsNil() {
if existing.Elem().Kind() != reflect.Slice {
// A map[string]any destination boxes its arrays in the
// interface; a typed map hands the slice itself.
if existing.Kind() == reflect.Interface {
existing = existing.Elem()
}
if existing.Kind() != reflect.Slice {
return nil, tp.errf("key %q is not an array of tables", leaf)
}
arr = existing.Elem()
arr = existing
}
var elem reflect.Value
switch {
@@ -645,62 +784,64 @@ func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []st
loc, ok = parent.schema.byName[resolved]
}
if !ok {
parent.recordStrictUnknown(leaf)
if opened, ok := tp.opened[pathKey(parent.path)]; ok {
return opened, nil
}
if tp.tableHas(parent, leaf) {
return nil, tp.errf("key %q is not an array of tables", leaf)
}
sink := &targetTable{sink: true, path: parent.path}
tp.opened[pathKey(parent.path)] = sink
tp.recordStrictUnknown(parent, leaf)
// Every element is a fresh namespace, the way a known array's is,
// registered under the header's path so a [sub-table] header reaches
// the last element, the tree's rule for a header under an array of
// tables; a dotted key skips it, the tree's rule for an array.
sink := &targetTable{sink: true, arrayElem: true, path: slices.Clone(key)}
tp.opened[pathKey(key)] = sink
return sink, nil
}
parent.markResolved(resolved)
if tp.keyAssigned(parent, resolved) {
// A key statement already assigned the field its own value; the tree
// holds a value array there and its header append is the
// `not an array of tables` parse error.
return nil, tp.errf("key %q is not an array of tables", leaf)
}
tp.markResolved(parent, resolved)
fv, err := fieldByIndex(parent.rv, loc.index)
if err != nil {
return nil, errTargetFallback
}
fieldPath := append(slices.Clone(parent.path), leaf)
if fv.Kind() == reflect.Array {
// A fixed-size array fills position by position; one element too many
// is the length mismatch the tree decode reports.
// is the length mismatch the tree decode reports, and one too few is
// the same mismatch, checked when the parse completes.
fill := tp.arrayFillFor(fv, fieldPath)
et := derefType(fv.Type().Elem())
switch et.Kind() {
case reflect.Struct:
if isScalarStruct(et) {
switch {
case et.Kind() == reflect.Struct && !isScalarStruct(et):
if fill.next >= fv.Len() {
return nil, errTargetFallback
}
addr := fv.Addr().Pointer()
n := tp.arrayNext[addr]
if n >= fv.Len() {
n := fill.next
fill.next = n + 1
return &targetTable{rv: fv.Index(n), schema: schemaRef(et), path: elementPath(parent.path, leaf, n), strict: parent.strict}, nil
case et.Kind() == reflect.Map && et.Key().Kind() == reflect.String:
if fill.next >= fv.Len() {
return nil, errTargetFallback
}
tp.arrayNext[addr] = n + 1
return &targetTable{rv: fv.Index(n), schema: schemaRef(et), path: parent.path, strict: parent.strict}, nil
case reflect.Map:
if et.Key().Kind() != reflect.String {
return nil, errTargetFallback
}
addr := fv.Addr().Pointer()
n := tp.arrayNext[addr]
if n >= fv.Len() {
return nil, errTargetFallback
}
tp.arrayNext[addr] = n + 1
n := fill.next
fill.next = n + 1
elem := reflect.MakeMap(et)
fv.Index(n).Set(elem)
return &targetTable{rv: elem, path: parent.path}, nil
return &targetTable{rv: elem, path: elementPath(parent.path, leaf, n)}, nil
}
return nil, errTargetFallback
}
if fv.Kind() != reflect.Slice {
return nil, tp.errf("key %q is not an array of tables", leaf)
if tp.tableHas(parent, leaf) {
return nil, tp.errf("key %q is not an array of tables", leaf)
}
// The tree builds an array here without asking the destination, and
// its decode answers with the type mismatch; the fallback keeps the
// message the tree path gives.
return nil, errTargetFallback
}
addr := fv.Addr().Pointer()
if !tp.appendedHere[addr] {
if fv.Len() > 0 {
// A slice the caller prefilled is replaced by the tree decode,
// not appended to; the fallback runs the document the tree's way.
return nil, errTargetFallback
}
tp.appendedHere[addr] = true
}
et := derefType(fv.Type().Elem())
switch et.Kind() {
@@ -708,16 +849,38 @@ func (tp *targetParser) appendElement(parent *targetTable, leaf string, key []st
if isScalarStruct(et) {
return nil, errTargetFallback
}
grown := reflect.Append(fv, reflect.New(et).Elem())
// A pointer element is appended as the allocated pointer and filled
// through its pointee, so []*T takes the same path []T does. The
// element the table fills is the slice's own: the value a New built
// stands apart from the backing array.
var el, appended reflect.Value
if fv.Type().Elem().Kind() == reflect.Pointer {
p := reflect.New(et)
el, appended = p.Elem(), p
} else {
el = reflect.New(et).Elem()
appended = el
}
grown := reflect.Append(fv, appended)
fv.Set(grown)
return &targetTable{rv: grown.Index(grown.Len() - 1), schema: schemaRef(et), path: elementPath(parent.path, leaf, grown.Len()-1), strict: parent.strict}, nil
if fv.Type().Elem().Kind() != reflect.Pointer {
el = grown.Index(grown.Len() - 1)
}
return &targetTable{rv: el, schema: schemaRef(et), path: elementPath(parent.path, leaf, grown.Len()-1), strict: parent.strict}, nil
case reflect.Map:
if et.Key().Kind() != reflect.String {
return nil, errTargetFallback
}
grown := reflect.Append(fv, reflect.MakeMap(et))
m := reflect.MakeMap(et)
var appended reflect.Value = m
if fv.Type().Elem().Kind() == reflect.Pointer {
p := reflect.New(et)
p.Elem().Set(m)
appended = p
}
grown := reflect.Append(fv, appended)
fv.Set(grown)
return &targetTable{rv: grown.Index(grown.Len() - 1), path: elementPath(parent.path, leaf, grown.Len()-1)}, nil
return &targetTable{rv: m, path: elementPath(parent.path, leaf, grown.Len()-1)}, nil
}
return nil, errTargetFallback
}
@@ -780,22 +943,18 @@ func (tp *targetParser) parseKeyStatement() error {
if verr != nil {
return verr
}
dupKey := first
if len(dest.path) > 0 || len(rest) > 0 {
full := make([]string, 0, len(dest.path)+len(rest)+1)
full = append(full, dest.path...)
full = append(full, first)
full = append(full, rest...)
dupKey = pathKey(full)
full := append([]string{first}, rest...)
if len(dest.path) > 0 {
full = append(slices.Clone(dest.path), full...)
}
if tp.tableHas(leafTable, dupKey) {
if tp.tableHas(leafTable, pathKey(full)) {
return p.errf("duplicate key %q", leaf)
}
tp.tableMark(leafTable, dupKey)
tp.tableMark(leafTable, pathKey(full))
if m, isMap := val.(map[string]any); isMap {
full := make([]string, 0, len(dest.path)+len(leaf)+1)
full = append(full, dest.path...)
full = append(full, leaf)
// An inline table freezes the whole path the statement wrote,
// intermediate segments included, so no later header or dotted
// key can extend it at any depth.
p.freezeInline(full, m)
}
return nil
@@ -804,27 +963,29 @@ func (tp *targetParser) parseKeyStatement() error {
return p.errf("duplicate key %q", leaf)
}
tp.tableMark(leafTable, leaf)
if dst.Kind() == reflect.Slice || dst.Kind() == reflect.Array {
// Only a field a later [[header]] could append to needs the
// key-assigned record; everything else never checks it.
et := derefType(dst.Type().Elem())
if et.Kind() == reflect.Struct && !isScalarStruct(et) || et.Kind() == reflect.Map {
tp.markKeyAssigned(leafTable, leaf)
}
}
val, err := tp.parseValueInto(dst)
if err != nil {
if errors.Is(err, errTargetFallback) {
return err
}
if _, isSyntax := errors.AsType[*SyntaxError](err); !isSyntax {
// A hook's own failure, which the fallback must not rerun: it
// returns wrapped the way the tree decode wraps a field's.
return newDecodeError(leaf, err)
}
return err
}
if mapDst.IsValid() {
mapDst.SetMapIndex(reflect.ValueOf(leaf), dst)
}
if m, isMap := val.(map[string]any); isMap {
// An inline table freezes its paths; the abs slice is built for it
// alone, after the parse proved one is needed.
abs := make([]string, 0, len(dest.path)+len(leaf)+1)
// An inline table freezes the whole path the statement wrote,
// intermediate segments included; the slice is built for it alone,
// after the parse proved one is needed.
abs := make([]string, 0, len(dest.path)+len(rest)+1)
abs = append(abs, dest.path...)
abs = append(abs, leaf)
abs = append(abs, first)
abs = append(abs, rest...)
p.freezeInline(abs, m)
}
return nil
@@ -847,7 +1008,10 @@ func (tp *targetParser) parseValueInto(dst reflect.Value) (any, error) {
return nil, err
}
if err := tp.d.assign(v, dst); err != nil {
return nil, errTargetFallback
// The hook has run; falling back would run it a second time on
// tree path, so its error returns as the tree path's own, for
// the caller to wrap the way the tree decode wraps a field's.
return nil, err
}
return v, nil
}
@@ -890,7 +1054,16 @@ func (tp *targetParser) parseValueInto(dst reflect.Value) (any, error) {
}
}
tok := tp.scanNumberToken()
if dtv, dtok := parseDateTime(tok); dtok {
if hasHighByte(tok) && invalidUTF8Offset(tok) >= 0 {
// The token route the targeted parse takes validates UTF-8 the
// way the tree scanner does, on the byte that does not decode.
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
}
dtv, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
if err := tp.d.assign(dtv, dst); err != nil {
return nil, errTargetFallback
}
@@ -1091,10 +1264,10 @@ func (tp *targetParser) descendDotted(dest *targetTable, first string, rest []st
loc, found = tbl.schema.byName[resolved]
}
if !found {
tbl.recordStrictUnknown(leaf)
tp.recordStrictUnknown(tbl, leaf)
return reflect.Value{}, reflect.Value{}, leafTable, false, nil
}
tbl.markResolved(resolved)
tp.markResolved(tbl, resolved)
fv, ferr := fieldByIndex(tbl.rv, loc.index)
if ferr != nil {
return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback
@@ -1118,15 +1291,14 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
return nil, errTargetFallback
}
if existing := tbl.rv.MapIndex(reflect.ValueOf(seg)); existing.IsValid() && !existing.IsNil() {
ev := existing.Elem()
if ev.Kind() != reflect.Map {
if existing.Kind() != reflect.Map {
return nil, tp.errf("key %q is not a table", seg)
}
return &targetTable{rv: ev, path: segAbs}, nil
return &targetTable{rv: existing, path: slices.Clone(segAbs)}, nil
}
next := reflect.MakeMap(elemT)
tbl.rv.SetMapIndex(reflect.ValueOf(seg), next)
return &targetTable{rv: next, path: segAbs}, nil
return &targetTable{rv: next, path: slices.Clone(segAbs)}, nil
}
resolved := seg
loc, ok := tbl.schema.byName[seg]
@@ -1135,17 +1307,21 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
loc, ok = tbl.schema.byName[resolved]
}
if !ok {
tbl.recordStrictUnknown(seg)
if opened, ok := tp.opened[pathKey(segAbs)]; ok {
tp.recordStrictUnknown(tbl, seg)
// A sink an array element created is entered by a [sub-table]
// header, through the last element, but never by a dotted key: the
// tree's descendKey rejects an array where tableAt follows it.
if opened, ok := tp.opened[pathKey(segAbs)]; ok && !opened.arrayElem {
return opened, nil
}
if tp.tableHas(tbl, seg) {
return nil, tp.errf("key %q is not a table", seg)
}
sink := &targetTable{sink: true, path: segAbs}
sink := &targetTable{sink: true, path: slices.Clone(segAbs)}
tp.opened[pathKey(segAbs)] = sink
return sink, nil
}
tp.markResolved(tbl, resolved)
fv, ferr := fieldByIndex(tbl.rv, loc.index)
if ferr != nil {
return nil, errTargetFallback
@@ -1167,7 +1343,7 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
}
return nil, errTargetFallback
}
return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: segAbs, strict: tbl.strict}, nil
return &targetTable{rv: fv, schema: schemaRef(fv.Type()), path: slices.Clone(segAbs), strict: tbl.strict}, nil
case reflect.Map:
if fv.Type().Key().Kind() != reflect.String {
return nil, errTargetFallback
@@ -1175,7 +1351,7 @@ func (tp *targetParser) dottedEnter(tbl *targetTable, seg string, segAbs []strin
if fv.IsNil() {
fv.Set(reflect.MakeMap(fv.Type()))
}
return &targetTable{rv: fv, path: segAbs}, nil
return &targetTable{rv: fv, path: slices.Clone(segAbs)}, nil
}
if tp.tableHas(tbl, seg) {
return nil, tp.errf("key %q is not a table", seg)
@@ -1193,7 +1369,9 @@ func (tp *targetParser) leafInTable(dest *targetTable, key string) (dst reflect.
if dest.rv.Kind() != reflect.Map {
return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback
}
if dest.rv.MapIndex(reflect.ValueOf(key)).IsValid() {
if tp.tableHas(dest, key) {
// The duplicate check reads the keys the document defined, not
// the destination map's own contents, which are the caller's.
return reflect.Value{}, reflect.Value{}, leafTable, false, tp.errf("duplicate key %q", key)
}
elem := reflect.New(dest.rv.Type().Elem()).Elem()
@@ -1206,10 +1384,10 @@ func (tp *targetParser) leafInTable(dest *targetTable, key string) (dst reflect.
loc, ok = dest.schema.byName[resolved]
}
if !ok {
dest.recordStrictUnknown(key)
tp.recordStrictUnknown(dest, key)
return reflect.Value{}, reflect.Value{}, leafTable, false, nil
}
dest.markResolved(resolved)
tp.markResolved(dest, resolved)
fv, ferr := fieldByIndex(dest.rv, loc.index)
if ferr != nil {
return reflect.Value{}, reflect.Value{}, leafTable, false, errTargetFallback
+350
View File
@@ -4,9 +4,12 @@
package interpres
import (
"errors"
"maps"
"net"
"reflect"
"strings"
"sync/atomic"
"testing"
"time"
)
@@ -522,3 +525,350 @@ func TestTargetedMapTableShapes(t *testing.T) {
})
}
}
// TestTargetedNestedMapDescents pins the descents into a map of maps that
// meet entries the document built earlier: a dotted key twice through the
// same sub-table, a header into a dotted-built sub-table, and a typed array
// under a map key. Each shape once panicked on a reflect Elem of a map.
func TestTargetedNestedMapDescents(t *testing.T) {
t.Run("dotted key through one sub-table twice", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\nm.a.c = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["c"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("header under a dotted-built sub-table", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\n[m.a.deep]\nx = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["deep"].(map[string]any)["x"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("typed array under a map key", func(t *testing.T) {
var cfg struct {
M map[string][]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("[[m.arr]]\nx = 1\n\n[[m.arr]]\ny = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.M["arr"]) != 2 || cfg.M["arr"][1]["y"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
}
// TestTargetedPointerElementSlice pins that an array of tables over a slice
// of pointer elements fills the pointed-to structs.
func TestTargetedPointerElementSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items []*item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n\n[[items]]\nn = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 2 || cfg.Items[0] == nil || cfg.Items[1].N != 2 {
t.Errorf("items = %#v", cfg.Items)
}
}
// TestTargetedArrayScopeResets pins that a new element of an array of tables
// starts a fresh definition scope, the contract the changelog documents.
func TestTargetedArrayScopeResets(t *testing.T) {
doc := "[[a]]\nb.c = 1\n\n[[a]]\n\n[a.b]\nx = 1\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
if refErr != nil {
t.Fatalf("tree decode: %v", refErr)
}
if err := Unmarshal([]byte(doc), &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
}
// TestTargetedUnknownArrayElements pins that every element of an unknown
// array of tables is a fresh namespace, and a sub-table header reaches the
// last element the way the tree parser's does.
func TestTargetedUnknownArrayElements(t *testing.T) {
doc := "[[zz]]\nk = 1\n\n[[zz]]\nk = 2\n\n[zz.sub]\nx = 3\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
tgtErr := Unmarshal([]byte(doc), &tgt)
if (refErr == nil) != (tgtErr == nil) {
t.Fatalf("error presence disagrees: tree %v, targeted %v", refErr, tgtErr)
}
if refErr != nil {
return
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
// A dotted key may not enter the array: the tree's own rule.
var dotted targetCfg
dErr := Unmarshal([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
refDotted := treeDecodeInto([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
if (dErr == nil) != (refDotted == nil) {
t.Errorf("dotted into an array: targeted %v, tree %v", dErr, refDotted)
}
}
// TestTargetedFixedArrayUnderFill pins that a fixed-size array the document
// under-fills is the length mismatch the tree decode raises, with the
// field's path.
func TestTargetedFixedArrayUnderFill(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items [2]item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n"), &cfg)
if err == nil {
t.Fatal("unmarshal accepted an under-filled array")
}
want := `interpres: items: cannot assign 1 elements to [2]interpres.item`
if err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedPrefilledSliceReplaced pins that a prefilled slice is replaced
// by the document's elements on both paths, not appended to.
func TestTargetedPrefilledSliceReplaced(t *testing.T) {
type item struct {
N int `toml:"n"`
}
doc := []byte("[[items]]\nn = 1\n")
var ref struct {
Items []item `toml:"items"`
}
ref.Items = []item{{N: 9}}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
var tgt struct {
Items []item `toml:"items"`
}
tgt.Items = []item{{N: 9}}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if len(tgt.Items) != 1 || tgt.Items[0].N != 1 {
t.Errorf("items = %#v, want the prefilled element replaced", tgt.Items)
}
}
// TestTargetedHeaderOverValueArrayKeepsCase pins that a value array assigned
// under a differently cased key than the field's name still blocks the
// array-of-tables header over it, the tree parse error.
func TestTargetedHeaderOverValueArrayKeepsCase(t *testing.T) {
var cfg struct {
Arr []targetNested `toml:"arr"`
}
err := Unmarshal([]byte("Arr = [{x = 1}]\n[[Arr]]\nx = 2\n"), &cfg)
if err == nil || err.Error() != `interpres: line 2: key "Arr" is not an array of tables` {
t.Errorf("err = %v, want the parse error over the assigned field", err)
}
}
// TestTargetedDottedInlineFreezePath pins that an inline table assigned by a
// dotted key freezes the whole path the statement wrote: a later header
// under that path is the extension error, and a key outside it stays free.
func TestTargetedDottedInlineFreezePath(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("m.a.b = {x = 1}\nb.y = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
err = Unmarshal([]byte("m.a.b = {x = 1}\n[m.a.b]\ny = 2\n"), &cfg)
want := `interpres: line 2: cannot extend inline table "m.a.b"`
if err == nil || err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedStrictThroughDottedKeys pins that strict and required findings
// survive the transient tables a dotted descent builds.
func TestTargetedStrictThroughDottedKeys(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("tab.zz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the strict failure through the dotted key", err)
}
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the path through the dotted key", err)
}
}
// TestTargetedRequiredThroughDottedKeys pins that a required tag is honoured
// when the table is reached only through dotted keys.
func TestTargetedRequiredThroughDottedKeys(t *testing.T) {
type nested struct {
X int `toml:"x,required"`
Y int `toml:"y"`
}
var cfg struct {
Tab nested `toml:"tab"`
}
err := Unmarshal([]byte("tab.y = 1\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), `missing required key "x"`) {
t.Errorf("err = %v, want the missing required key through the dotted key", err)
}
}
// TestTargetedOrderedMapSliceFallsBack pins that a slice of OrderedMap
// elements takes the tree path, whose fill keeps the written order.
func TestTargetedOrderedMapSliceFallsBack(t *testing.T) {
var cfg struct {
Items []OrderedMap `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nk = \"v\"\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 1 || cfg.Items[0].Keys()[0] != "k" {
t.Errorf("items = %#v, want the element filled in written order", cfg.Items)
}
}
// hookMap is a named map type whose decode hook counts its calls.
type hookMap map[string]any
var hookMapCalls atomic.Int32
func (h *hookMap) UnmarshalTOML(data any) error {
hookMapCalls.Add(1)
m, _ := data.(map[string]any)
if *h == nil {
*h = hookMap{}
}
maps.Copy((*h), m)
return nil
}
// TestTargetedMapFieldHookGetsWholeTable pins that a named map field with a
// decode hook receives the whole parsed table, even in its header form.
func TestTargetedMapFieldHookGetsWholeTable(t *testing.T) {
type cfg struct {
M hookMap `toml:"m"`
}
var c cfg
hookMapCalls.Store(0)
err := Unmarshal([]byte("[m]\na = 1\nb = 2\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if hookMapCalls.Load() != 1 {
t.Errorf("hook calls = %d, want exactly one with the whole table", hookMapCalls.Load())
}
if c.M["a"] != int64(1) || c.M["b"] != int64(2) {
t.Errorf("m = %#v", c.M)
}
}
// errHook fails every decode with a fixed error and counts its calls.
type errHook struct{ calls *int }
func (e *errHook) UnmarshalTOML(any) error {
if e.calls != nil {
*e.calls++
}
return errors.New("boom")
}
// TestTargetedHookErrorRunsOnce pins that a failing hook's error is the
// tree path's own, wrapped with the key, and that the hook is not run a
// second time by a fallback.
func TestTargetedHookErrorRunsOnce(t *testing.T) {
calls := 0
cfg := struct {
F errHook `toml:"f"`
}{F: errHook{calls: &calls}}
err := Unmarshal([]byte("f = 1\n"), &cfg)
if err == nil || err.Error() != "interpres: f: unmarshal: boom" {
t.Errorf("err = %v, want the wrapped hook failure", err)
}
if calls != 1 {
t.Errorf("hook calls = %d, want one", calls)
}
}
// TestTargetedUnknownBeforeRequired pins the report order the tree decode
// produces: an unknown key wins over a missing required one.
func TestTargetedUnknownBeforeRequired(t *testing.T) {
type inner struct {
X int `toml:"x,required"`
}
var cfg struct {
Tab inner `toml:"tab"`
}
err := Unmarshal([]byte("[tab]\nzz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the unknown key reported before the required one", err)
}
}
// TestTargetedStrictPathStableAcrossHeaders pins that the path a strict
// finding wraps does not alias the parser's key buffer: the table that owns
// the unknown key keeps its name after a later header.
func TestTargetedStrictPathStableAcrossHeaders(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("[tab]\nzz = 1\n\n[lims]\nx = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the finding on tab, not the later header", err)
}
}
// TestTargetedPrefilledMapFieldMergesUnderHeader pins that a prefilled map
// field merges the document's header-form table into it on both paths, the
// rule the root map has always followed.
func TestTargetedPrefilledMapFieldMergesUnderHeader(t *testing.T) {
doc := []byte("[lims]\nnew = 3\n")
var ref, tgt targetCfg
ref.Lims = map[string]any{"keep": "yes"}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
tgt.Lims = map[string]any{"keep": "yes"}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if tgt.Lims["keep"] != "yes" || tgt.Lims["new"] != int64(3) {
t.Errorf("lims = %#v, want the merge", tgt.Lims)
}
}
// TestTargetedNumberTokenValidatesUTF8 pins that the token route the
// targeted parse takes reports invalid UTF-8 with the scanner's own message
// and position.
func TestTargetedNumberTokenValidatesUTF8(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("num = 12\xff\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), "invalid UTF-8 in value at byte offset 8") {
t.Errorf("err = %v, want the UTF-8 complaint on the invalid byte", err)
}
}