44 Commits
Author SHA1 Message Date
petrbalvin e3dda5e115 fix: take the release-check version from the recipe argument
Test / test (push) Successful in 1m46s
Release / gates (push) Successful in 1m36s
Release / release (push) Successful in 39s
2026-09-22 21:49:05 +02:00
petrbalvin e35b4bb90d fix: take the release-check version from the recipe argument
Test / test (push) Canceled after 44s
2026-09-22 21:48:00 +02:00
petrbalvin 75ade89f34 chore: prepare release v2.0.0
Test / test (push) Successful in 2m3s
2026-09-22 21:45:25 +02:00
petrbalvin 41786c5bc6 chore: drop the internal reference from the justfile header 2026-09-22 21:45:25 +02:00
petrbalvin 010a7b2a1e ci: pin the actions by version tag and cap the test timeout at 10m 2026-09-22 21:45:25 +02:00
petrbalvin eb6ac1ab9d ci: drop the schedule triggers, race and fuzz run on dispatch alone
Test / test (push) Successful in 1m54s
Assisted-by: GLM 5.3
2026-09-22 21:28:30 +02:00
petrbalvin b0f40739c7 chore: keep the coverage floor at the documented 80 percent
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin fba53440c7 ci: state where the nightly race and fuzz sweeps run
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 332cd01d44 chore: check the suite colour before comparing documented counts
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 2fa075de00 docs: align every document with the reviewed behaviour
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 4900367970 fix(cmd): long-form flags, honest counts and safer inference
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin b7f39435e1 fix(document): rebuild set nodes and write documents back round-trip
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 0ded34da3c fix(encode): pointer table arrays, whole-minute offsets and emission checks
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin b45f4d65da fix(decode): keep the targeted parse on the tree path's contract
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin f7e427ae3f fix(decode): allocate embedded pointer maps and settle case collisions
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin e677e34508 fix(api): one statement per value array, parseas options and bounded reads
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 2efdb2d059 fix(parse): reject the lenient grammar edges and name out-of-range date-times
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 8f2b26bd33 docs: align the remaining Decoder and Encoder mentions with the options API
Test / test (push) Successful in 1m41s
Assisted-by: GLM 5.3 Flash
2026-09-22 19:27:12 +02:00
petrbalvin 2e61ad0ba9 feat!: rework the public API to json/v2-style variadic options
Test / test (push) Successful in 1m49s
Assisted-by: GLM 5.3 Flash
2026-09-22 18:42:18 +02:00
petrbalvin 7ee155d1e9 test(decode): seed the targeted fuzz with regression documents
Test / test (push) Successful in 1m39s
Assisted-by: GLM 5.3 Flash
2026-09-22 16:09:00 +02:00
petrbalvin a7d0041259 perf(decode): parse struct destinations without the value tree
Test / test (push) Successful in 1m50s
Assisted-by: GLM 5.3 Flash
2026-09-22 15:59:03 +02:00
petrbalvin ee32490452 docs: write the 2.0.0 migration guide into the changelog
Test / test (push) Successful in 1m36s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:48:23 +02:00
petrbalvin 80b2bc6e0f perf(encode): write scalars and scalar arrays without boxing
Test / test (push) Successful in 1m34s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:46:26 +02:00
petrbalvin ebaca18093 docs: note how the module resolves and verify it in release-check
Test / test (push) Successful in 1m40s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:31:37 +02:00
petrbalvin 3406955654 docs: add godoc examples, extend the basic example and map encoding/json
Test / test (push) Successful in 1m47s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:29:42 +02:00
petrbalvin f6a96379e6 ci: schedule nightly race and fuzz, pin actions, add release-check and docs drift
Test / test (push) Successful in 1m37s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:26:48 +02:00
petrbalvin 3ffae35a20 feat: add the Document edit pipeline with comment-preserving write
Test / test (push) Successful in 1m35s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:22:46 +02:00
petrbalvin a7a942a8e1 feat: add Statements, the top-level statement iterator
Test / test (push) Successful in 1m36s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:12:04 +02:00
petrbalvin d92bb56853 refactor: rename the encoder layout options
Test / test (push) Successful in 1m35s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:09:00 +02:00
petrbalvin bef1d3fbd9 test: add FuzzMarshal, golden messages, synctest cancellation and cross smoke
Test / test (push) Successful in 1m31s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:03:27 +02:00
petrbalvin 3e741e7790 feat(cmd): add version, plain json, struct inference and schema modes
Test / test (push) Successful in 1m35s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:55:02 +02:00
petrbalvin d18935ebc2 feat: add field comments, local time zone decoding and in-value cancellation
Test / test (push) Successful in 1m32s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:44:23 +02:00
petrbalvin 71bd82a7a5 feat: add ParseAs and NewSchema generics
Test / test (push) Successful in 1m38s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:36:10 +02:00
petrbalvin b02471c09a refactor: unify the error paths in one Path type
Test / test (push) Canceled after 1m31s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:34:48 +02:00
petrbalvin a8a2fcf8c3 feat: add the inline tag and json-style omitempty
Test / test (push) Successful in 1m30s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:28:52 +02:00
petrbalvin 13ac6dd521 test: pin the embedded map and map merge rules
Test / test (push) Successful in 1m34s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:23:41 +02:00
petrbalvin 10391a090f feat: bound the encoder walk and add UnmarshalWithOptions
Test / test (push) Successful in 1m34s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:21:46 +02:00
petrbalvin eaa69dc6f6 feat: add OrderedMap, the table that keeps its key order
Test / test (push) Successful in 1m30s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:15:17 +02:00
petrbalvin aefff80a28 style: gofmt the decode tests
Test / test (push) Successful in 1m29s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:04:49 +02:00
petrbalvin 0ba145ba0c feat: add the required tag option and UnmarshalerContext
Test / test (push) Canceled after 39s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:04:16 +02:00
petrbalvin 2e5dfc54c9 feat: decode arrays into [N]T and add MarshalAppend
Test / test (push) Successful in 2m31s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:58:23 +02:00
petrbalvin 10d49fbe60 feat: carry the byte offset and column in SyntaxError
Test / test (push) Canceled after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:55:58 +02:00
petrbalvin 3cc168f39a feat: add ParseFile and Valid
Test / test (push) Successful in 2m33s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:51:58 +02:00
petrbalvin ce0c1ebd9d feat: add Decoder.UseNumber and the Number type
Test / test (push) Successful in 1m52s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:49:39 +02:00
49 changed files with 9497 additions and 695 deletions
+36
View File
@@ -0,0 +1,36 @@
# Fuzz smoke, Go. Dispatched by hand when a change asks for it.
#
# Fuzzing is exploration, so it never belongs to the push pipeline; a 30 second
# smoke per target on a hand dispatch checks a change without holding the
# shared box. The targets run the seeds and whatever the corpus has gathered; a
# failure leaves its crashing input in testdata/fuzz, which the ordinary suite
# then reproduces on every push.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Fuzz
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
fuzz:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Fuzz the parser
run: go test -run '^$' -fuzz FuzzParse -fuzztime=30s -timeout 10m .
- name: Fuzz the encoder
run: go test -run '^$' -fuzz FuzzMarshal -fuzztime=30s -timeout 10m .
+6 -4
View File
@@ -1,8 +1,10 @@
# Race, Go. Dispatched by hand, and run as part of the release gates.
# Race, Go. Dispatched by hand.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is an explicit decision rather than a routine.
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once
# per task; here it is an explicit decision rather than a routine, a hand dispatch
# when a change asks for one. Development carries its race gate on every push through
# that local gate.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
@@ -32,4 +34,4 @@ jobs:
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 30m ./...
run: go test -race -count=1 -timeout 10m ./...
+3 -3
View File
@@ -4,8 +4,8 @@
# carries the CHANGELOG section as its body and nothing else. The gates still run first,
# in their own job and once, minus the race detector: race never runs on a push path or a
# tag, and the local gate raced this tree before the tag was cut. The write permission
# sits on the release job alone, and the version contract these steps implement is in the
# `release` skill.
# sits on the release job alone, and the version the binary reports is the one the
# toolchain records from the tag, with nothing injected.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
@@ -77,7 +77,7 @@ jobs:
- name: Tests
# Keep the pattern equal to `packages` in the project's justfile.
run: go test -count=1 -timeout 30m -coverprofile=coverage.out ./...
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./...
- name: Coverage floor
run: |
+4 -3
View File
@@ -1,8 +1,9 @@
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so race runs once inside
# the release pipeline instead. The box is one core and 2 GB beside Gitea, so
# runner box cannot afford the race detector on every push. Race has its own
# pipeline, dispatched by hand, and the local `just gates` runs it once per
# task. The box is one core and 2 GB beside Gitea, so
# parallelism is bounded on purpose and everything runs in one job. Extra jobs would
# duplicate the checkout, the Go setup and the dependency download three times without
# buying any parallelism.
@@ -79,7 +80,7 @@ jobs:
- name: Tests
# Scope the pattern to the packages that hold the logic when a thin cmd/ drags the
# total under the floor, and keep it equal to `packages` in the project's justfile.
run: go test -count=1 -timeout 30m -coverprofile=coverage.out ./...
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./...
- name: Coverage floor
run: |
+1 -1
View File
@@ -1,7 +1,7 @@
.idea/
.zcode/
# Build artifacts
# Build artefacts
bin/
*.test
*.out
+221 -8
View File
@@ -9,6 +9,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
-
## [2.0.0] - 2026-09-22
### Added
- `encoding.TextMarshaler` and `encoding.TextUnmarshaler` are honoured by
default, with no option to switch them off. A type that implements them is
encoded as a TOML string and decoded from one: `net.IP` becomes
@@ -20,11 +26,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- `time.Duration` is encoded in its canonical Go form as a TOML string,
`1h30m0s`, because TOML has no duration type; the decoder reads that string
back and still accepts a bare integer as the nanosecond count.
- `interpres-decode -encode`, the adapter's other direction: it reads the
- `interpres-decode --encode`, the adapter's other direction: it reads the
toml-test tagged JSON from stdin and writes the TOML document it describes.
The compliance suite now runs the encoder as well as the decoder, 214
encoder cases against the tagged JSON of the valid corpus.
- `Encoder.InlineTables(threshold)`: a sub-table whose single-line rendering is
- `InlineTables(threshold)`: a sub-table whose single-line rendering is
at most `threshold` bytes is written as an inline table instead of a header
section, which shortens a document of small tables. An array of tables keeps
its header form, because its inline form would re-parse as a value array.
@@ -32,8 +38,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
values together with the key order, whether a table was written as an inline
table or under a header, and the comments, with `Keys`, `Entries`, `Get`,
`Comments` and `SetComments` to read and write them. `ParseMap` returns the
plain `map[string]any` tree, the shape `Parse` used to give. `Marshal` does
not accept a `Document`; it writes values, so `doc.Map()` is the way through.
plain `map[string]any` tree, the shape `Parse` used to give.
- The edit pipeline on a `Document`: typed getters on `Document` and `Table`
(`GetString`, `GetInt`, `GetFloat`, `GetBool`, `GetArray`, `GetTable`),
`Set` and `Delete` that keep the surviving keys' positions and comments,
`UnmarshalDocument`, which decodes the document into a typed destination
without parsing again, and `Marshal` of a `Document`, which writes the keys
in written order, the comments above the lines and headers they belonged
to, and the inline tables inline.
- `OffsetDateTime`, the Go type of the offset date-time kind, so that all four
TOML date-time kinds have one of their own. `Parse` and `Unmarshal` hand it
back where they produced a bare `time.Time` before, and `Marshal` accepts it.
@@ -41,14 +53,118 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
the plain type takes an offset date-time as it always did; code that asserts
the tree's type, and `UnmarshalTOML` implementations that expect a
`time.Time`, need the new type.
- `Decoder.MaxDepth(depth)` and `Decoder.MaxInputSize(size)` bound the parse a
`Decode` performs, and every parse carries a nesting limit in any case
- `MaxNestingDepth(depth)` and `MaxInputSize(size)` options bound the parse an
`Unmarshal` performs, and every parse carries a nesting limit in any case
(10000 levels, which no hand-written document approaches): a document that
nests arrays or inline tables deeper used to run the stack out and is now
rejected with a `SyntaxError` naming the limit.
- `Statements(r)`, an iterator over the top-level statements of the document
the reader carries, in written order: key/value statements, a `[table]`
header as one statement with its node, an `[[array of tables]]` as one
statement per element. A caller that breaks after the statement it wanted
reads no further ones. `examples/statements` shows the walk.
- `ParseAs[T](data, opts...)`, the generic one-line decode, and `NewSchema[T]()`,
which precompiles the struct schema and the interface flags for a hot path
before the first document arrives.
- `EmitFieldComments(true)` prints the comment a field's `toml` tag
carries in a `comment=` option above the field's line or header, the
comments a round trip through the Go type would otherwise drop. Go doc
comments are not visible to reflection, so the tag is the channel that
carries the text.
- `LocalTimeLocation(loc)` lets a local date-time fill a plain
`time.Time` destination in the location given, relabelled rather than
shifted: `07:32` in the document is `07:32` in the zone. Without the
option the wrapper types remain the only destinations a local kind fills.
- The parse checks its context inside a value as well as between statements:
an array, an inline table and a multi-line string check every 64 elements
or lines, so one huge value cannot hold the parse past its cancellation.
- `OrderedMap`, the string-keyed table that remembers the order its keys were
set in: decoding into one fills it in the order the document wrote the
keys, and `Marshal` writes one back in that order, where a map carries no
order on decode and sorts on encode. It works as a decode target on its
own, in a struct field, and as the element of an array of tables; its
values are untyped, so a nested table stays a `map[string]any`.
- `Unmarshal(data, v, opts...)` and the other entries take variadic options,
the shape encoding/json/v2 uses: `RejectUnknownFields`,
`NumbersAsLiterals`, `MaxNestingDepth`, `MaxInputSize`,
`LocalTimeLocation`. `MarshalWrite(w, v, opts...)` and
`UnmarshalRead(r, v, opts...)` are the streaming forms.
- `Marshal` carries a nesting limit of 10000 levels, the parser's own figure:
cyclic data, which used to run the stack out, is now rejected with an error
that names the limit and the path it was met at.
- The `toml` tag gained the `required` option: a field tagged
`toml:"host,required"` makes the decode fail with
`missing required key "host"` when the document carries no key that
resolves to it. The option shapes decoding only, and the encoder ignores
it.
- `UnmarshalerContext`, the custom-decode interface that hands the decode's
context to the method, `UnmarshalTOMLContext(ctx, data)`. It wins over
`UnmarshalTOML` when a type implements both, so a long custom decode can
abort on cancellation; a non-cancellable entry point hands in
`context.Background`, never nil.
- A TOML array decodes into a Go fixed-size array, `[N]T`, where only a slice
was accepted before; the encoder could already encode one. A length mismatch
is an error wrapped with the key path.
- `MarshalAppend(buf, v)` appends the TOML encoding of v to buf and returns
the extended buffer, the shape `json.MarshalAppend` has.
- `interpres-decode --version` prints the binary's version, the module version
the toolchain recorded, so a release-built binary names its own tag.
- `interpres-decode --json` prints plain indented JSON instead of the tagged
form, the shape for people and diffs, with the date-time wrappers in their
TOML form.
- `interpres-decode --validate` walks a named directory for `.toml` files and
closes the sweep with a summary naming how many documents were checked and
how many were invalid; single files stay quiet on success as before.
- `interpres-decode --struct` infers a Go struct definition from a document:
one field per key in written order, nested tables as nested struct types,
an array of tables as a slice. The printed type compiles and decodes the
document it came from.
- `interpres-decode --schema TYPE file.go` writes a TOML template for the
named struct type of a Go source, the `comment=` tag option printed as a
comment and the `default=` option as the value. It is the inverse of
`--struct`.
- `ParseFile(path)` reads the file and parses it into a `Document`, with the
file name at the front of every error it returns, read failure and parse
failure alike. `Valid(data)` reports whether a document parses, nil on
success and the parse error on failure, the library call the `--validate`
mode of interpres-decode is built on.
- `SyntaxError` carries the byte `Offset` the scan stopped at and the 1-based
`Column` on the line, beside the line it always had, and `SourceLine(src)`
renders that line with a caret under the position, for messages shown under
the input. An input that is not valid UTF-8 names the offset of the first
invalid byte in its message. The new fields are additive: a `SyntaxError`
built from a line and a message alone is unchanged.
- `NumbersAsLiterals(true)` decodes the integers and floats of the document into
`Number`, which carries the literal the document wrote, so `0x1f`, `1_000`,
`+1.0` and `inf` survive a round trip with their spelling instead of the
normalised `31`, `1000` and `1.0`. Typed destinations take the evaluated
value as before, a `Number` field takes the literal, and `Marshal` writes a
`Number` back as its bare literal, rejecting one that is not a valid TOML
number.
### Changed
- The stateful `Decoder` and `Encoder` of 1.x are replaced by variadic
options on the entries, the shape encoding/json/v2 uses: `Layout(kind)`
with `LayoutKindGrouped` or `LayoutKindDeclaration`, `OmitEmptyArrays`,
`LiteralMultiline(threshold)`, `InlineTables(threshold)`,
`EmitFieldComments`, `RejectUnknownFields`, `NumbersAsLiterals`,
`MaxNestingDepth`, `MaxInputSize`, `LocalTimeLocation`.
- `DecodeError` and `EncodeError` carry one `Path` type, a list of segments
(`"items"`, `"[0]"`, `"weight"`) with a `String()` rendering the TOML
notation, `items[0].weight`. The decode error used to hold a bare
`[]string`, the encode error a plain string. Both messages render the same
way now, `interpres: items[0].weight: ...`, with one `interpres:` prefix
where the composition used to double it.
- `omitempty` follows the encoding/json semantics: the field is skipped when
it holds an empty string, a zero number, `false`, a nil pointer or
interface, or a nil or empty slice, array or map. In 1.x the option covered
only the collections.
- The `toml` tag gained the `inline` option: a struct or map field tagged
`toml:"retry,inline"` emits as `retry = {…}` instead of a header section,
whatever its size, a named embedded struct included. Forcing it on an array
of tables is an error, because the inline form would re-parse as a value
array and change the value's Go type.
- The output takes the TOML 1.1 form. A date-time writes its seconds only when
the value carries them and drops the trailing zeros of a fractional second,
so `07:32:00` is written `07:32` and half a second as `00.5`. Both are the
@@ -81,6 +197,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
**Performance**
- Struct destinations decode directly: for a type the direct skeleton can
model, the parser resolves tables and keys against the struct schema while
the document scans and no intermediate value tree is kept. The strict
decode of the representative document drops from 168 to 160 allocations
per call against the tree path in the same process, and the 2000-element
document reaches allocation parity; every document the skeleton cannot
model falls back to the tree path and its exact error contracts. A
differential fuzz target decodes every generated document both ways.
- Marshal writes plain scalars and typed scalar arrays straight from their
reflect cells instead of boxing them into interface values first, and skips
the per-element resolution for arrays that can never take the `[[header]]`
form. The representative document now costs 130 allocations per call
instead of 141, the long array-of-tables document 55 915 instead of
63 660, with byte-identical output.
- Parsing is faster than in 1.1.0 while carrying the new document layer:
the suite's representative document decodes at about 79 MB/s with 104
allocations per call, and the long array-of-tables document at about
@@ -103,6 +233,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Fixed
- An offset date-time written with the `+00:00` offset kept the anonymous
location `time.Parse` invents for it, so a round trip through the tree and
`Marshal`, which writes a zero offset as `Z`, changed the value's
reflection-visible location. The zero offset normalises to `time.UTC` at
parse, and the tree is stable across the round trip.
- Decoding into a defined type whose underlying kind is string or bool, such
as `type Name string`, panicked instead of storing the value, because a
value of the predeclared type is not assignable to a defined type and the
@@ -110,6 +245,84 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- A top-level value the encoder could not normalise reported its path with
a leading dot, `interpres: .port: ...`; the message now reads
`interpres: port: ...`, the shape `EncodeError.Path` already used.
- A token shaped like a date-time with a component out of range, such as an
hour of 24 or a February the 30th, fell through to the number decoder and
failed with the number complaint `invalid character "-" in number`; it now
fails as the date-time it visibly is, `invalid date-time "..."`.
- A `time.Time` or `OffsetDateTime` whose zone offset is not a whole number
of minutes wrote only the minutes, silently shifting the instant by the
seconds dropped; the encoder now refuses such an offset, which TOML has no
form for, instead of corrupting the value.
- An empty array of tables over pointer elements, `[]*T{}`, emitted as
`key = []` while its value form was omitted; it is omitted too now, the
rule TOML forces, because an empty `[[a]]` has no valid form.
- Two lenient grammar edges are closed: a sign in a `\u` or `\U` escape,
which is not a hex digit, is rejected instead of evaluating, and a bare
carriage return right after a multi-line string's opening delimiter is
the bare-CR error instead of a newline trimmed silently.
### Migration from 1.x
**The module path.** 2.0 lives at `sourcedock.dev/petrbalvin/interpres/v2`,
the suffix the Go toolchain requires of every major version 2 module. Change
every import and `go get` line:
```sh
go get sourcedock.dev/petrbalvin/interpres/v2
```
**TOML 1.1 only.** The acceptance contract is the TOML 1.1 corpus, and the
promise that every 1.0 document parses exactly as before is withdrawn.
Documents whose verdict changes are the ones 1.1 relaxed: `\e` and `\xHH`
escapes, times without seconds, multi-line inline tables with comments and a
trailing comma. Nothing that parsed in 1.x stops parsing, because the 1.1
grammar contains the 1.0 one.
**The output takes the 1.1 form.** A date-time writes seconds only when the
value carries them, a fraction drops its trailing zeros, and a long inline
table breaks across lines. A document written from the same value can come
out shorter; it re-parses to the same value.
**Text methods on by default.** A type implementing
`encoding.TextMarshaler` or `encoding.TextUnmarshaler` now takes the text
path with no option to switch it off. A struct that implemented the
interface encodes as a string where it was a table before. `MarshalTOML` and
`UnmarshalTOML` still win.
**One Go type per date-time kind.** Offset date-times hand back
`OffsetDateTime`, not a bare `time.Time`. Code that type-asserts the tree or
expects `time.Time` inside `UnmarshalTOML` needs the new wrapper; a
destination field of type `time.Time` keeps working.
**The document carries what the map could not.** `Parse` returns a
`*Document` with the key order, the inline distinction and the comments;
`ParseMap` gives the plain `map[string]any` tree the old `Parse` returned.
The document is writable, and `Marshal` writes it back with its comments.
**Options instead of Decoder and Encoder.** The stateful types of 1.x are
gone; the entries take variadic options, the shape encoding/json/v2 uses.
`NewDecoder().DisallowUnknownFields().Decode(data, &cfg)` becomes
`Unmarshal(data, &cfg, RejectUnknownFields(true))`, and the encoder
methods become options: `Layout(LayoutKindDeclaration)` replaces
`GroupByKind(false)`, `LiteralMultiline` replaces `UseLiteralMultiline`.
**Tag options.** `omitempty` follows encoding/json: it now also drops empty
strings, zero numbers, `false`, nil pointers and nil interfaces. `required`
demands a key at decode. `inline` forces the inline table form at encode.
`comment=text` carries a comment `EmitFieldComments` prints.
**Errors.** `DecodeError.Path` is a `Path` (segments with a `String()`
renderer), `EncodeError.Path` the same type instead of a plain string, and
both messages render `interpres: server.ports[2]: ...` with one prefix.
`SyntaxError` gained `Offset`, `Column` and `SourceLine`. Decode errors into
a `Number`-carrying tree and the fixed-size array decode are new shapes a
match on the old messages would not see.
**Decoding shapes.** `map[string]any` values merge into a non-empty map
destination; untagged embedded maps beyond the first stay empty; numbers can
stay literals under `NumbersAsLiterals`; local date-times can decode into
`time.Time` under `LocalTimeLocation`. All three are opt-in or additive
except where noted above.
## [1.1.0] - 2026-09-18
@@ -121,7 +334,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
comments and trailing commas. The compliance suite runs in TOML 1.1 mode:
214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document
parses exactly as before.
- `interpres-decode -validate [file ...]`: a validate mode beside the
- `interpres-decode --validate [file ...]`: a validate mode beside the
toml-test adapter. It parses each named file, or stdin when none are named,
prints one line per invalid document to stderr, and exits 0 when all are
valid, 1 when one is not, and 2 on a usage or read failure. Install it with
@@ -256,7 +469,7 @@ uses only the standard library and passes the entire
nested structs, slices, and `map[string]T`.
- `toml:"name"` field tags, case-insensitive name fallback, and `toml:"-"` to
skip a field.
- `Decoder` with `DisallowUnknownFields` for strict decoding that rejects keys
- `RejectUnknownFields(true)` option for strict decoding that rejects keys
without a destination field, at every struct depth.
- `Unmarshaler` interface (`UnmarshalTOML(data any) error`) for types that take
full control of their decode.
+1
View File
@@ -116,6 +116,7 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
|---|---|---|
| Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite |
| Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs |
| Fuzz | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector |
The local equivalent is `just gates`, which is the same set plus the race
+27 -20
View File
@@ -13,20 +13,21 @@ the entire official [toml-test](https://github.com/toml-lang/toml-test) suite:
`\xHH` escapes; integers in the four radixes with `_` separators; floats with
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds
optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and
`Marshal` for structs and maps, mirroring `encoding/json`.
- **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that
- **Decoding and encoding**: `Unmarshal` and `Marshal` for structs and maps,
mirroring `encoding/json`; `Parse` and `ParseMap` for the document with its
key order and the plain untyped tree.
- **Strict decoding**: `RejectUnknownFields(true)` rejects keys that
match no destination field, at every struct depth.
- **Custom types**: `Marshaler` and `Unmarshaler` let a type control its own
TOML representation in both directions, and `encoding.TextMarshaler` and
`TextUnmarshaler` are honoured by default, so `net.IP`, `time.Duration` and
user types with text methods need no configuration.
- **Cancellation**: every entry point has a `*Context` sibling that honours a
`context.Context`.
- **Cancellation**: the parse, decode and marshal entries have `*Context`
siblings that honour a `context.Context`, checked while the work runs.
- **Ordered documents**: `Parse` gives a `*Document` that keeps the key order,
tells an inline table from a header one, and carries the comments; `ParseMap`
gives the plain `map[string]any` tree.
- **Configurable emission**: `Encoder` options for declaration-order output,
- **Configurable emission**: `Marshal` options for declaration-order output,
omitting empty arrays, literal multiline strings, and inlining small
sub-tables.
@@ -40,6 +41,11 @@ go get sourcedock.dev/petrbalvin/interpres/v2
Requires Go 1.27.1 or newer. The module imports only the standard library.
The module is public and resolves through proxy.golang.org and sum.golang.org
like any other; no GOPROXY or GOPRIVATE setup is needed to fetch it. A machine
that sets `GOPRIVATE=sourcedock.dev` fetches directly from the forge instead,
which skips the proxy and the checksum database.
## Quick start
```sh
@@ -49,8 +55,7 @@ just example
```
`just example` runs the tour in `examples/basic`: it decodes an embedded
document into a struct, prints it, and re-encodes it under both `Encoder`
layouts.
document into a struct, prints it, and re-encodes it under both layouts.
## Usage
@@ -96,9 +101,7 @@ which is the layout that re-parses to the same tree.
### Strict decoding
```go
err := interpres.NewDecoder().
DisallowUnknownFields().
Decode(data, &cfg)
err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
```
A key with no matching field becomes an error instead of a silent drop.
@@ -127,16 +130,20 @@ func (ip *IP) UnmarshalTOML(data any) error {
The value `MarshalTOML` returns is encoded in place of the receiver;
`UnmarshalTOML` receives the parsed value verbatim.
### Encoder options
### Options
```go
out, err := interpres.NewEncoder().
GroupByKind(false). // preserve declaration order
OmitEmptyArrays(). // skip empty scalar arrays
UseLiteralMultiline(80). // long multi-line strings as literal blocks
Marshal(cfg)
out, err := interpres.Marshal(cfg,
interpres.Layout(interpres.LayoutKindDeclaration), // preserve declaration order
interpres.OmitEmptyArrays(true), // skip empty scalar arrays
interpres.LiteralMultiline(80), // long multi-line strings as literal blocks
)
```
The decode and encode calls take variadic options, the shape
encoding/json/v2 uses for its own. `UnmarshalRead(r, v, opts...)` and
`MarshalWrite(w, v, opts...)` are the streaming forms.
### Cancellation
```go
@@ -146,8 +153,7 @@ defer cancel()
out, err := interpres.MarshalContext(ctx, cfg)
```
`ParseContext`, `UnmarshalContext`, `(*Decoder).DecodeContext` and
`(*Encoder).MarshalContext` follow the same pattern.
`ParseContext`, `UnmarshalContext` and `MarshalContext` follow the same pattern.
The full rules for field matching, numeric conversion and emission live in
[docs/API.md](docs/API.md).
@@ -168,7 +174,8 @@ See [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the full workflow, and
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/API.md](docs/API.md): the API reference, decoding and encoding rules
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator,
also shipped as the manual page `man/interpres-decode.1`
## Licence
+89
View File
@@ -0,0 +1,89 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"errors"
"strings"
"testing"
)
// errReader fails every read with a fixed error.
type errReader struct{ err error }
func (r errReader) Read([]byte) (int, error) { return 0, r.err }
// TestUnmarshalRead covers the streaming entry: the happy path with options,
// a failing reader, and MaxInputSize bounding what a reader is drained into.
func TestUnmarshalRead(t *testing.T) {
var got struct {
Name string `toml:"name"`
N int `toml:"n"`
}
err := UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, RejectUnknownFields(true))
if err != nil {
t.Fatalf("UnmarshalRead: %v", err)
}
if got.Name != "x" {
t.Errorf("Name = %q", got.Name)
}
readErr := errors.New("boom")
if err := UnmarshalRead(errReader{readErr}, &got); !errors.Is(err, readErr) {
t.Errorf("err = %v, want the read error wrapped", err)
}
err = UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, MaxInputSize(4))
if err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit", err)
}
// The limit bounds the read itself: a reader that would supply far more
// than the limit is not drained into memory first.
big := strings.Repeat("x", 1<<20)
if err := UnmarshalRead(strings.NewReader(big), &got, MaxInputSize(16)); err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit before the read completes", err)
}
}
// TestParseAsWithOptions covers the generic shorthand carrying options.
func TestParseAsWithOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name"`
}
got, err := ParseAs[cfg]([]byte("name = \"x\"\nrogue = 1\n"), RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), "unknown field") {
t.Errorf("err = %v, want the strict failure", err)
}
// The statements before the failure stay written, the contract the
// targeted path documents and encoding/json follows.
if got.Name != "x" {
t.Errorf("Name = %q, want the statement before the failure kept", got.Name)
}
}
// TestStatementsValueArrays pins that a value array is one statement, a
// scalar array and an array of inline tables alike; only an array of tables
// yields per element.
func TestStatementsValueArrays(t *testing.T) {
src := strings.NewReader("port = [8080, 9090]\nmix = [{y = 1, x = 2}]\n[[items]]\nn = 1\n")
var got []Statement
for stmt, err := range Statements(src) {
if err != nil {
t.Fatal(err)
}
got = append(got, stmt)
}
if len(got) != 3 {
t.Fatalf("got %d statements, want 3", len(got))
}
if got[0].Index != -1 || got[0].Table != nil {
t.Errorf("port statement = %+v, want one plain key/value", got[0])
}
if got[1].Index != -1 || got[1].Table != nil {
t.Errorf("mix statement = %+v, want one plain key/value", got[1])
}
if got[2].Index != 0 || got[2].Table == nil {
t.Errorf("items statement = %+v, want the element with its node", got[2])
}
}
+39 -4
View File
@@ -4,6 +4,7 @@
package interpres
import (
"context"
"fmt"
"strings"
"testing"
@@ -108,11 +109,10 @@ func BenchmarkMarshal(b *testing.B) {
}
func BenchmarkStrictDecode(b *testing.B) {
dec := NewDecoder().DisallowUnknownFields()
b.ReportAllocs()
for b.Loop() {
var cfg benchConfig
if err := dec.Decode(benchDoc, &cfg); err != nil {
if err := Unmarshal(benchDoc, &cfg, RejectUnknownFields(true)); err != nil {
b.Fatal(err)
}
}
@@ -144,12 +144,11 @@ type benchLongDoc struct {
}
func BenchmarkStrictDecodeLong(b *testing.B) {
dec := NewDecoder().DisallowUnknownFields()
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
var doc benchLongDoc
if err := dec.Decode(longDoc, &doc); err != nil {
if err := Unmarshal(longDoc, &doc, RejectUnknownFields(true)); err != nil {
b.Fatal(err)
}
}
@@ -168,3 +167,39 @@ func BenchmarkMarshalLong(b *testing.B) {
}
}
}
// BenchmarkStrictDecodeTree measures the reference path the targeted decode
// is measured against: the full tree parse followed by the reflection walk.
// The pair runs in one process, so the A/B comparison shares the machine.
func BenchmarkStrictDecodeTree(b *testing.B) {
dec := newDecoder()
dec.disallowUnknown = true
b.ReportAllocs()
for b.Loop() {
tree, _, err := parseWithOptions(context.Background(), benchDoc, parseOptions{}, false)
if err != nil {
b.Fatal(err)
}
var cfg benchConfig
if err := dec.decode(tree, &cfg); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkStrictDecodeTreeLong(b *testing.B) {
dec := newDecoder()
dec.disallowUnknown = true
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
tree, _, err := parseWithOptions(context.Background(), longDoc, parseOptions{}, false)
if err != nil {
b.Fatal(err)
}
var doc benchLongDoc
if err := dec.decode(tree, &doc); err != nil {
b.Fatal(err)
}
}
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package main
import (
"fmt"
"io"
"strconv"
"strings"
"time"
"unicode"
"unicode/utf8"
"sourcedock.dev/petrbalvin/interpres/v2"
)
// inferStruct reads a TOML document and writes a Go struct definition shaped
// like the document: one field per key in written order, nested tables as
// nested struct types, an array of tables as a slice, and the field names
// invented from the keys. It is the onboarding aid: the printed type compiles
// and decodes the document it came from. The definition is built whole and
// written with a single call, so a failing standard output surfaces as one
// error instead of being dropped mid-print.
func inferStruct(data []byte, stdout io.Writer) error {
doc, err := interpres.Parse(data)
if err != nil {
return err
}
body := &strings.Builder{}
fmt.Fprintln(body, "// Generated by interpres-decode --struct; decode with")
fmt.Fprintln(body, "// sourcedock.dev/petrbalvin/interpres/v2.")
fmt.Fprintln(body, "type inferred struct {")
writeInferredFields(body, tableFields(doc.Root()), map[string]bool{})
fmt.Fprintln(body, "}")
_, err = io.WriteString(stdout, body.String())
return err
}
// inferredField is one document key with the entry it is inferred from.
type inferredField struct {
key string
entry *interpres.Entry
}
// tableFields lists a table's entries in written order.
func tableFields(t *interpres.Table) []inferredField {
out := make([]inferredField, 0, len(t.Keys()))
for _, key := range t.Keys() {
entry, _ := t.Get(key)
out = append(out, inferredField{key: key, entry: entry})
}
return out
}
// mergedTableFields merges the key sets of an array's elements in first-seen
// order. An array's type has to cover every element, and a key may appear
// only in a later one, so the first element alone does not decide the shape;
// each key is inferred from the first element that carries it.
func mergedTableFields(tables []*interpres.Table) []inferredField {
var out []inferredField
seen := map[string]bool{}
for _, t := range tables {
for _, f := range tableFields(t) {
if seen[f.key] {
continue
}
seen[f.key] = true
out = append(out, f)
}
}
return out
}
// writeInferredFields writes one field per entry, in the order given.
// invented tracks the field names already used at one level, so two keys
// that clean to the same name do not collide.
func writeInferredFields(w *strings.Builder, fields []inferredField, invented map[string]bool) {
for _, f := range fields {
writeInferredField(w, f, invented)
}
}
// writeInferredField writes one field for one entry: an array of tables as a
// slice of structs, a child table as a nested struct, and everything else as
// the scalar or slice the decoded value names.
func writeInferredField(w *strings.Builder, f inferredField, invented map[string]bool) {
name := goFieldName(f.key, invented)
// An array of tables carries a node per element; the nodes of a value
// array are nil wherever an element is not a table. Every node present
// is what tells the two apart: [1, {x=1}] stays a value array even
// though one of its elements is a table.
elements := f.entry.Elements()
allTables := len(elements) > 0
for _, el := range elements {
if el == nil {
allTables = false
break
}
}
if allTables {
fmt.Fprintf(w, "\t%s []struct {\n", name)
writeInferredFields(w, mergedTableFields(elements), map[string]bool{})
fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
}
if child := f.entry.Table(); child != nil {
fmt.Fprintf(w, "\t%s struct {\n", name)
writeInferredFields(w, tableFields(child), map[string]bool{})
fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
}
val := f.entry.Value()
if items, ok := val.([]any); ok {
fmt.Fprintf(w, "\t%s []%s %s\n", name, inferScalarType(items), structTag(f.key))
return
}
fmt.Fprintf(w, "\t%s %s %s\n", name, goTypeOf(val), structTag(f.key))
}
// structTag renders the toml tag of one key as a Go string literal. The raw
// backtick literal is the conventional shape, but a key carrying a backtick
// would end that literal early and the printed definition would not compile,
// so such tags are rendered with strconv.Quote instead.
func structTag(key string) string {
tag := `toml:"` + key + `"`
if !strings.ContainsAny(tag, "`\r") {
return "`" + tag + "`"
}
return strconv.Quote(tag)
}
// goTypeOf names the Go type the decoded value asks for.
func goTypeOf(val any) string {
switch val.(type) {
case string:
return "string"
case bool:
return "bool"
case int64:
return "int64"
case float64:
return "float64"
case interpres.OffsetDateTime:
return "interpres.OffsetDateTime"
case interpres.LocalDateTime:
return "interpres.LocalDateTime"
case interpres.LocalDate:
return "interpres.LocalDate"
case interpres.LocalTime:
return "interpres.LocalTime"
case time.Time:
return "time.Time"
case []any:
return "[]any"
case map[string]any:
return "map[string]any"
}
return "any"
}
// goFieldName cleans a document key into an exported Go identifier: the
// words the punctuation splits become capitalised runs, a leading digit
// gains a Field prefix, because an underscore would leave the field
// unexported and the decoder would skip it, and a collision with an earlier
// name gains a counter.
func goFieldName(key string, invented map[string]bool) string {
var b strings.Builder
nextUpper := true
for _, r := range key {
switch {
case unicode.IsLetter(r) || unicode.IsDigit(r):
if nextUpper {
r = unicode.ToUpper(r)
nextUpper = false
}
b.WriteRune(r)
default:
nextUpper = true
}
}
name := b.String()
if name == "" {
name = "Field"
}
// The first rune is decoded rather than taken as a byte, because a key
// may open with a digit beyond ASCII.
if first, _ := utf8.DecodeRuneInString(name); unicode.IsDigit(first) {
name = "Field" + name
}
for invented[name] {
name += "2"
}
invented[name] = true
return name
}
// inferScalarType names the Go element type of a scalar array when every
// element agrees, and any when they do not.
func inferScalarType(items []any) string {
seen := ""
for i, item := range items {
t := goTypeOf(item)
if i == 0 {
seen = t
} else if t != seen {
return "any"
}
}
if seen == "" {
return "any"
}
return seen
}
+209 -26
View File
@@ -4,16 +4,16 @@
// Command interpres-decode is the toml-test harness adapter and a TOML
// validator. Without flags it reads a TOML document from standard input and
// writes the toml-test "tagged JSON" representation to standard output. With
// -encode it is the reverse: it reads tagged JSON and writes the TOML document
// it describes. With -validate it checks the named documents, or standard
// input when none are named, and exits non-zero on the first invalid one:
// --encode it is the reverse: it reads tagged JSON and writes the TOML document
// it describes. With --validate it checks the named documents, or standard
// input when none are named, and exits non-zero when one is invalid:
//
// interpres-decode -validate config.toml
// interpres-decode -encode < case.json
// interpres-decode --validate config.toml
// interpres-decode --encode < case.json
//
// Run the official suite in both directions against the adapter with:
//
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode -encode'
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode --encode'
package main
import (
@@ -22,9 +22,13 @@ import (
"flag"
"fmt"
"io"
"io/fs"
"math"
"os"
"path/filepath"
"runtime/debug"
"strconv"
"strings"
"time"
"sourcedock.dev/petrbalvin/interpres/v2"
@@ -35,28 +39,74 @@ func main() {
}
// Run runs the command line and returns the process exit code: 0 success,
// 1 an invalid document, 2 a usage, reading, encoding, or
// 1 an invalid document, 2 a usage, reading, writing, encoding, or
// unsupported-value error.
func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError)
fs.SetOutput(stderr)
// The flag package's own diagnostics and default usage render flags
// with a single dash, while the command spells every flag in its
// two-dash long form, the form the manpage documents. Its output is
// therefore discarded and the usage below is the only one printed.
fs.SetOutput(io.Discard)
fs.Usage = func() {}
version := fs.Bool("version", false, "print the version and exit")
validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON")
encode := fs.Bool("encode", false, "read tagged JSON from stdin and write TOML instead")
plainJSON := fs.Bool("json", false, "with the default mode, print plain indented JSON instead of tagged JSON")
infer := fs.Bool("struct", false, "infer a Go struct definition from the document on stdin and print it")
schemaType := fs.String("schema", "", "write a TOML template for the named struct type; the source file follows as the first argument")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
usage(stdout)
return 0
}
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
usage(stderr)
return 2
}
if *validate && *encode {
fmt.Fprintln(stderr, "interpres-decode: -validate and -encode cannot be combined")
if *version {
if _, err := fmt.Fprintf(stdout, "interpres-decode %s\n", versionString()); err != nil {
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0
}
modes := 0
for _, on := range []*bool{validate, encode, infer} {
if *on {
modes++
}
}
if *schemaType != "" {
modes++
}
if modes > 1 {
fmt.Fprintln(stderr, "interpres-decode: --validate, --encode, --struct and --schema cannot be combined")
return 2
}
// --json shapes the decoding output only, so it is rejected with every
// mode uniformly instead of being silently ignored by some of them.
if *plainJSON && modes > 0 {
fmt.Fprintln(stderr, "interpres-decode: --json shapes the decoder output and cannot be combined with --encode, --struct, --validate or --schema")
return 2
}
if *schemaType != "" {
rest := fs.Args()
if len(rest) != 1 {
fmt.Fprintln(stderr, "interpres-decode: --schema needs the type name and exactly one Go source file")
return 2
}
if err := runSchema(*schemaType, rest[0], stdout); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
return 0
}
if *validate {
return validatePaths(fs.Args(), stdin, stderr)
}
if fs.NArg() > 0 {
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with -validate")
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with --validate")
return 2
}
if *encode {
@@ -64,37 +114,163 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
}
data, err := io.ReadAll(stdin)
if err != nil {
fmt.Fprintln(stderr, "read stdin:", err)
fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2
}
if *infer {
if err := inferStruct(data, stdout); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
// A document that fails to parse keeps the adapter's invalid
// exit; anything else, a failed write among them, is a tool
// failure.
if _, ok := errors.AsType[*interpres.SyntaxError](err); ok {
return 1
}
return 2
}
return 0
}
tree, err := interpres.ParseMap(data)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 1
}
if *plainJSON {
enc := json.NewEncoder(stdout)
enc.SetEscapeHTML(false)
enc.SetIndent("", " ")
if err := enc.Encode(plainJSONValue(tree)); err != nil {
fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2
}
return 0
}
tagged, err := tag(tree)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
enc := json.NewEncoder(stdout)
enc.SetEscapeHTML(false)
if err := enc.Encode(tagged); err != nil {
fmt.Fprintln(stderr, "encode:", err)
fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2
}
return 0
}
// usage prints the command line summary, with every flag in its two-dash
// long form: the flag package's default usage printer renders a single dash,
// and the manpage and docs/CLI.md spell the flags the way this text does.
func usage(w io.Writer) {
fmt.Fprint(w, `Usage: interpres-decode [flags]
Without a mode flag the command reads one TOML document from standard input
and writes the toml-test tagged JSON representation to standard output.
--encode read tagged JSON from standard input and write TOML
instead
--help print this usage
--json with the default mode, print plain indented JSON
instead of tagged JSON
--schema TYPE write a TOML template for the named struct type; the
Go source file follows as the first argument
--struct infer a Go struct definition from the document on
standard input and print it
--validate validate the documents instead of emitting tagged JSON
--version print the version and exit
`)
}
// versionString names the version the binary was built at: the module
// version the toolchain recorded, which is the tag when the release pipeline
// builds it, and (devel) for an ordinary build from a working tree.
func versionString() string {
if info, ok := debug.ReadBuildInfo(); ok {
if v := info.Main.Version; strings.HasPrefix(v, "v") {
return v
}
}
return "(devel)"
}
// plainJSONValue converts the parsed tree into the values encoding/json
// renders: the date-time wrappers print in their TOML form, which is the
// same text a reader of the document saw.
func plainJSONValue(v any) any {
switch x := v.(type) {
case map[string]any:
for k, val := range x {
x[k] = plainJSONValue(val)
}
return x
case []any:
for i, val := range x {
x[i] = plainJSONValue(val)
}
return x
case []map[string]any:
out := make([]any, len(x))
for i, val := range x {
out[i] = plainJSONValue(val)
}
return out
case time.Time:
return x.Format(time.RFC3339Nano)
case interpres.OffsetDateTime:
return x.String()
case interpres.LocalDateTime:
return x.String()
case interpres.LocalDate:
return x.String()
case interpres.LocalTime:
return x.String()
}
return v
}
// validatePaths parses every named file, or standard input when none are
// named, and reports each invalid document on stderr. It returns 0 when all
// documents parse, 1 when one does not, and 2 on a usage or read failure.
// named, and reports each invalid document on stderr. A named directory is
// walked for .toml files. It returns 0 when all documents parse, 1 when one
// does not, and 2 on a usage or read failure. A summary names the counts.
func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
if len(paths) == 0 {
paths = []string{"-"}
}
valid := true
var files []string
dirs := 0
for _, p := range paths {
if p == "-" {
files = append(files, "-")
continue
}
info, err := os.Stat(p)
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", p, err)
return 2
}
if !info.IsDir() {
files = append(files, p)
continue
}
dirs++
err = filepath.WalkDir(p, func(path string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if !d.IsDir() && strings.EqualFold(filepath.Ext(path), ".toml") {
files = append(files, path)
}
return nil
})
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: walk %s: %v\n", p, err)
return 2
}
}
checked := 0
invalid := 0
for _, p := range files {
name := p
var data []byte
var err error
@@ -108,12 +284,19 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
return 2
}
checked++
if _, err := interpres.ParseMap(data); err != nil {
fmt.Fprintf(stderr, "%s: %v\n", name, err)
valid = false
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
invalid++
}
}
if !valid {
// The single-document run stays quiet on success, the contract the
// compliance tooling relies on; a directory walk closes with the
// summary that makes the sweep readable.
if dirs > 0 {
fmt.Fprintf(stderr, "checked %d documents, %d invalid\n", checked, invalid)
}
if invalid > 0 {
return 1
}
return 0
@@ -124,17 +307,17 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
data, err := io.ReadAll(stdin)
if err != nil {
fmt.Fprintln(stderr, "read stdin:", err)
fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2
}
var desc any
if err := json.Unmarshal(data, &desc); err != nil {
fmt.Fprintln(stderr, "decode JSON:", err)
fmt.Fprintln(stderr, "interpres-decode: decode JSON:", err)
return 2
}
tree, err := untag(desc)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
doc, ok := tree.(map[string]any)
@@ -144,11 +327,11 @@ func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
}
out, err := interpres.Marshal(doc)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
if _, err := stdout.Write(out); err != nil {
fmt.Fprintln(stderr, "write stdout:", err)
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0
+567 -15
View File
@@ -7,7 +7,10 @@ import (
"bytes"
"encoding/json"
"errors"
"go/parser"
"go/token"
"os"
"path/filepath"
"reflect"
"strings"
"testing"
@@ -216,7 +219,7 @@ func TestTaggedHelper(t *testing.T) {
func TestValidateStdinAcceptsValidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"ok\"\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 0 {
if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
@@ -227,7 +230,7 @@ func TestValidateStdinAcceptsValidDocument(t *testing.T) {
func TestValidateStdinRejectsInvalidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"unterminated\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 1 {
if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 1 {
t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") {
@@ -249,10 +252,10 @@ func TestValidateFiles(t *testing.T) {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", good}, nil, &stdout, &stderr); code != 0 {
if code := Run([]string{"--validate", good}, nil, &stdout, &stderr); code != 0 {
t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String())
}
if code := Run([]string{"-validate", good, bad}, nil, &stdout, &stderr); code != 1 {
if code := Run([]string{"--validate", good, bad}, nil, &stdout, &stderr); code != 1 {
t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") {
@@ -262,7 +265,7 @@ func TestValidateFiles(t *testing.T) {
func TestValidateMissingFileReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
if code := Run([]string{"--validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
@@ -273,14 +276,14 @@ func TestAdapterModeRejectsPositionalArgument(t *testing.T) {
if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "-validate") {
t.Fatalf("stderr = %q, want it to point at -validate", stderr.String())
if !strings.Contains(stderr.String(), "--validate") {
t.Fatalf("stderr = %q, want it to point at --validate", stderr.String())
}
}
func TestUnknownFlagReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-nope"}, nil, &stdout, &stderr); code != 2 {
if code := Run([]string{"--nope"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
@@ -302,7 +305,7 @@ func TestRunEncoderScalars(t *testing.T) {
}
`
var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
@@ -333,7 +336,7 @@ func TestRunEncoderNested(t *testing.T) {
}
`
var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
@@ -354,7 +357,7 @@ func TestRunEncoderFloatTagDecides(t *testing.T) {
// tag decides the type; the output must stay a float.
var stdout, stderr bytes.Buffer
in := `{"whole": {"type": "float", "value": "1"}, "exp": {"type": "float", "value": "5e+22"}}`
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
@@ -379,7 +382,7 @@ func TestRunEncoderRejectsBadInput(t *testing.T) {
}
for _, c := range cases {
var stdout, stderr bytes.Buffer
code := Run([]string{"-encode"}, strings.NewReader(c.in), &stdout, &stderr)
code := Run([]string{"--encode"}, strings.NewReader(c.in), &stdout, &stderr)
if code != 2 {
t.Errorf("%s: Run returned %d, want 2; stderr = %q", c.name, code, stderr.String())
continue
@@ -395,7 +398,7 @@ func TestRunEncoderRejectsBadInput(t *testing.T) {
func TestRunEncoderFlagConflicts(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-encode", "-validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
if code := Run([]string{"--encode", "--validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for the two modes together", code)
}
if !strings.Contains(stderr.String(), "cannot be combined") {
@@ -404,7 +407,7 @@ func TestRunEncoderFlagConflicts(t *testing.T) {
stdout.Reset()
stderr.Reset()
if code := Run([]string{"-encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
if code := Run([]string{"--encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for an argument", code)
}
}
@@ -431,7 +434,7 @@ n = "a"
t.Fatalf("decode returned %d, stderr = %q", code, stderr.String())
}
var out bytes.Buffer
if code := Run([]string{"-encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
if code := Run([]string{"--encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
t.Fatalf("encode returned %d, stderr = %q", code, stderr.String())
}
want, err := interpres.ParseMap([]byte(doc))
@@ -446,3 +449,552 @@ n = "a"
t.Errorf("round trip changed the document:\noriginal: %#v\nencoded: %#v\noutput: %q", want, got, out.String())
}
}
func TestRunVersion(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--version"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.HasPrefix(out, "interpres-decode ") {
t.Errorf("output = %q, want the version prefix", out)
}
}
func TestRunPlainJSON(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nwhen = 1979-05-27T07:32:00-07:00\nitems = [1, 2]\n")
code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "\"host\": \"db\"") {
t.Errorf("output = %q, want plain JSON keys", out)
}
if strings.Contains(out, "\"type\"") {
t.Errorf("output = %q, want no tags", out)
}
if !strings.Contains(out, "\n \"") {
t.Errorf("output = %q, want indentation", out)
}
}
func TestValidateDirectorySummary(t *testing.T) {
dir := t.TempDir()
if err := os.WriteFile(filepath.Join(dir, "good.toml"), []byte("a = 1\n"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "bad.toml"), []byte("a =\n"), 0o644); err != nil {
t.Fatal(err)
}
sub := filepath.Join(dir, "nested")
if err := os.Mkdir(sub, 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(sub, "deep.toml"), []byte("b = true\n"), 0o644); err != nil {
t.Fatal(err)
}
// A second invalid document, so the summary's invalid count is
// exercised beyond the single failure the boolean tracked.
if err := os.WriteFile(filepath.Join(sub, "worse.toml"), []byte("c =\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--validate", dir}, strings.NewReader(""), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 for a directory with invalid files", code)
}
if !strings.Contains(stderr.String(), "checked 4 documents, 2 invalid") {
t.Errorf("stderr = %q, want the summary", stderr.String())
}
}
func TestInferStruct(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nport = 5432\ntags = [\"a\"]\n\n[server]\nname = \"edge\"\n\n[[items]]\nn = 1\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"type inferred struct {",
"Host string `toml:\"host\"`",
"Port int64 `toml:\"port\"`",
"Tags []string `toml:\"tags\"`",
"Server struct {",
"Items []struct {",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunSchemaTemplate(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := `package cfg
type Server struct {
Host string ` + "`toml:\"host,comment=The host to dial,default=example.org\"`" + `
Port int ` + "`toml:\"port,default=8080\"`" + `
}
type Config struct {
Name string ` + "`toml:\"name\"`" + `
Rate float64 ` + "`toml:\"rate,default=0.5\"`" + `
On bool ` + "`toml:\"on\"`" + `
Started time.Time ` + "`toml:\"started\"`" + `
Server Server ` + "`toml:\"server,comment=The server section\"`" + `
Items []Item ` + "`toml:\"items\"`" + `
}
type Item struct {
N int ` + "`toml:\"n\"`" + `
}
`
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"# The server section",
"[server]",
"# The host to dial",
"host = \"example.org\"",
"port = 8080",
"rate = 0.5",
"on = false",
"started = 1979-05-27T00:00:00Z",
"[[items]]",
"n = 0",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunPlainJSONShapes(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("when = 1979-05-27T07:32:00-07:00\nd = 1979-05-27\nt = 07:32:00\nwall = 1979-05-27T07:32:00\n" +
"items = [1, \"two\"]\n\n[[tables]]\nx = true\n")
code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"\"when\": \"1979-05-27T07:32-07:00\"",
"\"d\": \"1979-05-27\"",
"\"t\": \"07:32\"",
"\"wall\": \"1979-05-27T07:32\"",
"\"items\": [",
"\"x\": true",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestInferStructScalarShapes(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("f = 1.5\nb = true\nd = 1979-05-27\nldt = 1979-05-27T07:32:00\nlt = 07:32:00\nnums = [1, 2, 3]\nmixed = [1, \"a\"]\nempty = []\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"F float64",
"B bool",
"D interpres.LocalDate",
"Ldt interpres.LocalDateTime",
"Lt interpres.LocalTime",
"Nums []int64",
"Mixed []any",
"Empty []any",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunHelpPrintsLongFlags(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--help"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, flag := range []string{"--encode", "--help", "--json", "--schema", "--struct", "--validate", "--version"} {
if !strings.Contains(out, flag) {
t.Errorf("usage output missing %q:\n%s", flag, out)
}
}
// Every flag line of the list names its flag in the two-dash long form
// only, so no line opens with a single dash.
for line := range strings.SplitSeq(strings.TrimRight(out, "\n"), "\n") {
if after, ok := strings.CutPrefix(line, " -"); ok && !strings.HasPrefix(after, "-") {
t.Errorf("usage line %q lists a flag with one dash", line)
}
}
}
func TestGoFieldName(t *testing.T) {
cases := []struct{ key, want string }{
{"host", "Host"},
{"ab", "Ab"},
{"http-host", "HttpHost"},
{"3d", "Field3d"},
{"", "Field"},
}
for _, c := range cases {
if got := goFieldName(c.key, map[string]bool{}); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestGoFieldNameCollision(t *testing.T) {
// Two keys that clean to the same name must not collide; the counter
// keeps the fields apart and both stay exported.
invented := map[string]bool{}
cases := []struct{ key, want string }{
{"a-b", "AB"},
{"a b", "AB2"},
{"a_b", "AB22"},
}
for _, c := range cases {
if got := goFieldName(c.key, invented); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestInferStructDigitLeadingKey(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("3d = true\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Field3d bool") {
t.Errorf("output missing the exported Field3d field:\n%s", stdout.String())
}
}
func TestInferStructBacktickKey(t *testing.T) {
// A backtick in the key would end a raw string literal early, so the
// tag has to be rendered as an interpreted literal instead.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("\"a`b\" = 1\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "AB int64 \"toml:\\\"a`b\\\"\"") {
t.Errorf("output missing the quoted tag:\n%s", out)
}
// The printed definition has to compile; parsing it as Go is the
// syntax half of that proof.
if _, err := parser.ParseFile(token.NewFileSet(), "inferred.go", "package p\n\n"+out, 0); err != nil {
t.Errorf("the printed definition does not parse: %v\n%s", err, out)
}
}
func TestInferStructMergesArrayElements(t *testing.T) {
// The second element carries a key the first lacks, so the slice type
// has to be inferred from both.
var stdout, stderr bytes.Buffer
in := strings.NewReader("[[items]]\nn = 1\n\n[[items]]\nextra = \"late\"\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"Items []struct {",
"N int64 `toml:\"n\"`",
"Extra string `toml:\"extra\"`",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestInferStructMixedArrayStaysValueArray(t *testing.T) {
// One table element does not make the array an array of tables; a
// struct slice would not decode the scalar element.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("arr = [1, {x = 1}]\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "Arr []any") {
t.Errorf("output = %q, want a value array typed []any", out)
}
if strings.Contains(out, "[]struct") {
t.Errorf("output = %q, a mixed array must not become a struct slice", out)
}
}
func TestRunStructParseError(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a =\n"), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 {
t.Errorf("stdout should be empty on parse error, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "line 1") {
t.Errorf("stderr = %q, want the library's line number", stderr.String())
}
}
func TestRunStructWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a = 1\n"), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunSchemaNeedsTypeAndExactlyOneFile(t *testing.T) {
for _, args := range [][]string{{"--schema", "Config"}, {"--schema", "Config", "a.go", "b.go"}} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--schema") {
t.Errorf("Run(%v) stderr = %q, want it to name --schema", args, stderr.String())
}
}
}
func TestRunSchemaUnparsableSource(t *testing.T) {
src := filepath.Join(t.TempDir(), "broken.go")
if err := os.WriteFile(src, []byte("this is not Go\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "broken.go") {
t.Errorf("stderr = %q, want it to name the source file", stderr.String())
}
}
func TestRunSchemaUnknownType(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Missing", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), `no struct type "Missing"`) {
t.Errorf("stderr = %q, want it to name the missing type", stderr.String())
}
}
func TestRunSchemaRecursiveType(t *testing.T) {
// A self-referential struct has no finite template; the generator has
// to name the recursion instead of exhausting the stack.
src := filepath.Join(t.TempDir(), "node.go")
body := "package cfg\n\ntype Node struct {\n\tNext *Node `toml:\"next\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Node", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
for _, want := range []string{"recursive", "Node"} {
if !strings.Contains(stderr.String(), want) {
t.Errorf("stderr = %q, want it to mention %q", stderr.String(), want)
}
}
}
func TestRunSchemaMultiNameField(t *testing.T) {
// A field list may name several fields of one type; each name is one
// TOML key.
src := filepath.Join(t.TempDir(), "range.go")
if err := os.WriteFile(src, []byte("package cfg\n\ntype Range struct {\n\tMin, Max int\n}\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Range", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{"min = 0", "max = 0"} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunSchemaEmbeddedStructs(t *testing.T) {
// The library inlines only untagged embedded structs; a tagged one
// keeps its own section.
src := filepath.Join(t.TempDir(), "embed.go")
body := `package cfg
type Inner struct {
X int ` + "`toml:\"x\"`" + `
}
type Tagged struct {
Inner ` + "`toml:\"inner\"`" + `
Y int ` + "`toml:\"y\"`" + `
}
type Flat struct {
Inner
Z int ` + "`toml:\"z\"`" + `
}
`
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Tagged", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Tagged: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "y = 0") || !strings.Contains(out, "[inner]") {
t.Errorf("Tagged output = %q, want a y scalar and an [inner] section", out)
}
stdout.Reset()
stderr.Reset()
code = Run([]string{"--schema", "Flat", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Flat: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out = stdout.String()
if !strings.Contains(out, "x = 0") || !strings.Contains(out, "z = 0") {
t.Errorf("Flat output = %q, want x and z flattened as scalars", out)
}
if strings.Contains(out, "[inner]") {
t.Errorf("Flat output = %q, an untagged embedded struct must not become a section", out)
}
}
func TestRunVersionWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--version"}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunSchemaWriteFailure(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunEncodeWriteFailure(t *testing.T) {
var stderr bytes.Buffer
in := `{"a": {"type": "integer", "value": "1"}}`
code := Run([]string{"--encode"}, strings.NewReader(in), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunEmptyInput(t *testing.T) {
t.Run("default", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run(nil, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if strings.TrimSpace(stdout.String()) != "{}" {
t.Errorf("stdout = %q, want an empty table", stdout.String())
}
})
t.Run("encode", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2, empty input is not JSON", code)
}
})
t.Run("struct", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "type inferred struct {\n}") {
t.Errorf("stdout = %q, want an empty struct", stdout.String())
}
})
t.Run("validate", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--validate"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
t.Errorf("validate should be quiet, stdout %q stderr %q", stdout.String(), stderr.String())
}
})
}
func TestRunJSONFlagConflicts(t *testing.T) {
for _, args := range [][]string{
{"--encode", "--json"},
{"--struct", "--json"},
{"--validate", "--json"},
{"--json", "--schema", "Config", "config.go"},
} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--json") {
t.Errorf("Run(%v) stderr = %q, want it to explain the --json conflict", args, stderr.String())
}
}
}
+363
View File
@@ -0,0 +1,363 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package main
import (
"fmt"
"go/ast"
"go/parser"
"go/token"
"io"
"maps"
"path/filepath"
"reflect"
"slices"
"strconv"
"strings"
)
// runSchema writes a TOML template for the named struct type of a Go source
// file: one key per exported field, the comment a `comment=` tag option
// carries printed above it, and a `default=` option as the value, or the
// type's zero value where no default is given. Struct fields resolve into
// [sections], slices of them into [[array of tables]] blocks, and an
// untagged embedded struct flattens into its parent, the way the library
// decodes it.
func runSchema(typeName, sourcePath string, stdout io.Writer) error {
fset := token.NewFileSet()
file, err := parser.ParseFile(fset, sourcePath, nil, parser.ParseComments)
if err != nil {
return fmt.Errorf("%s: %w", filepath.Base(sourcePath), err)
}
types := declaredStructs(file)
st, ok := types[typeName]
if !ok {
return fmt.Errorf("no struct type %q in %s", typeName, filepath.Base(sourcePath))
}
body := &strings.Builder{}
if err := writeSchemaFields(body, st, types, "", nil); err != nil {
return err
}
_, err = io.WriteString(stdout, strings.TrimLeft(body.String(), "\n"))
return err
}
// declaredStructs collects the field lists of the file's top-level struct
// type declarations.
func declaredStructs(file *ast.File) map[string]*ast.StructType {
out := map[string]*ast.StructType{}
for _, decl := range file.Decls {
gd, ok := decl.(*ast.GenDecl)
if !ok {
continue
}
for _, spec := range gd.Specs {
ts, ok := spec.(*ast.TypeSpec)
if !ok {
continue
}
st, ok := ts.Type.(*ast.StructType)
if !ok {
continue
}
out[ts.Name.Name] = st
}
}
return out
}
// fieldMeta is what the generator reads off one struct field.
type fieldMeta struct {
key string
comment string
def string
typ ast.Expr
}
// writeSchemaFields writes the fields of one struct level: the scalar lines
// first, then the sections, so the template re-parses with every value under
// the header it belongs to. prefix is the dotted path the nested headers
// carry. path holds the struct types of the levels currently being written,
// so a type that reaches itself is reported as recursion instead of
// exhausting the stack.
func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]*ast.StructType, prefix string, path []*ast.StructType) error {
if slices.Contains(path, st) {
return recursionError(st, types)
}
path = append(path, st)
metas, err := metasOf(st, types)
if err != nil {
return err
}
for _, m := range metas {
if _, elemSt := elementStruct(m.typ, types); elemSt != nil {
continue
}
if isStructKind(m.typ, types) || isMapKind(m.typ) {
continue
}
writeComment(w, m.comment)
if _, ok := baseType(m.typ).(*ast.ArrayType); ok {
fmt.Fprintf(w, "%s = []\n", m.key)
continue
}
fmt.Fprintf(w, "%s = %s\n", m.key, scalarLiteral(m))
}
for _, m := range metas {
if !isStructKind(m.typ, types) && !isMapKind(m.typ) {
continue
}
writeComment(w, m.comment)
fmt.Fprintf(w, "[%s%s]\n", prefix, m.key)
if sub := structOf(m.typ, types); sub != nil {
if err := writeSchemaFields(w, sub, types, prefix+m.key+".", path); err != nil {
return err
}
}
fmt.Fprintln(w)
}
for _, m := range metas {
_, elemSt := elementStruct(m.typ, types)
if elemSt == nil {
continue
}
writeComment(w, m.comment)
fmt.Fprintf(w, "[[%s%s]]\n", prefix, m.key)
if err := writeSchemaFields(w, elemSt, types, "", path); err != nil {
return err
}
fmt.Fprintln(w)
}
return nil
}
// writeComment writes the comment lines above a binding.
func writeComment(w *strings.Builder, text string) {
if text == "" {
return
}
for line := range strings.SplitSeq(text, "\n") {
fmt.Fprintf(w, "# %s\n", line)
}
}
// metasOf flattens the exported fields of a struct. The key comes from the
// toml tag, or the lower-cased field name; a `-` key drops the field. An
// embedded struct without a tag name flattens into its parent, the way the
// library inlines it, while a tagged one keeps its own section.
func metasOf(st *ast.StructType, types map[string]*ast.StructType) ([]fieldMeta, error) {
return flattenMetas(st, types, nil)
}
// flattenMetas is metasOf with the chain of struct types currently being
// flattened, which stops a struct that embeds itself, directly or through
// another embedded type.
func flattenMetas(st *ast.StructType, types map[string]*ast.StructType, chain map[*ast.StructType]bool) ([]fieldMeta, error) {
if chain[st] {
return nil, recursionError(st, types)
}
// A copy per branch: the chain is the path being flattened now, not the
// set ever visited, so a type embedded in two siblings is not mistaken
// for recursion.
chain = maps.Clone(chain)
if chain == nil {
chain = map[*ast.StructType]bool{}
}
chain[st] = true
var out []fieldMeta
for _, field := range st.Fields.List {
tagText := ""
if field.Tag != nil {
tagText, _ = strconv.Unquote(field.Tag.Value)
}
toml := reflect.StructTag(tagText).Get("toml")
key, opts, _ := strings.Cut(toml, ",")
if len(field.Names) == 0 {
if key == "" {
// An untagged embedded struct flattens into its parent.
if ident, ok := baseType(field.Type).(*ast.Ident); ok {
if inner, ok := types[ident.Name]; ok {
metas, err := flattenMetas(inner, types, chain)
if err != nil {
return nil, err
}
out = append(out, metas...)
}
}
continue
}
if key == "-" {
continue
}
// A tagged embedded struct is a section of its own; the tag
// name is the only name it has.
out = append(out, fieldMeta{
key: key,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
continue
}
if key == "-" {
continue
}
// A field list may name several fields of one type, `Min, Max int`;
// each name is one TOML key.
for _, name := range field.Names {
if !ast.IsExported(name.Name) {
continue
}
fieldKey := key
if fieldKey == "" {
fieldKey = strings.ToLower(name.Name)
}
out = append(out, fieldMeta{
key: fieldKey,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
}
}
return out, nil
}
// recursionError names the struct type that reached itself. Such a type has
// no finite TOML template: every level would nest another copy of the same
// shape.
func recursionError(st *ast.StructType, types map[string]*ast.StructType) error {
return fmt.Errorf("recursive type %s: the struct contains itself, so it has no finite template", typeName(st, types))
}
// typeName names the declared struct type st refers to, and "anonymous
// struct" for a literal one that no declaration names.
func typeName(st *ast.StructType, types map[string]*ast.StructType) string {
for name, t := range types {
if t == st {
return name
}
}
return "anonymous struct"
}
// tagOption returns the text a `name=` option carries in the option part of
// a tag.
func tagOption(opts, name string) string {
for opts != "" {
var opt string
opt, opts, _ = strings.Cut(opts, ",")
if text, ok := strings.CutPrefix(opt, name); ok {
return text
}
}
return ""
}
// baseType unwraps pointers and parentheses.
func baseType(e ast.Expr) ast.Expr {
for {
switch x := e.(type) {
case *ast.StarExpr:
e = x.X
case *ast.ParenExpr:
e = x.X
default:
return e
}
}
}
// structOf returns the struct type an expression denotes when its
// declaration sits in the same file, or when it is an anonymous struct.
func structOf(e ast.Expr, types map[string]*ast.StructType) *ast.StructType {
if ident, ok := baseType(e).(*ast.Ident); ok {
return types[ident.Name]
}
if st, ok := baseType(e).(*ast.StructType); ok {
return st
}
return nil
}
// isStructKind reports whether the type is a struct the generator renders as
// a section.
func isStructKind(e ast.Expr, types map[string]*ast.StructType) bool {
return structOf(e, types) != nil
}
// isMapKind reports whether the type is a map, which renders as an empty
// section.
func isMapKind(e ast.Expr) bool {
_, ok := baseType(e).(*ast.MapType)
return ok
}
// elementStruct returns the struct type a slice's element denotes, for the
// [[array of tables]] blocks.
func elementStruct(e ast.Expr, types map[string]*ast.StructType) (ast.Expr, *ast.StructType) {
arr, ok := baseType(e).(*ast.ArrayType)
if !ok {
return nil, nil
}
return arr.Elt, structOf(arr.Elt, types)
}
// scalarLiteral renders the value line for a scalar field: the default=
// option when it is set, and the type's zero value otherwise.
func scalarLiteral(m fieldMeta) string {
kind := scalarKind(m.typ)
if m.def != "" {
if kind == "string" {
return strconv.Quote(m.def)
}
return m.def
}
switch kind {
case "int":
return "0"
case "float":
return "0.0"
case "bool":
return "false"
case "datetime":
return "1979-05-27T00:00:00Z"
}
return `""`
}
// scalarKind classifies a scalar type for the zero-value rendering.
func scalarKind(e ast.Expr) string {
switch t := baseType(e).(type) {
case *ast.Ident:
switch t.Name {
case "bool":
return "bool"
case "float32", "float64":
return "float"
case "int", "int8", "int16", "int32", "int64",
"uint", "uint8", "uint16", "uint32", "uint64", "uintptr", "byte", "rune":
return "int"
}
if t.Name != "string" {
// A named type in the file may be a scalar alias; the string
// zero value is the safe default for it and everything unknown.
return "unknown"
}
return "string"
case *ast.SelectorExpr:
if pkg, ok := t.X.(*ast.Ident); ok {
if pkg.Name == "time" && t.Sel.Name == "Time" {
return "datetime"
}
if pkg.Name == "interpres" {
switch t.Sel.Name {
case "OffsetDateTime", "LocalDateTime", "LocalDate", "LocalTime":
return "datetime"
}
}
}
}
return "unknown"
}
+38 -15
View File
@@ -79,7 +79,9 @@ func clockString(t time.Time) string {
// offsetString renders an offset date-time, the fourth TOML kind, in the same
// shape: no zero seconds, no trailing zeros in the fraction, and the offset
// written as "Z" when it is zero.
// written as "Z" when it is zero. A zone offset that is not a whole number of
// minutes loses its seconds to this rendering, which is why Marshal refuses
// such a value rather than writing it.
func offsetString(t time.Time) string {
buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
buf = appendClock(buf, t)
@@ -235,17 +237,21 @@ func normaliseDateTimeToken(tok string, kind dateTimeKind) string {
// parseDateTime classifies and parses a bare token as a TOML date-time value.
// It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or
// LocalTime) and whether the token was a date-time at all.
func parseDateTime(tok string) (any, bool) {
// LocalTime), whether the token was a date-time at all, and an error for a
// token whose shape is a date-time a component of which lies outside its
// range: an hour of 24, a day the month does not hold. Such a token is a
// broken date-time, not some other value, so the error names it instead of
// leaving it to the number decoder's complaint.
func parseDateTime(tok string) (any, bool, error) {
if tok == "" || tok[0] < '0' || tok[0] > '9' {
return nil, false
return nil, false, nil
}
if !strings.ContainsAny(tok, "-:") {
return nil, false
return nil, false, nil
}
kind, seconds := scanDateTimeShape(tok)
if kind == dateTimeNone {
return nil, false
return nil, false, nil
}
norm := normaliseDateTimeToken(tok, kind)
switch kind {
@@ -256,9 +262,15 @@ func parseDateTime(tok string) (any, bool) {
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return OffsetDateTime{t}, true
// A zero offset carries its own anonymous location from time.Parse,
// while the written form is "Z" either way; normalising to UTC keeps
// the tree identical across the round trip.
if _, off := t.Zone(); off == 0 {
t = t.In(time.UTC)
}
return OffsetDateTime{t}, true, nil
case dateTimeLocal:
layout := localClockLayout
if seconds {
@@ -266,15 +278,15 @@ func parseDateTime(tok string) (any, bool) {
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalDateTime{t}, true
return LocalDateTime{t}, true, nil
case dateTimeDate:
t, err := time.Parse(localDateOnlyLayout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalDate{t}, true
return LocalDate{t}, true, nil
case dateTimeClock:
layout := localTimeClockLayout
if seconds {
@@ -282,11 +294,22 @@ func parseDateTime(tok string) (any, bool) {
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalTime{t}, true
return LocalTime{t}, true, nil
}
return nil, false
return nil, false, nil
}
// wholeMinuteOffset reports an error when the zone offset carries seconds, a
// shape no TOML offset can hold: writing only the minutes would silently
// shift the instant on the way back, so the encoder refuses the value rather
// than corrupting it.
func wholeMinuteOffset(t time.Time) error {
if _, off := t.Zone(); off%60 != 0 {
return fmt.Errorf("interpres: date-time offset of %d seconds is not a whole number of minutes, which TOML cannot write", off)
}
return nil
}
// isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a
+232 -34
View File
@@ -4,8 +4,10 @@
package interpres
import (
"context"
"encoding"
"fmt"
"maps"
"reflect"
"slices"
"strings"
@@ -14,18 +16,38 @@ import (
"time"
)
// decoder maps a parsed TOML tree onto Go values via reflection.
// decoder maps a parsed TOML tree onto Go values via reflection. ctx is the
// context a cancellable entry point handed in, and reaches an
// UnmarshalerContext destination; entry points without one leave it nil.
// nodes is the document's node index, present only when a destination can
// reach an OrderedMap and the parse built the tree its key order is read
// from. loc is the zone a local date-time is carried in when it decodes into
// a time.Time destination; nil keeps the wrapper-only default.
type decoder struct {
disallowUnknown bool
ctx context.Context
nodes nodeIndex
loc *time.Location
}
func newDecoder() *decoder { return &decoder{} }
// ctxOrBackground returns the context the decode carries, and Background when
// none was given, so a custom decoder never receives a nil context.
func (d *decoder) ctxOrBackground() context.Context {
if d.ctx == nil {
return context.Background()
}
return d.ctx
}
var timeType = reflect.TypeFor[time.Time]()
var (
unmarshalerType = reflect.TypeFor[Unmarshaler]()
ctxUnmarshalerType = reflect.TypeFor[UnmarshalerContext]()
textUnmarshalerType = reflect.TypeFor[encoding.TextUnmarshaler]()
numberType = reflect.TypeFor[Number]()
)
// The per-type flags record which interface lookups a decode into that type
@@ -35,6 +57,8 @@ var (
const (
flagUnmarshaler uint8 = 1 << iota
flagAddrUnmarshaler
flagCtxUnmarshaler
flagAddrCtxUnmarshaler
flagTextUnmarshaler
flagAddrTextUnmarshaler
)
@@ -71,10 +95,16 @@ func typeFlags(t reflect.Type) uint8 {
if t.Implements(unmarshalerType) {
f |= flagUnmarshaler
}
if t.Implements(ctxUnmarshalerType) {
f |= flagCtxUnmarshaler
}
pt := reflect.PointerTo(t)
if pt.Implements(unmarshalerType) {
f |= flagAddrUnmarshaler
}
if pt.Implements(ctxUnmarshalerType) {
f |= flagAddrCtxUnmarshaler
}
// The date-time types are excluded from the text path: they carry
// time.Time's UnmarshalText through an embedded field while their only
// accepted form is a bare timestamp.
@@ -114,6 +144,24 @@ func unmarshalerOf(dst reflect.Value) (Unmarshaler, bool) {
return nil, false
}
// ctxUnmarshalerOf is the same resolution for UnmarshalerContext.
func ctxUnmarshalerOf(dst reflect.Value) (UnmarshalerContext, bool) {
if dst.Kind() == reflect.Interface {
u, ok := dst.Interface().(UnmarshalerContext)
return u, ok
}
f := typeFlags(dst.Type())
if f&flagCtxUnmarshaler != 0 {
u, ok := dst.Interface().(UnmarshalerContext)
return u, ok
}
if f&flagAddrCtxUnmarshaler != 0 && dst.CanAddr() {
u, ok := dst.Addr().Interface().(UnmarshalerContext)
return u, ok
}
return nil, false
}
func (d *decoder) decode(tree map[string]any, v any) error {
rv := reflect.ValueOf(v)
if rv.Kind() != reflect.Pointer || rv.IsNil() {
@@ -138,12 +186,18 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
return nil
}
// Types implementing Unmarshaler get the parsed data wholesale and are
// responsible for setting their own state. The decoder does not consult
// any return value; whatever the receiver stores is kept. The lookup
// covers both T and *T so a pointer-receiver UnmarshalTOML method is
// invoked on an addressable struct field.
// Types implementing UnmarshalerContext get the context beside the parsed
// data, and are responsible for setting their own state. They win over
// Unmarshaler, which wins over the text path. The lookups cover both T and
// *T so a pointer-receiver method is invoked on an addressable struct
// field.
if dst.CanInterface() {
if u, ok := ctxUnmarshalerOf(dst); ok {
if err := u.UnmarshalTOMLContext(d.ctxOrBackground(), data); err != nil {
return fmt.Errorf("unmarshal: %w", err)
}
return nil
}
u, ok := dst.Interface().(Unmarshaler)
if !ok && dst.CanAddr() {
u, ok = dst.Addr().Interface().(Unmarshaler)
@@ -181,6 +235,8 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
return setDuration(dst, v)
}
return setBasic(dst, reflect.ValueOf(v), "string")
case Number:
return setNumber(dst, v)
case bool:
return setBasic(dst, reflect.ValueOf(v), "bool")
case int64:
@@ -191,6 +247,24 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
return setOffsetDateTime(v, dst)
case time.Time:
return setDateTime(v, dst)
case LocalDateTime:
if dst.Type() == localDateTimeType {
dst.Set(reflect.ValueOf(v))
return nil
}
return d.setLocalTimeValue(v.Time, dst)
case LocalDate:
if dst.Type() == localDateType {
dst.Set(reflect.ValueOf(v))
return nil
}
return d.setLocalTimeValue(v.Time, dst)
case LocalTime:
if dst.Type() == localTimeType {
dst.Set(reflect.ValueOf(v))
return nil
}
return d.setLocalTimeValue(v.Time, dst)
default:
rv := reflect.ValueOf(data)
if rv.IsValid() && dst.Type() == rv.Type() {
@@ -224,6 +298,9 @@ func textUnmarshalerOf(dst reflect.Value) (encoding.TextUnmarshaler, bool) {
}
func (d *decoder) assignTable(tbl map[string]any, dst reflect.Value) error {
if dst.Type() == orderedMapType {
return d.fillOrderedMap(tbl, dst)
}
switch dst.Kind() {
case reflect.Struct:
return d.assignStruct(tbl, dst)
@@ -255,27 +332,42 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
return fmt.Errorf("interpres: unknown field %q for %s", unknown, dst.Type())
}
}
for key, val := range tbl {
// The keys that resolved to a field are remembered while the table walks,
// but only a struct that demands one pays for the set.
var seen map[string]bool
if len(schema.required) > 0 {
seen = make(map[string]bool, len(tbl))
}
for _, key := range d.tableKeys(tbl) {
val := tbl[key]
// A key that is already lowercase, which document keys usually are,
// hits the map directly; only a miss pays for the case fold.
resolved := key
field, ok := schema.byName[key]
if !ok {
field, ok = schema.byName[strings.ToLower(key)]
resolved = strings.ToLower(key)
field, ok = schema.byName[resolved]
}
if !ok {
if schema.embedMaps != nil {
// Leftover keys land in an untagged embedded map, the inverse
// of the encoder inlining that map's entries.
// of the encoder inlining that map's entries. The assign call
// rather than assignMap itself lets it allocate the embedded
// pointer the field may be, the way any other destination is
// reached.
mv, err := fieldByIndex(dst, schema.embedMaps[0])
if err != nil {
return newDecodeError(key, err)
}
if err := d.assignMap(map[string]any{key: val}, mv); err != nil {
if err := d.assign(map[string]any{key: val}, mv); err != nil {
return newDecodeError(key, err)
}
}
continue
}
if seen != nil {
seen[resolved] = true
}
fv, err := fieldByIndex(dst, field.index)
if err != nil {
return newDecodeError(key, err)
@@ -284,9 +376,26 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
return newDecodeError(key, err)
}
}
for _, key := range schema.required {
if !seen[key] {
return fmt.Errorf("interpres: missing required key %q", key)
}
}
return nil
}
// tableKeys returns the keys of tbl in the order the document wrote them
// when the node index knows it, and in sorted order otherwise, the order a
// hand-built tree or a node-free parse offers. The order settles which of
// two keys that differ only in case wins one field: the same key wins every
// run, instead of whichever a map iteration happened to hand out.
func (d *decoder) tableKeys(tbl map[string]any) []string {
if node := d.nodeOf(tbl); node != nil {
return node.Keys()
}
return slices.Sorted(maps.Keys(tbl))
}
func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
if dst.Type().Key().Kind() != reflect.String {
return fmt.Errorf("interpres: map key must be a string, got %s", dst.Type().Key())
@@ -306,31 +415,58 @@ func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
}
func (d *decoder) assignSlice(items []any, dst reflect.Value) error {
if dst.Kind() != reflect.Slice {
switch dst.Kind() {
case reflect.Slice:
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
dst.Set(out)
return nil
case reflect.Array:
// A fixed-size array takes the elements in place; a length mismatch is
// the error, because a TOML array carries no way to name a default for
// the elements it is short of, and the surplus has nowhere to go.
if dst.Len() != len(items) {
return fmt.Errorf("interpres: cannot assign %d elements to %s", len(items), dst.Type())
}
for i, item := range items {
if err := d.assign(item, dst.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
return nil
default:
return fmt.Errorf("interpres: cannot assign array to %s", dst.Type())
}
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
dst.Set(out)
return nil
}
func (d *decoder) assignTableSlice(items []map[string]any, dst reflect.Value) error {
if dst.Kind() != reflect.Slice {
switch dst.Kind() {
case reflect.Slice:
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
dst.Set(out)
return nil
case reflect.Array:
if dst.Len() != len(items) {
return fmt.Errorf("interpres: cannot assign %d elements to %s", len(items), dst.Type())
}
for i, item := range items {
if err := d.assign(item, dst.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
return nil
default:
return fmt.Errorf("interpres: cannot assign array of tables to %s", dst.Type())
}
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
dst.Set(out)
return nil
}
// --- low-level setters -----------------------------------------------------
@@ -365,6 +501,26 @@ func setDateTime(v time.Time, dst reflect.Value) error {
return nil
}
// setLocalTimeValue stores a local date-time value into a plain time.Time
// destination, which the decoder permits only when LocalTimeLocation fixed
// the zone the wall-clock value is carried in; without it the wrapper types
// are the only destinations a local kind fills, as they always have been.
func (d *decoder) setLocalTimeValue(t time.Time, dst reflect.Value) error {
if dst.Type() == timeType {
if d.loc != nil {
// A local value is a wall clock, so the zone choice relabels it
// rather than shifting the instant: 07:32 in the document is
// 07:32 in the location, not an hour later.
dst.Set(reflect.ValueOf(time.Date(
t.Year(), t.Month(), t.Day(),
t.Hour(), t.Minute(), t.Second(), t.Nanosecond(), d.loc)))
return nil
}
return fmt.Errorf("interpres: cannot assign local date-time to time.Time; set LocalTimeLocation to choose the zone")
}
return fmt.Errorf("interpres: cannot assign local date-time to %s", dst.Type())
}
func setBasic(dst, val reflect.Value, kind string) error {
if dst.Kind() != val.Kind() {
return fmt.Errorf("interpres: cannot assign %s to %s", kind, dst.Type())
@@ -389,6 +545,28 @@ func setDuration(dst reflect.Value, s string) error {
return nil
}
// setNumber stores a Number, the literal NumbersAsLiterals keeps. A Number destination
// takes the literal as it is; every other destination takes the evaluated
// value through the ordinary rules, so an integer field, a float field and a
// duration field all read a Number the way they read the evaluated kind.
func setNumber(dst reflect.Value, n Number) error {
if dst.Type() == numberType {
dst.SetString(string(n))
return nil
}
v, err := decodeNumber(string(n))
if err != nil {
return fmt.Errorf("interpres: %w", err)
}
switch v := v.(type) {
case int64:
return setInt(dst, v)
case float64:
return setFloat(dst, v)
}
return fmt.Errorf("interpres: cannot assign number to %s", dst.Type())
}
func setInt(dst reflect.Value, v int64) error {
switch dst.Kind() {
case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64:
@@ -436,10 +614,12 @@ func setFloat(dst reflect.Value, v float64) error {
// structFieldLoc locates one destination field by its index path from the
// struct root and by the depth the field sits at, which breaks name clashes
// in favour of the shallower field.
// in favour of the shallower field. required records the tag option of the
// field that won the name.
type structFieldLoc struct {
index []int
depth int
index []int
depth int
required bool
}
// structSchema flattens the exported fields of t for decode, mirroring the
@@ -447,10 +627,11 @@ type structFieldLoc struct {
// keys of the same table, and an untagged embedded map is recorded in
// embedMaps (first declaration first) as the destination for leftover keys.
// When two fields resolve to one name, the shallower wins, then the later
// declaration.
// declaration. required holds the keys a `toml:"...,required"` tag demands.
type structSchema struct {
byName map[string]structFieldLoc
embedMaps [][]int
required []string
}
// structSchemaCache holds one schema per struct type. A schema is immutable
@@ -486,11 +667,20 @@ func newStructSchema(t reflect.Type) structSchema {
}
path := append(append([]int{}, prefix...), i)
name := ""
required := false
if tag, ok := f.Tag.Lookup("toml"); ok {
name, _, _ = strings.Cut(tag, ",")
var opts string
name, opts, _ = strings.Cut(tag, ",")
if name == "-" {
continue
}
for opts != "" {
var opt string
opt, opts, _ = strings.Cut(opts, ",")
if opt == "required" {
required = true
}
}
}
if f.Anonymous && name == "" {
ft := f.Type
@@ -514,11 +704,19 @@ func newStructSchema(t reflect.Type) structSchema {
}
key := strings.ToLower(name)
if existing, ok := s.byName[key]; !ok || depth <= existing.depth {
s.byName[key] = structFieldLoc{index: path, depth: depth}
s.byName[key] = structFieldLoc{index: path, depth: depth, required: required}
}
}
}
walk(t, nil, 0)
// The missing-key error must not depend on map order, so the demanded keys
// come out sorted.
for key, loc := range s.byName {
if loc.required {
s.required = append(s.required, key)
}
}
slices.Sort(s.required)
return s
}
+685 -10
View File
@@ -11,7 +11,9 @@ import (
"net"
"slices"
"strings"
"sync/atomic"
"testing"
"testing/synctest"
"time"
)
@@ -167,7 +169,7 @@ func TestDecoderDecodeContextHonoursCancellation(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
var cfg map[string]any
err := NewDecoder().DecodeContext(ctx, []byte("a = 1\n"), &cfg)
err := UnmarshalContext(ctx, []byte("a = 1\n"), &cfg)
if !errors.Is(err, context.Canceled) {
t.Fatalf("DecodeContext returned %v, want context.Canceled", err)
}
@@ -192,7 +194,7 @@ count = 3
if out.Title != "x" || out.Count != 3 {
t.Errorf("out = %#v", out)
}
if err := NewDecoder().DecodeContext(context.Background(), in, &map[string]any{}); err != nil {
if err := Unmarshal(in, &map[string]any{}); err != nil {
t.Fatalf("Decoder.DecodeContext: %v", err)
}
}
@@ -693,8 +695,7 @@ func TestUnmarshalStrictEmbeddedMapStaysStrict(t *testing.T) {
RoundTripExtra
Name string `toml:"name"`
}
dec := NewDecoder().DisallowUnknownFields()
err := dec.Decode([]byte("name = \"n\"\nrogue = 1\n"), &Cfg{})
err := Unmarshal([]byte("name = \"n\"\nrogue = 1\n"), &Cfg{}, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), "unknown field") {
t.Fatalf("expected unknown field error, got: %v", err)
}
@@ -725,8 +726,8 @@ func TestDecodeErrorCarriesPath(t *testing.T) {
if de.Err == nil || !strings.Contains(de.Err.Error(), "overflows uint8") {
t.Fatalf("Err = %v", de.Err)
}
// The rendered message keeps its shape: segments joined with ": ".
wantMsg := "items: [0]: weight: interpres: integer 300 overflows uint8"
// The rendered message uses the Path notation.
wantMsg := "interpres: items[0].weight: integer 300 overflows uint8"
if err.Error() != wantMsg {
t.Fatalf("message = %q, want %q", err.Error(), wantMsg)
}
@@ -978,10 +979,10 @@ func TestDecoderMaxDepth(t *testing.T) {
var cfg struct {
V any `toml:"v"`
}
if err := NewDecoder().MaxDepth(4).Decode(deep(4), &cfg); err != nil {
if err := Unmarshal(deep(4), &cfg, MaxNestingDepth(4)); err != nil {
t.Fatalf("at the limit: %v", err)
}
err := NewDecoder().MaxDepth(4).Decode(deep(5), &cfg)
err := Unmarshal(deep(5), &cfg, MaxNestingDepth(4))
if err == nil {
t.Fatal("expected a nesting error")
}
@@ -995,10 +996,10 @@ func TestDecoderMaxInputSize(t *testing.T) {
var cfg struct {
V string `toml:"v"`
}
if err := NewDecoder().MaxInputSize(len(doc)).Decode(doc, &cfg); err != nil {
if err := Unmarshal(doc, &cfg, MaxInputSize(len(doc))); err != nil {
t.Fatalf("at the limit: %v", err)
}
err := NewDecoder().MaxInputSize(len(doc)-1).Decode(doc, &cfg)
err := Unmarshal(doc, &cfg, MaxInputSize(len(doc)-1))
if err == nil {
t.Fatal("expected a size error")
}
@@ -1097,3 +1098,677 @@ func TestUnmarshalerReceivesOffsetDateTime(t *testing.T) {
t.Errorf("local kind = %q, want interpres.LocalDateTime", cfg.L.Kind)
}
}
func TestDecoderUseNumber(t *testing.T) {
data := []byte(`hex = 0x1f
sep = 1_000
signed = +1.0
exp = 1e6
posinf = inf
negzero = -0.0
plain = 42
frac = 2.5
`)
t.Run("the tree keeps the literal", func(t *testing.T) {
var tree map[string]any
if err := Unmarshal(data, &tree, NumbersAsLiterals(true)); err != nil {
t.Fatal(err)
}
for lit, key := range map[string]string{
"0x1f": "hex", "1_000": "sep", "+1.0": "signed", "1e6": "exp",
"inf": "posinf", "-0.0": "negzero", "42": "plain", "2.5": "frac",
} {
got, ok := tree[key].(Number)
if !ok {
t.Errorf("%s = %T, want Number", key, tree[key])
continue
}
if string(got) != lit {
t.Errorf("%s = %q, want %q", key, got, lit)
}
}
})
t.Run("typed fields take the evaluated value", func(t *testing.T) {
var cfg struct {
Hex Number `toml:"hex"`
Plain int64 `toml:"plain"`
Frac float64 `toml:"frac"`
Rate time.Duration
}
if err := Unmarshal([]byte("hex = 0x1f\nplain = 42\nfrac = 2.5\nRate = 1_000\n"), &cfg, NumbersAsLiterals(true)); err != nil {
t.Fatal(err)
}
if cfg.Hex != "0x1f" {
t.Errorf("hex = %q, want 0x1f", cfg.Hex)
}
if cfg.Plain != 42 {
t.Errorf("plain = %d, want 42", cfg.Plain)
}
if cfg.Frac != 2.5 {
t.Errorf("frac = %g, want 2.5", cfg.Frac)
}
if cfg.Rate != 1000 {
t.Errorf("rate = %s, want 1µs", cfg.Rate)
}
})
t.Run("invalid numbers are still parse errors", func(t *testing.T) {
for _, in := range []string{"a = 01\n", "a = 1__0\n", "a = 1x\n"} {
var tree map[string]any
if err := Unmarshal([]byte(in), &tree, NumbersAsLiterals(true)); err == nil {
t.Errorf("%q decoded without an error", in)
}
}
})
t.Run("without UseNumber the tree holds the evaluated kinds", func(t *testing.T) {
var tree map[string]any
if err := Unmarshal([]byte("hex = 0x1f\nfrac = 2.5\n"), &tree); err != nil {
t.Fatal(err)
}
if v, ok := tree["hex"].(int64); !ok || v != 31 {
t.Errorf("hex = %#v, want int64 31", tree["hex"])
}
if v, ok := tree["frac"].(float64); !ok || v != 2.5 {
t.Errorf("frac = %#v, want float64 2.5", tree["frac"])
}
})
}
func TestNumberMethods(t *testing.T) {
tests := []struct {
lit Number
wantI int64
wantF float64
intErr bool
}{
{lit: "42", wantI: 42, wantF: 42},
{lit: "0x1f", wantI: 31, wantF: 31},
{lit: "1_000", wantI: 1000, wantF: 1000},
{lit: "+1.0", wantF: 1, intErr: true},
{lit: "1e6", wantF: 1e6, intErr: true},
{lit: "inf", wantF: math.Inf(1), intErr: true},
{lit: "-2.5", wantF: -2.5, intErr: true},
}
for _, tt := range tests {
i, err := tt.lit.Int64()
if tt.intErr && err == nil {
t.Errorf("%q.Int64() succeeded with %d, want an error", tt.lit, i)
}
if !tt.intErr {
if err != nil {
t.Errorf("%q.Int64() = %v", tt.lit, err)
continue
}
if i != tt.wantI {
t.Errorf("%q.Int64() = %d, want %d", tt.lit, i, tt.wantI)
}
}
f, err := tt.lit.Float64()
if err != nil {
t.Errorf("%q.Float64() = %v", tt.lit, err)
continue
}
if f != tt.wantF {
t.Errorf("%q.Float64() = %g, want %g", tt.lit, f, tt.wantF)
}
}
for _, lit := range []Number{"01", "1__0", "abc", ""} {
if _, err := lit.Float64(); err == nil {
t.Errorf("%q.Float64() succeeded, want an error", lit)
}
if _, err := lit.Int64(); err == nil {
t.Errorf("%q.Int64() succeeded, want an error", lit)
}
}
}
func TestSyntaxErrorPosition(t *testing.T) {
src := []byte("alpha = 1\nbeta x = 2\n")
_, err := ParseMap(src)
se, ok := errors.AsType[*SyntaxError](err)
if !ok {
t.Fatalf("err = %v, want a SyntaxError", err)
}
if se.Line != 2 {
t.Errorf("Line = %d, want 2", se.Line)
}
if want := strings.Index(string(src), "x"); se.Offset != want {
t.Errorf("Offset = %d, want %d", se.Offset, want)
}
if se.Column != 6 {
t.Errorf("Column = %d, want 6", se.Column)
}
want := "beta x = 2\n ^"
if got := se.SourceLine(src); got != want {
t.Errorf("SourceLine =\n%s\nwant:\n%s", got, want)
}
}
func TestSyntaxErrorUTF8Offset(t *testing.T) {
src := []byte("a = \"ok\"\nb = \"\xff\xfe\"\n")
_, err := ParseMap(src)
se, ok := errors.AsType[*SyntaxError](err)
if !ok {
t.Fatalf("err = %v, want a SyntaxError", err)
}
if !strings.Contains(se.Msg, "byte offset 14") {
t.Errorf("Msg = %q, want it to name byte offset 14", se.Msg)
}
if se.Offset != 14 {
t.Errorf("Offset = %d, want 14", se.Offset)
}
want := "b = \"\xff\xfe\"\n ^"
if got := se.SourceLine(src); got != want {
t.Errorf("SourceLine =\n%q\nwant:\n%q", got, want)
}
}
func TestSourceLineEdgePositions(t *testing.T) {
src := []byte("a = 1\n")
e := &SyntaxError{Line: 1, Msg: "no position"}
if got, want := e.SourceLine(src), "a = 1\n^"; got != want {
t.Errorf("SourceLine(zero offset) =\n%q\nwant:\n%q", got, want)
}
e = &SyntaxError{Line: 2, Offset: 100, Msg: "past the end"}
if got, want := e.SourceLine(src), "\n^"; got != want {
t.Errorf("SourceLine(offset past end) =\n%q\nwant:\n%q", got, want)
}
}
func TestDecodeFixedArray(t *testing.T) {
t.Run("value array", func(t *testing.T) {
var cfg struct {
Ports [2]int `toml:"ports"`
Label [2]string `toml:"label"`
Grid [2][2]int64 `toml:"grid"`
}
in := []byte("ports = [8080, 9090]\nlabel = [\"a\", \"b\"]\ngrid = [[1, 2], [3, 4]]\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Ports != [2]int{8080, 9090} || cfg.Label != [2]string{"a", "b"} || cfg.Grid != [2][2]int64{{1, 2}, {3, 4}} {
t.Errorf("decoded %+v", cfg)
}
})
t.Run("array of tables", func(t *testing.T) {
type Item struct {
Name string `toml:"name"`
Qty int `toml:"qty"`
}
var cfg struct {
Items [2]Item `toml:"items"`
}
in := []byte("[[items]]\nname = \"a\"\nqty = 1\n[[items]]\nname = \"b\"\nqty = 2\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Items != [2]Item{{"a", 1}, {"b", 2}} {
t.Errorf("decoded %+v", cfg)
}
})
t.Run("a length mismatch is an error", func(t *testing.T) {
var cfg struct {
Ports [3]int `toml:"ports"`
}
err := Unmarshal([]byte("ports = [8080, 9090]\n"), &cfg)
want := "interpres: ports: cannot assign 2 elements to [3]int"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
}
func TestRequiredTag(t *testing.T) {
type Config struct {
Host string `toml:"host,required"`
Radius int `toml:"radius"`
}
t.Run("a present key satisfies the tag", func(t *testing.T) {
var cfg Config
if err := Unmarshal([]byte("radius = 2\nhost = \"example.org\"\n"), &cfg); err != nil {
t.Fatal(err)
}
if cfg.Host != "example.org" || cfg.Radius != 2 {
t.Errorf("decoded %+v", cfg)
}
})
t.Run("a missing key is an error", func(t *testing.T) {
var cfg Config
err := Unmarshal([]byte("radius = 2\n"), &cfg)
want := `interpres: missing required key "host"`
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("the error carries the key path", func(t *testing.T) {
var outer struct {
Server Config `toml:"server"`
}
err := Unmarshal([]byte("[server]\nradius = 1\n"), &outer)
want := `interpres: server: missing required key "host"`
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("case-insensitive match satisfies the tag", func(t *testing.T) {
var cfg Config
if err := Unmarshal([]byte("HOST = \"x\"\n"), &cfg); err != nil {
t.Errorf("err = %v, want nil", err)
}
})
}
type ctxRecorder struct {
got context.Context
value any
}
func (r *ctxRecorder) UnmarshalTOMLContext(ctx context.Context, data any) error {
r.got = ctx
r.value = data
return nil
}
func TestUnmarshalerContext(t *testing.T) {
t.Run("the context reaches the method", func(t *testing.T) {
type keyT struct{}
ctx := context.WithValue(context.Background(), keyT{}, "sentinel")
var r ctxRecorder
if err := UnmarshalContext(ctx, []byte("a = 1\n"), &r); err != nil {
t.Fatal(err)
}
if v, _ := r.got.Value(keyT{}).(string); v != "sentinel" {
t.Errorf("ctx = %v, want the caller's context", r.got)
}
tree, isMap := r.value.(map[string]any)
if !isMap || tree["a"] != int64(1) {
t.Errorf("value = %#v, want the tree with a = 1", r.value)
}
})
t.Run("the context wins over Unmarshaler", func(t *testing.T) {
var v struct {
R ctxBoth `toml:"r"`
}
if err := Unmarshal([]byte("r = 1\n"), &v); err != nil {
t.Fatal(err)
}
if !v.R.ctxCalled {
t.Error("UnmarshalTOMLContext was not called")
}
if v.R.plainCalled {
t.Error("UnmarshalTOML was called although the context method exists")
}
})
t.Run("a non-cancellable entry point hands in Background", func(t *testing.T) {
var r ctxRecorder
if err := Unmarshal([]byte("a = 1\n"), &r); err != nil {
t.Fatal(err)
}
if r.got != context.Background() {
t.Errorf("ctx = %v, want context.Background", r.got)
}
})
}
type ctxBoth struct {
ctxCalled bool
plainCalled bool
}
func (b *ctxBoth) UnmarshalTOMLContext(ctx context.Context, data any) error {
b.ctxCalled = true
return nil
}
func (b *ctxBoth) UnmarshalTOML(data any) error {
b.plainCalled = true
return nil
}
func TestEmbeddedMapRule(t *testing.T) {
t.Run("only the first embedded map takes the leftover keys", func(t *testing.T) {
type ExtraMap map[string]any
type MoreMap map[string]any
type Config struct {
Port int
ExtraMap
MoreMap
}
var cfg Config
in := []byte("port = 8080\nlang = \"cs\"\nregion = \"EU\"\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Port != 8080 {
t.Errorf("port = %d", cfg.Port)
}
if cfg.ExtraMap["lang"] != "cs" || cfg.ExtraMap["region"] != "EU" {
t.Errorf("extra = %v, want the leftover keys", cfg.ExtraMap)
}
if len(cfg.MoreMap) != 0 {
t.Errorf("more = %v, want empty: only the first embedded map is the filler", cfg.MoreMap)
}
})
t.Run("a tagged embedded map is an ordinary field", func(t *testing.T) {
type Config struct {
Extra map[string]any `toml:"extra"`
}
var cfg Config
if err := Unmarshal([]byte("extra = {a = 1}\n"), &cfg); err != nil {
t.Fatal(err)
}
if cfg.Extra["a"] != int64(1) {
t.Errorf("extra = %v", cfg.Extra)
}
})
}
func TestDecodeMergesIntoNonEmptyMap(t *testing.T) {
dst := map[string]any{"keep": "me", "port": 1}
if err := Unmarshal([]byte("port = 8080\nlang = \"cs\"\n"), &dst); err != nil {
t.Fatal(err)
}
if dst["keep"] != "me" {
t.Errorf("keep = %v, want the pre-existing key kept", dst["keep"])
}
if dst["port"] != int64(8080) {
t.Errorf("port = %v, want the document's value to win", dst["port"])
}
if dst["lang"] != "cs" {
t.Errorf("lang = %v, want the key added", dst["lang"])
}
}
func TestPathString(t *testing.T) {
p := Path{"server", "ports", "[2]", "host"}
if got, want := p.String(), "server.ports[2].host"; got != want {
t.Errorf("String() = %q, want %q", got, want)
}
if got := (Path{}).String(); got != "" {
t.Errorf("String() of an empty path = %q, want the empty string", got)
}
}
func TestLocalTimeLocation(t *testing.T) {
zone := time.FixedZone("CET", 3600)
t.Run("without the option a local kind fills only its wrapper", func(t *testing.T) {
var cfg struct {
When time.Time `toml:"when"`
}
err := Unmarshal([]byte("when = 1979-05-27T07:32:00\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), "LocalTimeLocation") {
t.Errorf("err = %v, want the option hint", err)
}
})
t.Run("with the option the value lands in the zone", func(t *testing.T) {
var cfg struct {
When time.Time `toml:"when"`
Date LocalDate `toml:"date"`
Wall LocalDateTime `toml:"wall"`
}
in := []byte("when = 1979-05-27T07:32:00\ndate = 1979-05-27\nwall = 1979-05-27T07:32:00\n")
if err := Unmarshal(in, &cfg, LocalTimeLocation(zone)); err != nil {
t.Fatal(err)
}
if got := cfg.When.Format("15:04:05 MST"); got != "07:32:00 CET" {
t.Errorf("when = %s, want 07:32:00 CET", got)
}
if cfg.Date != (LocalDate{time.Date(1979, 5, 27, 0, 0, 0, 0, time.UTC)}) {
t.Errorf("date = %v", cfg.Date)
}
})
t.Run("the wrapper still takes the value with the option on", func(t *testing.T) {
var cfg struct {
Wall LocalDateTime `toml:"wall"`
}
if err := Unmarshal([]byte("wall = 1979-05-27T07:32:00\n"), &cfg, LocalTimeLocation(zone)); err != nil {
t.Fatal(err)
}
if cfg.Wall.Hour() != 7 {
t.Errorf("wall = %v", cfg.Wall)
}
})
}
// errAfterN is a context that reports cancelled once its Err has been read
// more than n times, which drives the in-value cancellation checks: the
// parser reads Err a fixed number of times per statement, so a huge array
// fails only where the checks inside the value run.
type errAfterN struct {
context.Context
n int
how atomic.Int32
}
func (c *errAfterN) Err() error {
if c.how.Add(1) > int32(c.n) {
return context.Canceled
}
return nil
}
func TestCancelInsideValue(t *testing.T) {
// Two top-level checks happen before the value (the entry check and the
// statement loop's first); the array checks follow inside the value, so
// the third read is the first that can fail today. The document only
// parses to the end when the checks inside the value are missing, which
// is the defect this test pins.
var b strings.Builder
b.WriteString("a = [")
for i := range 4000 {
if i > 0 {
b.WriteByte(',')
}
b.WriteString("1")
}
b.WriteString("]\n")
ctx := &errAfterN{Context: context.Background(), n: 2}
var tree map[string]any
err := UnmarshalContext(ctx, []byte(b.String()), &tree)
if err == nil {
t.Fatal("a cancelled context did not stop the parse inside the value")
}
if !errors.Is(err, context.Canceled) {
t.Errorf("err = %v, want context.Canceled", err)
}
}
func TestCancellationSynctest(t *testing.T) {
// The bubble makes the cost of the immediate-cancellation path visible in
// virtual microseconds, and synctest.Wait holds the test to leaving no
// goroutine behind.
synctest.Test(t, func(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
cancel()
start := time.Now()
var tree map[string]any
err := UnmarshalContext(ctx, []byte("a = 1\n"), &tree)
if !errors.Is(err, context.Canceled) {
t.Errorf("err = %v, want context.Canceled", err)
}
if d := time.Since(start); d != 0 {
t.Errorf("the parse consumed %v of virtual time, want none", d)
}
synctest.Wait()
})
}
func TestErrorMessagesGolden(t *testing.T) {
// The exact texts the library promises, pinned against unintended edits.
type Config struct {
Weight uint8 `toml:"weight"`
}
tests := []struct {
name string
read func() error
want string
}{
{
name: "missing equals",
read: func() error { _, err := ParseMap([]byte("a 1\n")); return err },
want: "interpres: line 1: expected '=' after key",
},
{
name: "duplicate key",
read: func() error { _, err := ParseMap([]byte("a = 1\na = 2\n")); return err },
want: `interpres: line 2: duplicate key "a"`,
},
{
name: "unterminated string",
read: func() error { _, err := ParseMap([]byte("a = \"open\n")); return err },
want: "interpres: line 1: unterminated string",
},
{
name: "leading zero",
read: func() error { _, err := ParseMap([]byte("a = 01\n")); return err },
want: "interpres: line 1: leading zeros are not allowed in numbers",
},
{
name: "bad escape",
read: func() error { _, err := ParseMap([]byte(`a = "\q"` + "\n")); return err },
want: `interpres: line 1: invalid escape sequence \q`,
},
{
name: "nesting limit",
read: func() error {
var b strings.Builder
b.WriteString("a = ")
for range 11 {
b.WriteString("[")
}
for range 11 {
b.WriteString("]")
}
b.WriteString("\n")
var tree map[string]any
err := Unmarshal([]byte(b.String()), &tree, MaxNestingDepth(10))
return err
},
want: "interpres: line 1: nesting exceeds the limit of 10",
},
{
name: "decode overflow",
read: func() error {
var cfg Config
return Unmarshal([]byte("weight = 300\n"), &cfg)
},
want: "interpres: weight: integer 300 overflows uint8",
},
{
name: "unknown field",
read: func() error {
var cfg struct {
Known int `toml:"known"`
}
return Unmarshal([]byte("mystery = 1\n"), &cfg, RejectUnknownFields(true))
},
want: `interpres: unknown field "mystery" for struct { Known int "toml:\"known\"" }`,
},
{
name: "decode target",
read: func() error { return Unmarshal([]byte("a = 1\n"), Config{}) },
want: "interpres: decode target must be a non-nil pointer",
},
{
name: "encode nil pointer",
read: func() error {
var p *Config
_, err := Marshal(p)
return err
},
want: "interpres: cannot marshal nil pointer",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
err := tt.read()
if err == nil {
t.Fatalf("no error, want %q", tt.want)
}
if err.Error() != tt.want {
t.Errorf("message = %q, want %q", err.Error(), tt.want)
}
})
}
}
// TestLocalTimeLocationLeavesOffsetsAlone pins that the option's zone is
// used for local date-times only: an offset date-time keeps the offset the
// document wrote.
func TestLocalTimeLocationLeavesOffsetsAlone(t *testing.T) {
var cfg struct {
Stamp time.Time `toml:"stamp"`
}
err := Unmarshal([]byte("stamp = 1979-05-27T07:32:00-07:00\n"), &cfg,
LocalTimeLocation(time.FixedZone("Prague", 2*60*60)))
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if _, off := cfg.Stamp.Zone(); off != -7*60*60 {
t.Errorf("offset = %d, want the document's -07:00", off/3600)
}
}
// TestUnmarshalCaseCollisionIsDeterministic pins that two keys differing
// only in case, both matching one field, resolve the same way on every run
// and on both decode paths.
func TestUnmarshalCaseCollisionIsDeterministic(t *testing.T) {
type cfg struct {
Host string `toml:"host"`
}
in := []byte("Host = \"upper\"\nhost = \"lower\"\n")
// The tree path iterates a map, so pin the winner across many runs.
var want string
for range 50 {
var viaTree cfg
if err := treeDecodeInto(in, &viaTree); err != nil {
t.Fatalf("tree decode: %v", err)
}
if want == "" {
want = viaTree.Host
} else if viaTree.Host != want {
t.Fatalf("tree decode is not deterministic: %q then %q", want, viaTree.Host)
}
}
var targeted cfg
if err := Unmarshal(in, &targeted); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if targeted.Host != want {
t.Errorf("targeted Host = %q, tree %q", targeted.Host, want)
}
}
// TestUnmarshalEmbeddedPointerMap pins that leftover keys reach an embedded
// pointer to a map, allocating it, rather than panicking on the pointer.
func TestUnmarshalEmbeddedPointerMap(t *testing.T) {
type Extra map[string]int
type cfg struct {
*Extra
Name string `toml:"name"`
}
var c cfg
err := Unmarshal([]byte("name = \"x\"\nrogue = 7\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Extra == nil || (*c.Extra)["rogue"] != 7 {
t.Errorf("embedded map = %v, want rogue allocated and filled", c.Extra)
}
}
// TestUnmarshalIgnoresEncodeTagOptions pins that the emission-only tag
// options change nothing on the decode side.
func TestUnmarshalIgnoresEncodeTagOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name,omitempty"`
Port int `toml:"port,omitzero,comment=The port"`
}
var c cfg
err := Unmarshal([]byte("name = \"x\"\nport = 8080\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Name != "x" || c.Port != 8080 {
t.Errorf("cfg = %+v, want both fields filled", c)
}
}
+389 -93
View File
@@ -50,6 +50,100 @@ tree, err := interpres.ParseMap([]byte("title = \"x\"\nport = 8080\n"))
The cancellable variant of `ParseMap`.
### `func ParseFile(path string) (*Document, error)`
Reads the file at `path` and parses it into a [Document](#documents), the shape
`Parse` gives. Both a read failure and a parse failure come back with the file
name as their first words, wrapped so `errors.AsType` still reaches the
`SyntaxError` inside a parse failure.
```go
doc, err := interpres.ParseFile("config.toml")
```
### `func Valid(data []byte) error`
Reports whether `data` is a valid TOML document: `nil` when the parser accepts
it, the parse error when it does not. It is the library call the `-validate`
mode of interpres-decode is built on.
```go
if err := interpres.Valid(data); err != nil {
fmt.Println("invalid:", err)
}
```
### Options
The decode and encode calls take variadic options, the shape
encoding/json/v2 uses for its own. Each is a function value over the private
settings of one call, and they compose by listing:
```go
cfg, err := interpres.Unmarshal(data, &cfg2,
interpres.RejectUnknownFields(true),
interpres.NumbersAsLiterals(true))
```
Decode options:
| Option | Default | Effect |
|---|---|---|
| `RejectUnknownFields(v bool)` | off | a key with no matching struct field is an error |
| `NumbersAsLiterals(v bool)` | off | integers and floats decode into `Number`, which carries the literal; see [Numbers as literals](#numbers-as-literals) |
| `MaxNestingDepth(depth int)` | `10000` | bound how deeply arrays and inline tables may nest |
| `MaxInputSize(size int)` | no limit | bound the size of the document, in bytes |
| `LocalTimeLocation(loc)` | nil | the zone a local date-time is carried in when it decodes into a `time.Time` |
Encode options:
| Option | Default | Effect |
|---|---|---|
| `Layout(kind LayoutKind)` | `LayoutKindGrouped` | group entries as scalars, then sub-tables, then arrays of tables; `LayoutKindDeclaration` preserves declaration order |
| `OmitEmptyArrays(v bool)` | off | skip `key = []` for empty scalar arrays |
| `LiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
| `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header |
### `func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error)`
The generic shorthand for `Unmarshal` with a destination variable:
```go
cfg, err := interpres.ParseAs[Config](data)
```
The zero `T` comes back with the error.
### `func NewSchema[T any]()`
Precompiles the codec for `T`: the struct schema both directions walk and the
interface flags the decoder and encoder resolve through are built once and
cached, so the first document pays the cost instead of the hot path. A `T`
that is not a struct warms nothing.
### `func Statements(r io.Reader) iter.Seq2[Statement, error]`
Iterates the top-level statements of the document r carries, in written
order: key/value statements, a value array or an inline table among them as
one statement whatever it holds, a `[table]` header as one statement carrying
its `Table` node, and an `[[array of tables]]` as one statement per element
with the element's node and its `Index`. Iteration stops at the first error
and at a false yield, so a caller looking for one section reads no further.
The reader is consumed in full before the first yield, because the parser
scans the source in place.
```go
for stmt, err := range interpres.Statements(file) {
if err != nil {
return err
}
if stmt.Table != nil {
fmt.Println(stmt.Key, stmt.Table.Keys())
}
}
```
## Documents
`Parse` returns a `Document`: the value tree together with what a map cannot
@@ -82,6 +176,37 @@ document and from `doc.Map()` is the same value.
array of tables, and the inline tables inside a value array, with `nil` for the
elements that are not tables.
### Editing a document
The document is writable, which makes the read-change-write loop a round trip
through one value. The typed getters read with one call:
| Getter | Returns |
|---|---|
| `GetString(key)` | `(string, bool)` |
| `GetInt(key)` | `(int64, bool)` |
| `GetFloat(key)` | `(float64, bool)` |
| `GetBool(key)` | `(bool, bool)` |
| `GetArray(key)` | `([]any, bool)` |
| `GetTable(key)` | `(*Table, bool)` |
`Set(key, value)` stores a value, keeping an existing key's position and
comments and appending a new key to the end; a `map[string]any` value becomes
a table of its own under a header, its keys in sorted order. `Delete(key)`
removes a key and everything it holds. Every method exists on `Document` for
the root table and on `Table` for the table itself.
`Marshal` writes the document back as it stands: keys in written order, the
comments above the lines and headers they belonged to, tables that were
written inline written inline again. `UnmarshalDocument(doc, v)` decodes the
edited document into a typed destination without parsing again.
```go
doc, err := interpres.Parse(data)
doc.Set("port", 9090)
out, err := interpres.Marshal(doc)
```
### Comments
A comment belongs to the line it precedes or follows, and to the node that line
@@ -99,11 +224,7 @@ introduced:
and no surrounding space, so `# note` is stored as `note` and a bare `#` as
`""`.
A `Document` is not a value to marshal: `Marshal` writes values, so it refuses
one and points at `doc.Map()`. Writing a document back, with its order and its
comments, belongs with the editing API.
### `func Unmarshal(data []byte, v any) error`
### `func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error`
Parses `data` and stores the result in the value pointed to by `v`, typically a
pointer to a struct or to `map[string]any`. Equivalent to
@@ -116,11 +237,11 @@ if err := interpres.Unmarshal(data, &cfg); err != nil {
}
```
### `func UnmarshalContext(ctx context.Context, data []byte, v any) error`
### `func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...UnmarshalOption) error`
The cancellable variant of `Unmarshal`.
### `func Marshal(v any) ([]byte, error)`
### `func Marshal(v any, opts ...MarshalOption) ([]byte, error)`
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
TOML document. The emission rules are in the [Encoding](#encoding) section
@@ -130,11 +251,16 @@ below. Equivalent to `MarshalContext(context.Background(), v)`.
out, err := interpres.Marshal(cfg)
```
### `func MarshalContext(ctx context.Context, v any) ([]byte, error)`
### `func MarshalContext(ctx context.Context, v any, opts ...MarshalOption) ([]byte, error)`
The cancellable variant of `Marshal`. The context is checked before any work
and every 64 fields during the reflection walk.
### `func MarshalAppend(buf []byte, v any, opts ...MarshalOption) ([]byte, error)`
Appends the TOML encoding of `v` to `buf` and returns the extended buffer, the
shape `json.MarshalAppend` has. A failed encoding leaves `buf` untouched.
## Decoding
### Value mapping
@@ -157,15 +283,20 @@ and every 64 fields during the reflection walk.
When decoding into a struct, these values convert onto the destination's
concrete types: any integer or unsigned width, floats, slices, nested structs
and `map[string]T`.
and `map[string]T`. `NumbersAsLiterals` replaces the two numeric rows of the
table with `Number`, which keeps the literal; see
[Numbers as literals](#numbers-as-literals).
### Target constraints
`Unmarshal` and `(*Decoder).Decode` write into a non-nil pointer:
`Unmarshal`, `UnmarshalRead` and `UnmarshalContext` write into a non-nil pointer:
- `*struct`, matched per the field rules below
- `*map[string]any` or `*map[string]T`, keys become map keys and values decode
into `T` recursively
into `T` recursively; a map that already holds entries is merged into, the
document's values replacing same-named keys and the rest left standing
- `*OrderedMap`, the keys fill in the order the document wrote them; see
[Ordered tables](#ordered-tables)
- `*any`, receives the whole parsed tree unchanged
Anything else returns `interpres: decode target must be a non-nil pointer`.
@@ -181,7 +312,9 @@ For a struct destination, a TOML key matches a field as follows:
into the embedded struct and matches its own fields against the same keys,
mirroring how the encoder flattens it. A nil embedded pointer struct is
allocated on demand. An untagged embedded map receives the keys no field
claims.
claims; when a struct embeds several untagged maps, the first one
declared takes all of them and the rest stay untouched, so the rule stays
predictable.
4. The key itself is lower-cased before lookup, so the match is
case-insensitive on both sides: `DATABASEURL` matches a field named
`DatabaseUrl`.
@@ -195,6 +328,11 @@ one declared later wins.
Unknown keys are ignored by default, landing in an untagged embedded map when
the struct has one; [Strict decoding](#strict-decoding) rejects them instead.
The tag may carry the `required` option, `toml:"host,required"`: the decode
fails with `missing required key "host"` when no key of the document resolved
to the field. The check runs after the table is read, so the other fields
carry their values whether the required one is present or not.
### Numeric conversion
The parser produces `int64` for every integer and `float64` for every float.
@@ -211,6 +349,29 @@ The decoder converts to the destination type with explicit overflow checks:
A conversion that the rules do not allow produces an error wrapped with the
offending key or index, for example `p: interpres: integer 300 overflows uint8`.
### Numbers as literals
`NumbersAsLiterals(true)` decodes every integer and float into `Number`, a
string type that carries the literal the document wrote: `0x1f`, `1_000`,
`+1.0`, `inf`. The shape is validated as strictly as ever, so `01` and `1__0`
remain parse errors; only the evaluated value is replaced by the literal. A
round trip through the value tree and `Marshal` keeps the spelling, where the
default tree normalises `0x1f` to `31` and `+1.0` to `1.0`.
```go
var tree map[string]any
err := interpres.Unmarshal(data, &tree, interpres.NumbersAsLiterals(true))
lit := tree["rate"].(interpres.Number) // "1_000"
```
A destination of a concrete kind is unaffected: an `int64` field, a `float64`
field and a `time.Duration` field take the evaluated value they always took,
and a `Number` field takes the literal. `Number.Float64` and `Number.Int64`
evaluate the literal on demand, with an error for a float asked as an integer
and for a literal that is not a valid TOML number. `Marshal` writes a `Number`
as its bare literal and rejects one that is not a valid TOML number, whether it
stands alone or inside a value array.
### Date-time values
Offset date-times decode into `OffsetDateTime`, whose embedded `time.Time` is the
@@ -227,6 +388,12 @@ error. The date-time types take a bare timestamp and never a quoted string, so a
document that writes a date-time with quotes does not decode into them, and
neither `encoding.TextUnmarshaler` nor the embedded `time.Time` changes that.
`LocalTimeLocation(loc)` lets a local date-time fill a plain
`time.Time` destination as well: the wall-clock value is carried in the
location given, relabelled rather than shifted, so `07:32` in the document is
`07:32` in the zone. Without the option the wrapper types are the only
destinations a local kind fills.
### Arrays of tables
A `[[a]]` block parses into a `[]map[string]any` element of the tree. When the
@@ -234,6 +401,11 @@ destination is a slice, each element decodes into the slice's element type
(`[]struct` or `[]map[string]V`); a mismatch on one element surfaces as an
error wrapped with `[i]:` and the element index.
A value array also decodes into a fixed-size array, `[N]T`, the mirror of the
encoder's ability to encode one. The element count has to match: an array
whose length differs from `N` is an error, `interpres: cannot assign 2
elements to [3]int`, wrapped with the key path.
### Custom decoding: `Unmarshaler`
A type that wants full control of its decode implements:
@@ -256,6 +428,21 @@ automatically, and a nil pointer destination is allocated first. An error
returned from `UnmarshalTOML` halts the decode and propagates wrapped with the
key path, for example `addr: unmarshal: not a string`.
### Custom decoding: `UnmarshalerContext`
`UnmarshalerContext` is `Unmarshaler` with the decode's context handed in:
```go
type UnmarshalerContext interface {
UnmarshalTOMLContext(ctx context.Context, data any) error
}
```
A type that implements both gets `UnmarshalTOMLContext`, so a long custom
decode can abort on cancellation instead of running to completion. The
context a non-cancellable entry point carries is `context.Background`, never
nil.
### Custom decoding: `encoding.TextUnmarshaler`
A destination type that implements `encoding.TextUnmarshaler` receives a TOML
@@ -288,15 +475,33 @@ and `from_text = "1h30m"` decode to the same duration. Text that
`time.ParseDuration` rejects, `d = "90"` among it, fails with
`interpres: invalid duration "90"`.
### Strict decoding
### Ordered tables
By default unknown keys are dropped silently. A `Decoder` built with
`DisallowUnknownFields` rejects them instead:
`OrderedMap` is a string-keyed table that remembers the order its keys were
set in, the shape a `map[string]any` cannot carry. Decoding into one fills it
in the order the document wrote the keys, and `Marshal` writes one back in
that order, where a map destination carries no order and a map source sorts
its keys. The type is a decode target on its own, in a struct field, and as
the element of an array of tables.
```go
err := interpres.NewDecoder().
DisallowUnknownFields().
Decode(data, &cfg)
var cfg OrderedMap
err := interpres.Unmarshal(data, &cfg)
out, err := interpres.Marshal(&cfg) // the keys come back in written order
```
The values are untyped, the shape the parser produces, so a nested table
inside an `OrderedMap` is a plain `map[string]any`; the order is kept at the
level the `OrderedMap` sits at. Inside a value array an `OrderedMap` renders
as an ordinary inline table, whose keys are sorted.
### Strict decoding
By default unknown keys are dropped silently. The `RejectUnknownFields`
option rejects them instead:
```go
err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
```
A typo such as `database_urls` then fails with
@@ -306,12 +511,31 @@ depth, including struct elements inside slices; map destinations accept every
key by nature. When several keys are unknown, the message names the smallest
one, so it does not depend on map iteration order.
### Direct decoding
For a struct destination whose type graph carries no untagged embedded map and
no custom decode hook, the decode parses straight into
the destination: the table skeleton is resolved against the struct schema while
the document scans, and no intermediate value tree is kept. Values still flow
through the ordinary assignment rules, so every conversion, hook and error the
[Decoding](#decoding) section states holds verbatim; the parity with the tree
path is pinned by a differential fuzz target that decodes every generated
document both ways and compares the results.
A document or destination the direct skeleton cannot model (an unknown table
under strictness it must sink, a hook that needs the whole parsed value, an
embedded map filler) falls back to the tree path and reruns, so the
observable behaviour is always the tree path's, exactly. Nothing changes for
`Parse`, `ParseMap` or the document API: the tree remains theirs.
### Cancellation
`ParseContext`, `UnmarshalContext` and `(*Decoder).DecodeContext` accept a
`ParseContext`, `UnmarshalContext` and `MarshalContext` accept a
`context.Context`. An already-cancelled context short-circuits with
`context.Canceled` before any work begins; afterwards the context is checked
every 64 top-level statements.
every 64 top-level statements, and inside a value too: an array, an inline
table and a multi-line string check every 64 elements or lines, so one huge
value cannot hold the parse past its cancellation.
### Flow
@@ -320,12 +544,12 @@ sequenceDiagram
participant Caller
participant Unmarshal as Unmarshal
participant Parser as parser
participant Decoder as decoder
participant Decode as decode
Caller->>Unmarshal: data, v
Unmarshal->>Parser: ParseContext(ctx, data)
Parser-->>Unmarshal: tree or *SyntaxError
Unmarshal->>Decoder: decode(tree, reflect value)
Decoder-->>Unmarshal: nil or wrapped field error
Unmarshal->>Parser: targeted parse straight into v
Parser-->>Unmarshal: nil, *SyntaxError, or fallback
Unmarshal->>Decode: tree rerun on fallback
Decode-->>Unmarshal: nil or wrapped field error
Unmarshal-->>Caller: error
```
@@ -333,9 +557,11 @@ sequenceDiagram
### Input constraints
`Marshal` and `(*Encoder).Marshal` accept a `struct`, a `map[string]V`, or a
non-nil pointer to one, where `V` is any value `Marshal` itself understands. A
different top-level value fails:
`Marshal` and `MarshalWrite` accept a `struct`, a `map[string]V`, or a
non-nil pointer to one, where `V` is any value `Marshal` itself understands.
An `OrderedMap` and a `Document` are accepted as themselves: the first in its
written key order, the second written back as it stands. A different
top-level value fails:
| Input | Error |
|---|---|
@@ -343,6 +569,11 @@ different top-level value fails:
| a nil `any` | `interpres: cannot marshal nil value` |
| a nil pointer | `interpres: cannot marshal nil pointer` |
The encoding walk carries a nesting limit of 10000 levels, the parser's own
figure: a value that nests deeper, which cyclic data always does, is rejected
with an error that names the limit and suggests the cycle, instead of running
the stack out.
### Field matching
Struct fields become TOML keys as follows:
@@ -360,22 +591,35 @@ nil map emits nothing.
### Tag options
The part of a `toml` tag after the first comma carries options. Both options
shape emission only; the decoder ignores them.
The part of a `toml` tag after the first comma carries options. They shape
emission only; the decoder ignores them, so a value that round-trips keeps
its key whether the table it came from was written inline or under a header.
- `omitzero` skips the field when its value is the zero value of its type. A
type with an `IsZero() bool` method (time.Time among them) decides through
that method, so a zero `time.Time` or an all-zero struct disappears from
the output.
- `omitempty` skips the field when it holds an empty collection: a nil or
empty slice or array, or a nil or empty map. Strings and other scalars are
not covered by `omitempty`; use `omitzero` for those.
- `omitempty` skips the field when it holds an empty value in the
encoding/json sense: an empty string, a zero number, `false`, a nil pointer
or interface, and a nil or empty slice, array or map. This is a change of
semantics against 1.x, where only collections were covered.
- `inline` forces a struct or map field to emit as `name = {…}`, the inline
table form, instead of a header section, whatever its size; a named
embedded struct tagged this way does the same. A field holding an array of
tables is an error under `inline`, because the inline form would re-parse
as a value array and change the value's Go type.
- `comment=text` carries a comment for the field, which
`EmitFieldComments(true)` prints above the field's line or
header, each line of a multi-line text with its own `# ` marker. Go doc
comments are not visible to reflection, so the tag is the channel that
carries the text; without the encoder option the tag is ignored.
```go
type Config struct {
Host string `toml:"host,omitzero"`
Started time.Time `toml:"started,omitzero"`
Tags []string `toml:"tags,omitempty"`
Retry Retry `toml:"retry,inline"`
}
```
@@ -402,11 +646,11 @@ parsed as keys of the sub-table.
### Preserving declaration order
`GroupByKind(false)` on an `Encoder` walks the entries in declaration order
`Layout(LayoutKindDeclaration)` walks the entries in declaration order
instead, emitting each header immediately before its content:
```go
out, err := interpres.NewEncoder().GroupByKind(false).Marshal(cfg)
out, err := interpres.Marshal(cfg, interpres.Layout(interpres.LayoutKindDeclaration))
```
The output remains parseable, but a scalar declared after a sub-table lands
@@ -491,18 +735,18 @@ across a round-trip.
A nil slice is always omitted. An empty (length 0) array of tables is always
omitted, because TOML forbids an empty `[[a]]`. Other empty arrays emit as
`key = []` by default; `OmitEmptyArrays()` skips them as well, so
`key = []` by default; `OmitEmptyArrays(true)` skips them as well, so
`[]string{}` is treated like a nil slice.
### Long strings
By default every string is emitted as a basic `"..."` string with the escapes
TOML requires, a newline among them as `\n`. `UseLiteralMultiline(threshold)`
TOML requires, a newline among them as `\n`. `LiteralMultiline(threshold)`
switches strings that contain a newline and are at least `threshold` bytes long
to the literal `'''...'''` form, which carries the newlines verbatim:
```go
out, err := interpres.NewEncoder().UseLiteralMultiline(80).Marshal(cfg)
out, err := interpres.Marshal(cfg, interpres.LiteralMultiline(80))
```
Single-line strings keep the basic form regardless of the threshold, and a
@@ -536,7 +780,7 @@ single-line rendering is at most `threshold` bytes, and as a table header
section when it is longer. A document of small tables therefore grows shorter:
```go
out, err := interpres.NewEncoder().InlineTables(60).Marshal(cfg)
out, err := interpres.Marshal(cfg, interpres.InlineTables(60))
```
With `60` and a table of three short entries, the same value is written
@@ -555,7 +799,7 @@ header is not read back as part of that header's section.
### Cancellation
`MarshalContext` and `(*Encoder).MarshalContext` accept a `context.Context`. The
`MarshalContext` accepts a `context.Context`. The
context is checked before any work and every 64 fields during the reflection
walk.
@@ -592,55 +836,118 @@ sequenceDiagram
Marshal-->>Caller: bytes, error
```
## Coming from encoding/json and encoding/json/v2
The API follows the shapes encoding/json made familiar and the option style
encoding/json/v2 made current, with the differences TOML asks for:
| encoding/json or encoding/json/v2 | interpres | Notes |
|---|---|---|
| `json.Unmarshal(data, v)` | `Unmarshal(data, v)` | the same shape; the value mapping is TOML's |
| `json.Marshal(v)` | `Marshal(v)` | the same shape; the output is TOML 1.1 |
| `json.MarshalAppend(buf, v)` | `MarshalAppend(buf, v)` | the same shape, options included |
| `json.MarshalWrite(w, v)` | `MarshalWrite(w, v)` | the same shape, options included |
| `json.UnmarshalRead(r, v)` | `UnmarshalRead(r, v)` | the same shape, options included |
| `json/v2 RejectUnknownMembers` | `RejectUnknownFields(true)` | the same effect under TOML vocabulary |
| `(*json.Decoder).DisallowUnknownFields` | `RejectUnknownFields(true)` | the variadic option replaces the stateful decoder |
| `json.Number`, `StringifyNumbers` | `Number`, `NumbersAsLiterals(true)` | the TOML literal carries its radix and separators, so `0x1f` stays `0x1f` |
| `json/v2 MarshalOptions` fields | `MarshalOption` values | `Layout`, `OmitEmptyArrays`, `LiteralMultiline`, `InlineTables`, `EmitFieldComments` |
| `json/v2 JoinOptions` | listing | options compose by listing them in the call |
| `json.MarshalIndent` | none | TOML is the presentation format; the `-json` mode of interpres-decode prints plain JSON |
| tag `json:"name,omitempty"` | tag `toml:"name,omitempty"` | the empty-value rules match encoding/json as of 2.0 |
| tag `json:"name,omitzero"` | tag `toml:"name,omitzero"` | the same, `IsZero()` honoured |
| tag `json:"name,inline"` (v2) | tag `toml:"name,inline"` | forces the inline table form on encode |
| `json/v2 Marshalers` | `Marshaler` (`MarshalTOML`) | the TOML method returns a value the encoder renders, not bytes |
| `json/v2 Unmarshalers` | `Unmarshaler` (`UnmarshalTOML`) | the data arrives decoded, not as bytes |
| `encoding.TextMarshaler`, `TextUnmarshaler` | honoured, the same | a type that renders itself as text becomes a TOML string, both ways |
| `*json.UnmarshalTypeError` | `*DecodeError` | the path is segments with a `String()` renderer, not a dotted string |
| `*json.SyntaxError` | `*SyntaxError` | the TOML error adds the byte `Offset` and the `Column` to the line |
| context support | `*Context` variants of the parse, decode and marshal entries | encoding/json has none |
## Types
### `type SyntaxError struct{ Line int; Msg string }`
### `type SyntaxError struct{ Line, Offset, Column int; Msg string }`
Describes a document the parser rejected, with the 1-based `Line` at which it
gave up and `Error()` rendering as `interpres: line N: msg`. A malformed
document is the usual cause; the nesting limit and an input that is not valid
UTF-8 report through the same type. Read the structured fields with a type
assertion or `errors.AsType`:
Describes a document the parser rejected: the 1-based `Line` at which it gave
up, the `Offset` in bytes the scan stopped at, the 1-based `Column` on that
line, and `Error()` rendering as `interpres: line N: msg`. A malformed document
is the usual cause; the nesting limit and an input that is not valid UTF-8
report through the same type, with the UTF-8 message naming the offset of the
first invalid byte. Read the structured fields with a type assertion or
`errors.AsType`:
```go
if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
fmt.Println(se.Line, se.Msg)
fmt.Println(se.Line, se.Offset, se.Column, se.Msg)
fmt.Println(se.SourceLine(data)) // the line, with a caret under Offset
}
```
### `type DecodeError struct{ Path []string; Err error }`
`SourceLine(src)` renders the source line the error points at from `src`,
followed by a caret line marking the column, for messages the reader sees
under the input.
### `type DecodeError struct{ Path Path; Err error }`
Wraps a decoding failure with the key path at which it happened. `Path` lists
one segment per level from the document root, the outermost key first: a key
contributes its name, an array element its bracketed index, so the path of the
`weight` field in the first item reads `["items", "[0]", "weight"]`. The
rendered message is unchanged by the type; read the fields instead of parsing
the message:
`weight` field in the first item reads `["items", "[0]", "weight"]` and its
`String()` renders `items[0].weight`. Read the fields instead of parsing the
message:
```go
if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
fmt.Println(de.Path, de.Err)
fmt.Println(de.Path.String(), de.Err)
}
```
### `type EncodeError struct{ Path string; Err error }`
### `type EncodeError struct{ Path Path; Err error }`
Wraps an encoding failure with the key path of the value that failed, in the
document's own notation: `server.ports[2]`. Read it with `errors.AsType` the
same way.
Wraps an encoding failure with the key path of the value that failed, the
same `Path` type the decode error carries, so `server.ports[2]` reads the
same on both sides. Read it with `errors.AsType` the same way.
### `type Decoder`
### `type Path []string`
Configurable strictness for decoding, constructed with `NewDecoder`. Set up
with `DisallowUnknownFields`, then call `Decode` or `DecodeContext` any number
of times. A configured `Decoder` holds no per-call state and is safe for
concurrent use.
The path both error wrappers carry, one segment per level from the document
root. `String()` renders the TOML notation: keys join with dots, an index
attaches to the previous segment in brackets, `items[0].weight`.
| Method | Default | Effect |
### Options
The decode and encode entries take variadic options, the shape
encoding/json/v2 uses for its own. Each option is a stateless function value
over the private settings of one call; they compose by listing in the call,
and there is no stateful Decoder or Encoder to share or guard.
Decode options:
| Option | Default | Effect |
|---|---|---|
| `DisallowUnknownFields()` | off | a key with no matching struct field is an error |
| `MaxDepth(depth int)` | `10000` | bound how deeply arrays and inline tables may nest |
| `RejectUnknownFields(v bool)` | off | a key with no matching struct field is an error |
| `NumbersAsLiterals(v bool)` | off | integers and floats decode into `Number`, which carries the literal; see [Numbers as literals](#numbers-as-literals) |
| `MaxNestingDepth(depth int)` | `10000` | bound how deeply arrays and inline tables may nest |
| `MaxInputSize(size int)` | no limit | bound the size of the document, in bytes |
| `LocalTimeLocation(loc)` | nil | the zone a local date-time is carried in when it decodes into a `time.Time` |
Encode options:
| Option | Default | Effect |
|---|---|---|
| `Layout(kind LayoutKind)` | `LayoutKindGrouped` | group entries as scalars, then sub-tables, then arrays of tables; `LayoutKindDeclaration` preserves declaration order |
| `OmitEmptyArrays(v bool)` | off | skip `key = []` for empty scalar arrays |
| `LiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
| `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header |
```go
out, err := interpres.MarshalContext(ctx, cfg,
interpres.Layout(interpres.LayoutKindDeclaration),
interpres.OmitEmptyArrays(true),
interpres.LiteralMultiline(80),
interpres.InlineTables(60))
```
The nesting limit protects the stack, because the parser is a recursive
descent: a deeper document is rejected with a `SyntaxError` naming the limit
@@ -649,36 +956,12 @@ default but take no options. The size limit is off by default, because the
caller already holds the bytes and the size is therefore a policy, not a
protection the library can impose on its own.
### `type Encoder`
Configurable emission policy, constructed with `NewEncoder`. The option state
is private; set it with the chainable methods, each of which returns the
encoder:
| Method | Default | Effect |
|---|---|---|
| `GroupByKind(v bool)` | `true` | group entries as scalars, then sub-tables, then arrays of tables; `false` preserves declaration order |
| `OmitEmptyArrays()` | off | skip `key = []` for empty scalar arrays |
| `UseLiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
```go
out, err := interpres.NewEncoder().
GroupByKind(false).
OmitEmptyArrays().
UseLiteralMultiline(80).
InlineTables(60).
MarshalContext(ctx, cfg)
```
A configured `Encoder` holds no per-call state; each `Marshal` or
`MarshalContext` call copies the options and is safe for concurrent use, as
long as no setter races with a call.
### `type Document`, `type Table`, `type Entry`
See [Documents](#documents). A `Document` is what `Parse` returns, and it is
not a value `Marshal` accepts.
See [Documents](#documents). A `Document` is what `Parse` returns, and
`Marshal` writes it back: the keys in written order, the comments in place,
the inline tables inline. `UnmarshalDocument(doc, v)` decodes it without
parsing again.
### `type Marshaler interface{ MarshalTOML() (any, error) }`
@@ -686,7 +969,20 @@ See [Custom encoding](#custom-encoding-marshaler).
### `type Unmarshaler interface{ UnmarshalTOML(data any) error }`
See [Custom decoding](#custom-decoding-unmarshaler).
See [Custom decoding](#custom-decoding-unmarshaler). `UnmarshalerContext`
carries the decode's context through `UnmarshalTOMLContext(ctx, data)` and
wins when a type implements both.
### `type Number string`
The literal a number was written with, what `NumbersAsLiterals` decodes into and what
`Marshal` writes back as it is. See
[Numbers as literals](#numbers-as-literals).
### `type OrderedMap`
The string-keyed table that keeps its key order on both the encode and the
decode side. See [Ordered tables](#ordered-tables).
### Date-time wrappers
+30 -16
View File
@@ -5,16 +5,17 @@ source tree; nothing is aspirational.
## Overview
interpres is one public library package, one command, and one example. The
interpres is one public library package, one command, and two examples. The
library implements the whole of TOML 1.1, decoding and encoding, in the
standard library alone; the command wraps the parser and the encoder for the
toml-test compliance harness, against which it stands at 214 valid, 467 invalid
and 214 encoder cases with zero failures; the example demonstrates the API.
and 214 encoder cases with zero failures; the examples demonstrate the API: one the document round trip, one the statement iterator.
```mermaid
flowchart TD
CLI[cmd/interpres-decode<br/>toml-test adapter] --> API
EX[examples/basic<br/>usage demo] --> API
EX2[examples/statements<br/>statement iterator demo] --> API
subgraph Lib [package interpres]
API[interpres.go<br/>public API and types]
API --> P[parser.go<br/>recursive-descent parser]
@@ -35,9 +36,10 @@ strict validation.
| Path | Responsibility |
|---|---|
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, `Decoder`, `Encoder`, `Marshaler`, `Unmarshaler`, `SyntaxError`, the local date-time types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `-encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, the option constructors, `Marshaler`, `Unmarshaler`, `SyntaxError`, the error and option types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `--encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
| `examples/basic` | A runnable tour of the API. Documentation in executable form, not part of the library. |
| `examples/statements` | The `Statements` iterator over a document, the shape a configuration tool reads. Documentation in executable form. |
Inside the library package, one file owns one concern:
@@ -46,18 +48,28 @@ Inside the library package, one file owns one concern:
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree, records the nodes a [Document](API.md#documents) is built from, and enforces the structural rules of TOML 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
| `document.go` | The parsed-document types: `Document`, `Table` and `Entry`, which carry the key order, whether a table was written inline, and the comments. The values they expose are the parser's own tree, not a copy. |
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `datetime.go` | The four date-time types (`OffsetDateTime` and the three local wrappers) and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `orderedmap.go` | `OrderedMap`, the table that keeps its key order, and the node index the decoder reads the written order from. |
| `target.go` | The targeted parse: the struct skeleton resolved against the document while it scans, no intermediate tree. Falls back to the tree path for every shape it does not model. |
| `docwrite.go` | The write side of the document pipeline: `UnmarshalDocument` and the writer that renders a `Document` back with its order and comments. |
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
| `encode.go` | The reverse walk: builds an intermediate `tomlDoc` per table (which is what preserves declaration order and enables the group-by-kind partition) and then emits it as TOML. |
The boundary that matters: `parser.go` produces only untyped trees
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); `decode.go` and
`encode.go` are the only files that touch `reflect`; the command never touches
either, it consumes `Parse` alone.
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); the reflection work
lives in `decode.go`, `encode.go`, `target.go` and `orderedmap.go`; the command
consumes `ParseMap`, `Parse` and `Marshal`, and owns no parsing or emission
logic of its own.
## Data flow
Decoding is parse, then one reflection walk. `SyntaxError` values are produced
Decoding has two paths. The direct one parses straight into a struct
destination: `target.go` resolves the table skeleton against the struct
schema while the document scans, and values assign through the ordinary
decoder rules, so no intermediate tree exists; that is the hot path every
`Unmarshal` into a struct takes. A document or destination the direct
skeleton cannot model falls back to the tree path: parse the whole document,
then one reflection walk over the tree. `SyntaxError` values are produced
inside `parser.go` and returned as-is; conversion errors are produced inside
`decode.go` and wrapped with the key path as they unwind.
@@ -66,10 +78,13 @@ sequenceDiagram
participant Caller
participant API as interpres.go
participant P as parser.go
participant T as target.go
participant D as decode.go
Caller->>API: Unmarshal(data, v)
API->>P: ParseContext(ctx, data)
P->>P: number and datetime atoms
API->>T: targeted parse into the struct
T->>P: scanner, grammar, atoms
T-->>API: result, error or fallback
API->>P: on fallback, ParseContext(ctx, data)
P-->>API: map tree or *SyntaxError
API->>D: decode(tree, reflect value)
D-->>API: nil or wrapped field error
@@ -77,7 +92,7 @@ sequenceDiagram
```
Encoding walks the other way. `encode.go` first builds a `tomlDoc` from the
value, then emits it; the two phases are why `GroupByKind` can reorder entries
value, then emits it; the two phases are why `Layout` can reorder entries
without a second reflection pass, and why cancellation is checked during both.
```mermaid
@@ -96,10 +111,9 @@ sequenceDiagram
## State and lifetime
- The exported `Decoder` and `Encoder` hold configuration only. Every
`Decode`, `DecodeContext`, `Marshal` and `MarshalContext` call allocates its
own unexported worker, so a configured type is safe for concurrent use; the
setter methods are not, and must finish before the value is shared.
- The option values are stateless: every `Unmarshal`, `Marshal` and their
variants apply their own options into a per-call unexported worker, so the
entries are safe for concurrent use.
- The parser is allocated per `ParseContext` call; the parser itself caches
nothing between documents.
- The shared state is a set of caches and pools whose entries are immutable
+4 -2
View File
@@ -11,10 +11,12 @@ The benchmarks live in `bench_test.go`, next to the code they measure:
|---|---|
| `BenchmarkParse` | `ParseMap` over a representative configuration document |
| `BenchmarkMarshal` | `Marshal` of the tree `ParseMap` produced from the same document |
| `BenchmarkStrictDecode` | `Decode` into a struct under `DisallowUnknownFields` |
| `BenchmarkStrictDecode` | `Unmarshal` into a struct under `RejectUnknownFields` (the targeted parse) |
| `BenchmarkParseLong` | `ParseMap` over a generated document with about 2000 array-of-tables entries |
| `BenchmarkStrictDecodeLong` | `Decode` into a typed document under `DisallowUnknownFields`, over the same long document |
| `BenchmarkStrictDecodeLong` | `Unmarshal` into a typed document under `RejectUnknownFields`, over the same long document |
| `BenchmarkMarshalLong` | `Marshal` of the tree `ParseMap` produced from the long document |
| `BenchmarkStrictDecodeTree` | the tree-path reference decode of the representative document: parse, then the reflection walk |
| `BenchmarkStrictDecodeTreeLong` | the tree-path reference decode of the long document, the A/B baseline of the targeted parse |
## Running
+105 -23
View File
@@ -3,6 +3,7 @@
The reference below is taken from the program itself. `interpres-decode` is
the toml-test harness adapter in both directions, decoding TOML into tagged
JSON and encoding tagged JSON back into TOML, and it also validates documents.
The same reference ships as the manual page `man/interpres-decode.1`.
Install it with Go itself, no release assets involved:
```sh
@@ -13,41 +14,74 @@ go install sourcedock.dev/petrbalvin/interpres/v2/cmd/interpres-decode@latest
```sh
interpres-decode [flags]
interpres-decode -encode
interpres-decode -validate [file ...]
interpres-decode --encode
interpres-decode --validate [file ...]
interpres-decode --validate [directory ...]
interpres-decode --json
interpres-decode --struct
interpres-decode --schema TYPE file.go
interpres-decode --version
```
Without `-validate` or `-encode` the program is the decoding half of the
toml-test adapter: it takes no arguments, reads one TOML document from stdin,
and writes the toml-test tagged-JSON form to stdout. Build it locally with
`just build`, which compiles it into `bin/interpres-decode`, or run it
straight from the module directory with `just run`.
Without `--validate`, `--encode`, `--json`, `--struct` or `--schema` the
program is the decoding half of the toml-test adapter: it takes no arguments,
reads one TOML document from stdin, and writes the toml-test tagged-JSON form
to stdout. Build it locally with `just build`, which compiles it into
`bin/interpres-decode`, or run it straight from the module directory with
`just run`.
With `-encode` the direction is reversed: the program reads a tagged-JSON
With `--encode` the direction is reversed: the program reads a tagged-JSON
description from stdin and writes the TOML document it describes to stdout,
which is the shape toml-test expects of an encoder command. It takes no
arguments either, and `-validate` and `-encode` cannot be combined.
arguments either, and the mode flags cannot be combined.
With `-validate` the program parses each named file instead, or stdin when no
With `--validate` the program parses each named file instead, or stdin when no
file is named, and prints one line per invalid document to stderr. It is
quiet on valid documents, which is the shape a CI step wants. The `-` name
means stdin.
means stdin. A named directory is walked for `.toml` files, every one of them
validated, and the walk closes with a summary on stderr naming how many
documents were checked and how many were invalid.
With `--json` the decoding half prints plain indented JSON instead of the
tagged form, the shape for people and diffs: the values keep their types as
JSON sees them, and the date-time wrappers print in their TOML form. The flag
shapes the decoding output only, so it is rejected together with the mode
flags.
With `--struct` the program reads a TOML document from stdin and prints a Go
struct definition shaped like it: one field per key in written order, nested
tables as nested struct types, and an array of tables as a slice. The
printed type compiles and decodes the document it came from.
With `--schema` the program reads a Go source file and writes a TOML template
for the named struct type: one key per exported field, the `comment=` tag
option printed as a comment above it, and the `default=` option as the value
where one is set. It is the inverse of `--struct`, for config-driven
applications that generate their example configuration from the type.
`--version` prints the binary's version and exits. The release pipeline builds
at the tag, so a released binary prints its own tag; a build from a working
tree prints `(devel)`.
## Flags
| Flag | Effect |
|---|---|
| `-validate` | validate the documents instead of emitting tagged JSON |
| `-encode` | read tagged JSON from stdin and write TOML instead |
| `-h` | print the usage |
| `--validate` | validate the documents instead of emitting tagged JSON; directories are walked for `.toml` files |
| `--encode` | read tagged JSON from stdin and write TOML instead |
| `--json` | with the default mode, print plain indented JSON instead of tagged JSON |
| `--struct` | infer a Go struct definition from the document on stdin and print it |
| `--schema TYPE` | write a TOML template for the struct type TYPE from the Go source file named as the first argument |
| `--version` | print the version and exit |
| `--help` | print the usage |
## Exit codes
| Code | Meaning |
|---|---|
| `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed |
| `1` | adapter: parse error; validate: at least one document is invalid |
| `2` | a usage error, a read failure, malformed tagged JSON, or a value with no TOML representation |
| `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed; schema, struct, version: the output was written |
| `1` | adapter: parse error; validate: at least one document is invalid; struct: the document on stdin failed to parse |
| `2` | a usage error, a read or write failure, malformed tagged JSON, or a value with no TOML representation |
## Wire format
@@ -74,7 +108,7 @@ wrapped in an object with a `type` and a `value`:
| local date | `date-local` | `1979-05-27` |
| local time | `time-local` | `07:32:00.999999` |
The `-encode` mode reads exactly this form back. Two properties of it are
The `--encode` mode reads exactly this form back. Two properties of it are
worth knowing. A float whose value has no fraction and no exponent is written
as a bare integer string, `{"type": "float", "value": "1"}`, so there the tag
decides the type and not the literal. And the form cannot tell an array of
@@ -95,10 +129,10 @@ port = 9090
```
The output is the equivalent value tree as one JSON object. Turn a description
back into TOML with `-encode`:
back into TOML with `--encode`:
```sh
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode -encode
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode --encode
```
```toml
@@ -108,18 +142,66 @@ title = "hello"
Validate the TOML files of another repository in CI:
```sh
interpres-decode -validate config.toml deploy/example.toml
interpres-decode --validate config.toml deploy/example.toml
```
An invalid document reports the file and the library's line number:
```sh
$ interpres-decode -validate bad.toml
bad.toml: interpres: line 1: expected a value
$ interpres-decode --validate bad.toml
interpres-decode: bad.toml: interpres: line 1: expected a value
$ echo $?
1
```
Sweep a whole directory tree of configuration, with the summary the walk
closes on:
```sh
$ interpres-decode --validate configs/
interpres-decode: configs/old.toml: interpres: line 3: duplicate key "port"
checked 14 documents, 1 invalid
$ echo $?
1
```
See the document a `--struct` template would decode:
```sh
echo 'host = "db"
port = 5432
' | ./bin/interpres-decode --struct
```
```go
// Generated by interpres-decode --struct; decode with
// sourcedock.dev/petrbalvin/interpres/v2.
type inferred struct {
Host string `toml:"host"`
Port int64 `toml:"port"`
}
```
Write the template back from the type, comments and defaults included, where
the Go source declares fields tagged
`toml:"host,comment=The host to dial,default=example.org"`:
```sh
./bin/interpres-decode --schema Config config.go
```
```toml
# The host to dial
host = "example.org"
```
Print the binary's version:
```sh
$ ./bin/interpres-decode --version
interpres-decode v2.0.0
```
Run the official compliance suite against the adapter:
```sh
+4
View File
@@ -46,6 +46,9 @@ prints the same list.
| `just install` | builds, then copies the binary into `~/.local/bin` (`BINDIR` overrides) |
| `just uninstall` | removes the installed binary |
| `just clean` | removes `bin/` and `coverage.out` |
| `just cross` | cross-compile smoke of the library and the command for arm64, loong64, riscv64 and the browser and edge runtimes; a hand-run convenience, not a gate |
| `just release-check X.Y.Z` | the release pre-flight: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release |
| `just docs-drift` | compares the toml-test counts the documentation quotes with a live suite run |
## Running a single test
@@ -99,6 +102,7 @@ pipeline.
|---|---|---|
| `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite |
| `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally |
| `fuzz.yml` | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut |
## Releases
+270 -18
View File
@@ -3,6 +3,11 @@
package interpres
import (
"maps"
"slices"
)
// A Document is a parsed TOML document: the values, plus what the map shape
// cannot carry, which is the order the keys were written in, whether a table
// was written inline or under a header, and the comments.
@@ -20,19 +25,72 @@ type Document struct {
footer []string
}
// Root returns the document's root table.
func (d *Document) Root() *Table { return d.root }
// Root returns the document's root table. A nil document has no root.
func (d *Document) Root() *Table {
if d == nil {
return nil
}
return d.root
}
// Map returns the value tree, the shape ParseMap gives. It is the tree the
// document was parsed into, not a copy.
func (d *Document) Map() map[string]any { return d.root.values }
// document was parsed into, not a copy. A nil document or one with no root
// holds no values.
func (d *Document) Map() map[string]any {
if d == nil || d.root == nil {
return nil
}
return d.root.values
}
// Footer returns the comment lines that follow the last statement, and every
// line of a document that holds no statement at all.
func (d *Document) Footer() []string { return d.footer }
func (d *Document) Footer() []string {
if d == nil {
return nil
}
return d.footer
}
// SetFooter replaces those lines.
func (d *Document) SetFooter(lines []string) { d.footer = lines }
func (d *Document) SetFooter(lines []string) {
if d == nil {
return
}
d.footer = lines
}
// The document-level convenience forms of the Table edit API; they act on
// the root table.
// Get returns the root table's entry for key, and whether the document has
// one. See Table.Get.
func (d *Document) Get(key string) (*Entry, bool) { return d.Root().Get(key) }
// GetString returns the string the key holds, and whether it holds one.
func (d *Document) GetString(key string) (string, bool) { return d.Root().GetString(key) }
// GetInt returns the integer the key holds, and whether it holds one.
func (d *Document) GetInt(key string) (int64, bool) { return d.Root().GetInt(key) }
// GetFloat returns the float the key holds, and whether it holds one.
func (d *Document) GetFloat(key string) (float64, bool) { return d.Root().GetFloat(key) }
// GetBool returns the boolean the key holds, and whether it holds one.
func (d *Document) GetBool(key string) (bool, bool) { return d.Root().GetBool(key) }
// GetArray returns the value array the key holds, and whether it holds one.
func (d *Document) GetArray(key string) ([]any, bool) { return d.Root().GetArray(key) }
// GetTable returns the node of the table the key holds, and whether it holds
// one.
func (d *Document) GetTable(key string) (*Table, bool) { return d.Root().GetTable(key) }
// Set stores value under the key in the root table. See Table.Set.
func (d *Document) Set(key string, value any) { d.Root().Set(key, value) }
// Delete removes the key from the root table. See Table.Delete.
func (d *Document) Delete(key string) { d.Root().Delete(key) }
// A Table is one TOML table: its values, its keys in written order, and the
// comments around the header or the key that introduced it.
@@ -45,6 +103,11 @@ type Table struct {
// rather than under a header or as a dotted key.
inline bool
// dotted records that a dotted key introduced the table, `a.b = 1`
// building the a around the leaf: the write side gives such a table back
// as dotted key lines, the form that holds the position of a line.
dotted bool
// comments are the lines above the table's header, trailing is the comment
// on the header's own line. Both are empty for a table a dotted key
// introduced, which has no line of its own.
@@ -56,8 +119,12 @@ func newTable(values map[string]any) *Table {
return &Table{values: values, index: map[string]*Entry{}}
}
// Keys returns the table's keys in the order they were written.
// Keys returns the table's keys in the order they were written. A nil table
// holds none, the answer a document without a root gives through Root.
func (t *Table) Keys() []string {
if t == nil {
return nil
}
keys := make([]string, len(t.entries))
for i, e := range t.entries {
keys[i] = e.key
@@ -67,43 +134,75 @@ func (t *Table) Keys() []string {
// Values returns the table's values, which is the map the value tree holds for
// it.
func (t *Table) Values() map[string]any { return t.values }
func (t *Table) Values() map[string]any {
if t == nil {
return nil
}
return t.values
}
// Entries returns the table's entries in written order.
func (t *Table) Entries() []*Entry { return t.entries }
func (t *Table) Entries() []*Entry {
if t == nil {
return nil
}
return t.entries
}
// Get returns the entry for key, and whether the table has one.
func (t *Table) Get(key string) (*Entry, bool) {
if t == nil {
return nil, false
}
e, ok := t.index[key]
return e, ok
}
// Inline reports whether the table was written as an inline table, `{…}`,
// rather than under a header or introduced by a dotted key.
func (t *Table) Inline() bool { return t.inline }
func (t *Table) Inline() bool { return t != nil && t.inline }
// Comments returns the comment lines above the table's header, or above the
// key that introduced it. Lines carry no leading '#' and no surrounding space.
func (t *Table) Comments() []string { return t.comments }
func (t *Table) Comments() []string {
if t == nil {
return nil
}
return t.comments
}
// SetComments replaces those lines. Each line is written back with a "# " in
// front of it, so a line should not carry one.
func (t *Table) SetComments(lines []string) { t.comments = lines }
func (t *Table) SetComments(lines []string) {
if t == nil {
return
}
t.comments = lines
}
// Trailing returns the comment on the header's own line, without the '#'.
func (t *Table) Trailing() string { return t.trailing }
func (t *Table) Trailing() string {
if t == nil {
return ""
}
return t.trailing
}
// SetTrailing replaces that comment.
func (t *Table) SetTrailing(line string) { t.trailing = line }
func (t *Table) SetTrailing(line string) {
if t == nil {
return
}
t.trailing = line
}
// addValue records a key of the table, in written order.
// addValue records a key of the table, in written order. The caller gives
// the entry a table node or element nodes when the value has that shape; a
// map value left without a node writes as an inline table.
func (t *Table) addValue(key string, val any, inline bool) *Entry {
e := &Entry{table: t, key: key, inline: inline}
t.entries = append(t.entries, e)
t.index[key] = e
if node, ok := val.(map[string]any); ok {
e.child = newTable(node)
}
return e
}
@@ -136,6 +235,9 @@ func (t *Table) addElement(key string, values map[string]any) *Table {
// child returns the node of a table-valued key, or nil.
func (t *Table) child(key string) *Table {
if t == nil {
return nil
}
if e, ok := t.index[key]; ok {
return e.child
}
@@ -194,3 +296,153 @@ func (e *Entry) Trailing() string { return e.trailing }
// SetTrailing replaces that comment.
func (e *Entry) SetTrailing(line string) { e.trailing = line }
// GetString returns the string the key holds, and whether it holds one.
func (t *Table) GetString(key string) (string, bool) {
if t == nil {
return "", false
}
v, ok := t.values[key]
s, ok := v.(string)
return s, ok
}
// GetInt returns the integer the key holds, and whether it holds one.
func (t *Table) GetInt(key string) (int64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key]
i, ok := v.(int64)
return i, ok
}
// GetFloat returns the float the key holds, and whether it holds one.
func (t *Table) GetFloat(key string) (float64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key]
f, ok := v.(float64)
return f, ok
}
// GetBool returns the boolean the key holds, and whether it holds one.
func (t *Table) GetBool(key string) (bool, bool) {
if t == nil {
return false, false
}
v, ok := t.values[key]
b, ok := v.(bool)
return b, ok
}
// GetArray returns the value array the key holds, and whether it holds one.
func (t *Table) GetArray(key string) ([]any, bool) {
if t == nil {
return nil, false
}
v, ok := t.values[key]
a, ok := v.([]any)
return a, ok
}
// GetTable returns the node of the table the key holds, and whether it holds
// one, whichever way the document wrote the table.
func (t *Table) GetTable(key string) (*Table, bool) {
c := t.child(key)
return c, c != nil
}
// Set stores value under key. A key the table already has keeps its position
// and its comments; a new one joins the end. A value of map[string]any
// becomes a table node of its own, written under a header like any other
// table, and replaces the node the key held, which belonged to the value the
// key held; a Go map carries no order, so its keys take sorted order. A value
// of []map[string]any becomes an array-of-tables node.
func (t *Table) Set(key string, value any) {
if t == nil {
return
}
e, ok := t.index[key]
if !ok {
t.values[key] = value
e = t.addValue(key, value, false)
if m, isMap := value.(map[string]any); isMap {
e.child = newOrderedTable(m)
}
if items, isArray := value.([]map[string]any); isArray {
e.elements = make([]*Table, len(items))
for i, item := range items {
e.elements[i] = newOrderedTable(item)
}
}
return
}
t.values[key] = value
switch v := value.(type) {
case map[string]any:
// The node is rebuilt rather than patched: the entries and the index
// belong to the table the key held, and writing the new value
// through them would leave the old table's keys in the output.
e.child = newOrderedTable(v)
e.elements = nil
case []map[string]any:
e.child = nil
e.elements = make([]*Table, len(v))
for i, item := range v {
e.elements[i] = newOrderedTable(item)
}
default:
e.child = nil
e.elements = nil
}
}
// newOrderedTable builds a table node for a value the caller set, its keys
// entered in sorted order, the order Marshal writes maps in.
func newOrderedTable(m map[string]any) *Table {
return orderedTable(m, 0)
}
// orderedTable is newOrderedTable's recursion. The depth bound is the value
// encoder's: a cyclic map stopped here is written by the value writer, which
// reports it instead of running the stack out.
func orderedTable(m map[string]any, depth int) *Table {
t := newTable(m)
for _, k := range slices.Sorted(maps.Keys(m)) {
v := m[k]
e := t.addValue(k, v, false)
if depth >= maxEncodeDepth {
continue
}
switch val := v.(type) {
case map[string]any:
e.child = orderedTable(val, depth+1)
case []map[string]any:
e.elements = make([]*Table, len(val))
for i, item := range val {
e.elements[i] = orderedTable(item, depth+1)
}
}
}
return t
}
// Delete removes key and everything it holds.
func (t *Table) Delete(key string) {
if t == nil {
return
}
if _, ok := t.values[key]; !ok {
return
}
delete(t.values, key)
delete(t.index, key)
for i, e := range t.entries {
if e.key == key {
t.entries = append(t.entries[:i], t.entries[i+1:]...)
break
}
}
}
+261 -16
View File
@@ -4,6 +4,7 @@
package interpres
import (
"reflect"
"slices"
"strings"
"testing"
@@ -288,27 +289,271 @@ func TestParseMapIsTheValueTree(t *testing.T) {
}
}
func TestMarshalRejectsDocument(t *testing.T) {
// A Document is not a value to marshal: its order and comments would be
// dropped, and a struct walk would silently write nothing at all.
doc, err := Parse([]byte("a = 1\n"))
func TestMarshalDocument(t *testing.T) {
// A Document writes back: the keys in written order, the comments above
// the lines and headers they belonged to, and inline tables inline again.
doc, err := Parse([]byte("# leading\na = 1 # trailing\n\n[t]\nb = \"x\"\n\ninline = { n = 1 }\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
if _, err := Marshal(doc); err == nil {
t.Fatal("expected an error for a Document")
} else if !strings.Contains(err.Error(), "Map()") {
t.Errorf("err = %v, want it to point at Map()", err)
}
if _, err := Marshal(*doc); err == nil {
t.Fatal("expected an error for a Document value")
}
// The tree marshals, which is the way through.
out, err := Marshal(doc.Map())
out, err := Marshal(doc)
if err != nil {
t.Fatalf("marshal of the tree: %v", err)
t.Fatalf("marshal of a Document: %v", err)
}
if want := "a = 1\n"; string(out) != want {
want := "# leading\na = 1 # trailing\n\n[t]\nb = \"x\"\ninline = {n = 1}\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
// The written document parses back to the same values.
re, err := Parse(out)
if err != nil {
t.Fatalf("re-parse: %v", err)
}
if got := re.Map()["a"]; got != int64(1) {
t.Errorf("a = %#v", got)
}
if _, err := Marshal(*doc); err != nil {
t.Errorf("marshal of a Document value: %v", err)
}
}
func TestDocumentEditPipeline(t *testing.T) {
doc, err := Parse([]byte("host = \"db\"\nport = 5432\n\n# The cache section\ntimeout = 1.5\n"))
if err != nil {
t.Fatal(err)
}
t.Run("typed getters", func(t *testing.T) {
if s, ok := doc.GetString("host"); !ok || s != "db" {
t.Errorf("host = %q, %v", s, ok)
}
if i, ok := doc.GetInt("port"); !ok || i != 5432 {
t.Errorf("port = %d, %v", i, ok)
}
if f, ok := doc.GetFloat("timeout"); !ok || f != 1.5 {
t.Errorf("timeout = %g, %v", f, ok)
}
if _, ok := doc.GetBool("host"); ok {
t.Error("host claimed as bool")
}
})
t.Run("set keeps the position and the comments", func(t *testing.T) {
doc.Set("port", int64(9090))
if got := doc.Root().Keys(); !slices.Equal(got, []string{"host", "port", "timeout"}) {
t.Fatalf("keys = %v", got)
}
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(out), "port = 9090") {
t.Errorf("output missing the new value:\n%s", out)
}
})
t.Run("a new key joins the end", func(t *testing.T) {
doc.Set("lang", "cs")
if got := doc.Root().Keys(); !slices.Equal(got, []string{"host", "port", "timeout", "lang"}) {
t.Fatalf("keys = %v", got)
}
})
t.Run("a set table keeps an order of its own", func(t *testing.T) {
sub := map[string]any{"z": int64(1), "a": int64(2)}
doc.Set("cache", sub)
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(out), "[cache]\na = 2\nz = 1\n") {
t.Errorf("output missing the new table in order:\n%s", out)
}
})
t.Run("delete removes the key", func(t *testing.T) {
doc.Delete("lang")
if _, ok := doc.Get("lang"); ok {
t.Fatal("lang survived Delete")
}
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
if strings.Contains(string(out), "lang") {
t.Errorf("output still names lang:\n%s", out)
}
})
t.Run("UnmarshalDocument decodes without reparsing", func(t *testing.T) {
type Cfg struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
var cfg Cfg
if err := UnmarshalDocument(doc, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Host != "db" || cfg.Port != 9090 {
t.Errorf("decoded %+v", cfg)
}
})
t.Run("comments survive the round trip", func(t *testing.T) {
src := "# header comment\n[a]\n# key comment\nb = 2\n"
doc, err := Parse([]byte(src))
if err != nil {
t.Fatal(err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
for _, want := range []string{"# header comment", "[a]", "# key comment", "b = 2"} {
if !strings.Contains(string(out), want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
})
t.Run("a nil document refuses to decode", func(t *testing.T) {
var cfg struct {
A int `toml:"a"`
}
if err := UnmarshalDocument(nil, &cfg); err == nil {
t.Error("UnmarshalDocument(nil) succeeded, want an error")
}
})
}
// TestMarshalDocumentRoundTrips pins that a parsed document written back
// re-parses to the same tree: arrays of tables keep exactly one header per
// element, dotted keys hold their line position without swallowing the keys
// after them, inline tables inside value arrays keep their written order,
// and comments travel with their statements.
func TestMarshalDocumentRoundTrips(t *testing.T) {
tests := []struct {
name string
src string
}{
{"array of tables", "[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n"},
{"array of tables with comments", "# about items\n[[items]] # first\nname = \"a\"\n"},
{"dotted key before a later key", "a.b = 1\nc = 2\n"},
{"dotted keys grouped", "a.b = 1\na.c = 2\nd = 3\n"},
{"dotted key with a nested leaf", "a.b.c = 1\nz = 2\n"},
{"header section after a dotted key", "a.b = 1\n\n[a.x]\ny = 2\n"},
{"inline tables in a value array keep order", "arr = [{y = 1, x = 2}, {second = true, first = false}]\n"},
{"nested array of tables", "[[items]]\nn = 1\n\n[items.sub]\nk = \"v\"\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
doc, err := Parse([]byte(tt.src))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
reparsed, err := Parse(out)
if err != nil {
t.Fatalf("re-parse of %q: %v", out, err)
}
if !reflect.DeepEqual(doc.Map(), reparsed.Map()) {
t.Errorf("round trip changed the tree:\nin: %#v\nout: %#v", doc.Map(), reparsed.Map())
}
if got, want := reparsed.Root().Keys(), doc.Root().Keys(); !slices.Equal(got, want) {
t.Errorf("root keys = %v, want %v", got, want)
}
})
}
}
// TestMarshalDocumentArrayComments pins where the comments of an array of
// tables land: above and beside the [[header]] itself.
func TestMarshalDocumentArrayComments(t *testing.T) {
doc, err := Parse([]byte("# element one\n[[items]] # trailing\nname = \"a\"\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "# element one\n[[items]] # trailing\nname = \"a\"\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetReplacesTableNode pins that Set over a key holding a table
// rebuilds the node, so the new map's keys are the ones written.
func TestTableSetReplacesTableNode(t *testing.T) {
doc, err := Parse([]byte("[cache]\nz = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("cache", map[string]any{"a": int64(2)})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "[cache]\na = 2\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetNestedArraysOfTables pins that a value set through the edit API
// carries its arrays of tables into the header form.
func TestTableSetNestedArraysOfTables(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("t", map[string]any{"items": []map[string]any{{"n": int64(1)}, {"n": int64(2)}}})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if !strings.Contains(string(out), "[[t.items]]") {
t.Errorf("output = %q, want the array of tables under a header", out)
}
}
// TestTableSetCyclicMapErrors pins that a cyclic map set through the edit API
// reaches the depth limit instead of the stack.
func TestTableSetCyclicMapErrors(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
m := map[string]any{}
m["self"] = m
doc.Set("cyclic", m)
if _, err := Marshal(doc); err == nil || !strings.Contains(err.Error(), "nests deeper") {
t.Errorf("err = %v, want the depth-limit complaint", err)
}
}
// TestDocumentNilSafety pins that the nil document answers its readers
// instead of panicking, the contract Root already carries.
func TestDocumentNilSafety(t *testing.T) {
var doc *Document
if doc.Map() != nil {
t.Errorf("Map = %v", doc.Map())
}
if doc.Footer() != nil {
t.Errorf("Footer = %v", doc.Footer())
}
doc.SetFooter([]string{"x"})
if e, ok := doc.Get("k"); e != nil || ok {
t.Errorf("Get = %v, %v", e, ok)
}
if _, ok := doc.GetString("k"); ok {
t.Error("GetString on a nil document reports a value")
}
if _, ok := doc.GetTable("k"); ok {
t.Error("GetTable on a nil document reports a value")
}
doc.Set("k", 1)
doc.Delete("k")
if keys := doc.Root().Keys(); keys != nil {
t.Errorf("Keys = %v", keys)
}
if doc.Root().Entries() != nil {
t.Errorf("Entries = %v", doc.Root().Entries())
}
}
+314
View File
@@ -0,0 +1,314 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"fmt"
)
// UnmarshalDocument decodes a parsed Document into v without parsing again,
// the shape an edit pipeline wants: read the document, change the values it
// holds, decode the result into a typed destination. The key order and the
// comments the document carries are untouched; the decode reads the value
// tree the document shares with its nodes.
//
// UnmarshalDocument accepts the same destinations Unmarshal does.
func UnmarshalDocument(doc *Document, v any) error {
if doc == nil {
return fmt.Errorf("interpres: cannot decode a nil Document")
}
dec := newDecoder()
dec.nodes = indexNodes(doc.Root())
return dec.decode(doc.Map(), v)
}
// writeDocument renders a Document back to TOML: the keys in written order,
// the comments above the lines and headers they belonged to, tables that
// were written inline written inline again, and an array of tables in its
// header form. It is the write side of the edit pipeline: read with Parse,
// change with the Table and Document mutators, write with Marshal.
func (e *encoder) writeDocument(doc *Document) error {
if err := e.checkCtx(); err != nil {
return err
}
if doc == nil || doc.root == nil {
return fmt.Errorf("interpres: cannot marshal a nil Document")
}
if err := e.writeTableEntries(doc.root, nil); err != nil {
return err
}
e.writeDocumentFooter(doc.footer)
return nil
}
// writeDocumentFooter writes the comment lines that follow the last
// statement. The parser collects them wherever they sit after it, so the
// writer needs no blank line of its own to have them read back.
func (e *encoder) writeDocumentFooter(footer []string) {
for _, line := range footer {
e.buf.WriteString("# ")
e.buf.WriteString(line)
e.buf.WriteByte('\n')
}
}
// writeTableEntries writes one table at the given header path, nil for the
// document root, whose keys need no header: the blank line, the comments,
// the header line with its trailing comment, then the body.
func (e *encoder) writeTableEntries(t *Table, path []string) error {
if t == nil {
return nil
}
if path != nil {
e.writeBlankLine()
e.writeComments(t.Comments())
e.buf.WriteString("[")
if err := e.writeKeyPath(path); err != nil {
return err
}
e.buf.WriteString("]")
if tr := t.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
}
return e.writeTableBody(t, path)
}
// writeTableBody writes one table's entries: the value lines first, in
// written order, then the header sections. In a valid document every line at
// one level precedes the headers below it, so the split reorders nothing;
// what it prevents is a table a dotted key introduced, which the parse nests
// as a sub-table at the position of a line, from swallowing the lines that
// follow it into its header.
func (e *encoder) writeTableBody(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if !e.isLineEntry(entry) {
continue
}
if err := e.writeLineEntry(entry, path); err != nil {
return err
}
}
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
// A dotted table writes as lines above; its own header-form
// sub-tables are sections the document placed after those lines,
// so the section pass reaches through the dotted entry.
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeDottedSections writes the header-form sub-tables of a dotted table:
// the sections the document placed after the dotted lines, reached through
// the dotted entry itself.
func (e *encoder) writeDottedSections(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeSectionEntry writes one entry the line pass left behind: a table or
// an array of tables under its header, at the path this level carries.
func (e *encoder) writeSectionEntry(entry *Entry, path []string) error {
if _, isTables := entry.Value().([]map[string]any); isTables {
// An array of tables keeps its header form, one element per header
// with the element's own comments above it; the body that follows is
// the element's, with no header of its own to repeat.
elemPath := append(append([]string{}, path...), entry.Key())
for i, el := range entry.Elements() {
e.writeBlankLine()
if i == 0 {
e.writeComments(entry.Comments())
}
e.writeComments(el.Comments())
e.buf.WriteString("[[")
if err := e.writeKeyPath(elemPath); err != nil {
return err
}
e.buf.WriteString("]]")
if tr := el.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
if err := e.writeTableBody(el, elemPath); err != nil {
return err
}
}
return nil
}
headerPath := append(append([]string{}, path...), entry.Key())
return e.writeTableEntries(entry.Table(), headerPath)
}
// isLineEntry reports whether an entry writes as one or more "key = value"
// lines at its own level: a value, an inline table, or a table a dotted key
// introduced, which goes back as dotted keys. An emptied array of tables
// counts as one only so the line pass can drop it, the omission the value
// encoder applies to an empty array of tables too.
func (e *encoder) isLineEntry(entry *Entry) bool {
if child := entry.Table(); child != nil {
return entry.Inline() || child.dotted
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return len(entry.Elements()) == 0
}
return true
}
// writeLineEntry writes one entry as lines at this level, and drops an
// emptied array of tables, which has no TOML form.
func (e *encoder) writeLineEntry(entry *Entry, path []string) error {
if child := entry.Table(); child != nil && !entry.Inline() {
return e.writeDottedTable(child, append(append([]string{}, path...), entry.Key()))
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return nil
}
return e.writeDocumentEntry(entry)
}
// writeDottedTable writes a table a dotted key introduced as one dotted line
// per leaf, in written order: `a.b = 1`. A sub-table the document added
// under a header stays a section and is left to the section pass.
func (e *encoder) writeDottedTable(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && !entry.Inline() && !child.dotted {
continue
}
leafPath := append(append([]string{}, path...), entry.Key())
if child := entry.Table(); child != nil && !entry.Inline() {
if err := e.writeDottedTable(child, leafPath); err != nil {
return err
}
continue
}
e.writeComments(entry.Comments())
if err := e.writeKeyPath(leafPath); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
}
return nil
}
// writeDocumentEntry writes one "key = value" line of a document, with the
// comments the key carried. A value that is itself an inline table renders
// inline from its node, in the written order.
func (e *encoder) writeDocumentEntry(entry *Entry) error {
e.writeComments(entry.Comments())
if err := e.writeKey(entry.Key()); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
return nil
}
// writeEntryValueNodes writes the value of a document entry. An inline table
// node keeps the written key order even inside a value array, where the
// ordinary value writer would sort the keys.
func (e *encoder) writeEntryValueNodes(entry *Entry) error {
if child := entry.Table(); child != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err
}
} else if arr, ok := entry.Value().([]any); ok {
elems := entry.Elements()
e.buf.WriteByte('[')
for i, item := range arr {
if i > 0 {
e.buf.WriteString(", ")
}
if i < len(elems) && elems[i] != nil {
if err := e.writeInlineTableNode(elems[i]); err != nil {
return err
}
continue
}
if err := e.writeValue(item); err != nil {
return err
}
}
e.buf.WriteByte(']')
} else if err := e.writeValue(entry.Value()); err != nil {
return err
}
if tr := entry.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
return nil
}
// writeInlineTableNode renders a table node as an inline table, its keys in
// written order, values that are tables inline in turn.
func (e *encoder) writeInlineTableNode(t *Table) error {
e.buf.WriteByte('{')
for i, key := range t.Keys() {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeKey(key); err != nil {
return err
}
e.buf.WriteString(" = ")
entry, _ := t.Get(key)
if child := entry.Table(); child != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err
}
continue
}
if err := e.writeValue(t.Values()[key]); err != nil {
return err
}
}
e.buf.WriteByte('}')
return nil
}
+613 -125
View File
File diff suppressed because it is too large Load Diff
+498 -48
View File
@@ -67,7 +67,7 @@ func TestMarshalFloatSpecials(t *testing.T) {
}
}
func TestMarshalFloatNormalizesNegativeZero(t *testing.T) {
func TestMarshalFloatNormalisesNegativeZero(t *testing.T) {
// The output contract normalises negative zero to "0.0".
type Cfg struct {
Z float64 `toml:"z"`
@@ -91,13 +91,10 @@ func TestMarshalContextHonoursCancellation(t *testing.T) {
if _, err := MarshalContext(ctx, C{A: 1}); !errors.Is(err, context.Canceled) {
t.Fatalf("MarshalContext returned %v, want context.Canceled", err)
}
if _, err := NewEncoder().MarshalContext(ctx, C{A: 1}); !errors.Is(err, context.Canceled) {
t.Fatalf("Encoder.MarshalContext returned %v, want context.Canceled", err)
}
}
func TestEncoderGroupByKindDefault(t *testing.T) {
// NewEncoder must default to GroupByKind=true so legacy callers keep the
func TestEncoderLayoutGroupedDefault(t *testing.T) {
// NewEncoder must default to LayoutKindGrouped so legacy callers keep the
// scalars-first ordering.
type Cfg struct {
Name string `toml:"name"`
@@ -105,7 +102,7 @@ func TestEncoderGroupByKindDefault(t *testing.T) {
Host string `toml:"host"`
} `toml:"s"`
}
out, err := NewEncoder().Marshal(Cfg{Name: "x", S: struct {
out, err := Marshal(Cfg{Name: "x", S: struct {
Host string `toml:"host"`
}{Host: "h"}})
if err != nil {
@@ -117,7 +114,7 @@ func TestEncoderGroupByKindDefault(t *testing.T) {
}
}
func TestEncoderGroupByKindFalsePreservesOrder(t *testing.T) {
func TestEncoderLayoutDeclarationPreservesOrder(t *testing.T) {
type Inner struct {
Host string `toml:"host"`
}
@@ -131,11 +128,11 @@ func TestEncoderGroupByKindFalsePreservesOrder(t *testing.T) {
Server: Inner{Host: "h"},
Debug: true,
}
out, err := NewEncoder().GroupByKind(false).Marshal(in)
out, err := Marshal(in, Layout(LayoutKindDeclaration))
if err != nil {
t.Fatalf("marshal: %v", err)
}
// With GroupByKind(false) the encoder walks entries in declaration order.
// With Layout(LayoutKindDeclaration) the encoder walks entries in declaration order.
// The output is still parseable, but a scalar that follows a header is
// parsed as a sub-table key. That is the user's trade-off; see
// docs/API.md.
@@ -145,8 +142,8 @@ func TestEncoderGroupByKindFalsePreservesOrder(t *testing.T) {
}
}
func TestEncoderGroupByKindTrueDefaultOrder(t *testing.T) {
// The default (GroupByKind=true) must lift the trailing scalar ahead of
func TestEncoderLayoutGroupedDefaultOrder(t *testing.T) {
// The default (LayoutKindGrouped) must lift the trailing scalar ahead of
// the [server] block so the document round-trips losslessly.
type Inner struct {
Host string `toml:"host"`
@@ -161,7 +158,7 @@ func TestEncoderGroupByKindTrueDefaultOrder(t *testing.T) {
Server: Inner{Host: "h"},
Debug: true,
}
out, err := NewEncoder().Marshal(in)
out, err := Marshal(in)
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -176,10 +173,10 @@ func TestEncoderOmitEmptyArrays(t *testing.T) {
Tags []string `toml:"tags"`
Secrets []string `toml:"secrets"`
}
out, err := NewEncoder().OmitEmptyArrays().Marshal(Cfg{
out, err := Marshal(Cfg{
Tags: []string{"a", "b"},
Secrets: []string{},
})
}, OmitEmptyArrays(true))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -193,7 +190,7 @@ func TestEncoderDefaultEmitsEmptyArray(t *testing.T) {
type Cfg struct {
Tags []string `toml:"tags"`
}
out, err := NewEncoder().Marshal(Cfg{Tags: []string{}})
out, err := Marshal(Cfg{Tags: []string{}})
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -211,10 +208,10 @@ func TestEncoderOmitEmptyArrayOfTablesStillSkipped(t *testing.T) {
Title string `toml:"title"`
Items []Item `toml:"items"`
}
out, err := NewEncoder().OmitEmptyArrays().Marshal(Cfg{
out, err := Marshal(Cfg{
Title: "demo",
Items: nil,
})
}, OmitEmptyArrays(true))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -224,12 +221,12 @@ func TestEncoderOmitEmptyArrayOfTablesStillSkipped(t *testing.T) {
}
}
func TestEncoderUseLiteralMultiline(t *testing.T) {
func TestEncoderLiteralMultiline(t *testing.T) {
type Cfg struct {
Long string `toml:"long"`
}
long := strings.Repeat("a", 50) + "\nline two\nline three"
out, err := NewEncoder().UseLiteralMultiline(20).Marshal(Cfg{Long: long})
out, err := Marshal(Cfg{Long: long}, LiteralMultiline(20))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -239,12 +236,12 @@ func TestEncoderUseLiteralMultiline(t *testing.T) {
}
}
func TestEncoderUseLiteralMultilineBelowThreshold(t *testing.T) {
func TestEncoderLiteralMultilineBelowThreshold(t *testing.T) {
// A multi-line value shorter than the threshold must remain escaped.
type Cfg struct {
Short string `toml:"short"`
}
out, err := NewEncoder().UseLiteralMultiline(1000).Marshal(Cfg{Short: "one\ntwo"})
out, err := Marshal(Cfg{Short: "one\ntwo"}, LiteralMultiline(1000))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -254,12 +251,12 @@ func TestEncoderUseLiteralMultilineBelowThreshold(t *testing.T) {
}
}
func TestEncoderUseLiteralMultilineThresholdZero(t *testing.T) {
// UseLiteralMultiline(0) disables the literal form entirely.
func TestEncoderLiteralMultilineThresholdZero(t *testing.T) {
// LiteralMultiline(0) disables the literal form entirely.
type Cfg struct {
S string `toml:"s"`
}
out, err := NewEncoder().UseLiteralMultiline(0).Marshal(Cfg{S: "a\nb\nc\nd"})
out, err := Marshal(Cfg{S: "a\nb\nc\nd"}, LiteralMultiline(0))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -282,7 +279,7 @@ func TestEncoderLiteralMultilineFallsBackWhenUnsafe(t *testing.T) {
{"lone carriage return", "first\rsecond\nthird"},
}
for _, c := range cases {
out, err := NewEncoder().UseLiteralMultiline(5).Marshal(map[string]any{"s": c.in})
out, err := Marshal(map[string]any{"s": c.in}, LiteralMultiline(5))
if err != nil {
t.Fatalf("%s: marshal: %v", c.name, err)
}
@@ -402,8 +399,8 @@ func TestMarshalRejectsNilMarshalerResult(t *testing.T) {
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "f" {
t.Fatalf("Path = %q, want %q", ee.Path, "f")
if ee.Path.String() != "f" {
t.Fatalf("Path = %v, want f", ee.Path)
}
// Inside a value array the nil result used to reach reflection as a zero
@@ -482,11 +479,10 @@ func TestEncoderChainedOptions(t *testing.T) {
I Inner `toml:"i"`
}
long := strings.Repeat("x", 200)
out, err := NewEncoder().
GroupByKind(false).
OmitEmptyArrays().
UseLiteralMultiline(50).
Marshal(Cfg{S: "short", I: Inner{V: long}})
out, err := Marshal(Cfg{S: "short", I: Inner{V: long}},
Layout(LayoutKindDeclaration),
OmitEmptyArrays(true),
LiteralMultiline(50))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -1302,7 +1298,7 @@ func TestEncoderEquivalenceToMarshal(t *testing.T) {
if err != nil {
t.Fatalf("marshal: %v", err)
}
b, err := NewEncoder().Marshal(in)
b, err := Marshal(in)
if err != nil {
t.Fatalf("encoder marshal: %v", err)
}
@@ -1339,8 +1335,8 @@ func TestEncodeErrorCarriesPath(t *testing.T) {
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "server.port" {
t.Fatalf("Path = %q, want %q", ee.Path, "server.port")
if ee.Path.String() != "server.port" {
t.Fatalf("Path = %v, want server.port", ee.Path)
}
if ee.Err == nil || ee.Err.Error() != "bad timestamp" {
t.Fatalf("Err = %v", ee.Err)
@@ -1359,8 +1355,8 @@ func TestEncodeErrorTopLevelPathHasNoLeadingDot(t *testing.T) {
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "port" {
t.Fatalf("Path = %q, want %q", ee.Path, "port")
if ee.Path.String() != "port" {
t.Fatalf("Path = %v, want port", ee.Path)
}
if err.Error() != "interpres: port: bad timestamp" {
t.Fatalf("message = %q", err.Error())
@@ -1382,8 +1378,8 @@ func TestEncodeErrorHeterogeneousArrayPath(t *testing.T) {
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "items[0]" {
t.Fatalf("Path = %q, want %q", ee.Path, "items[0]")
if ee.Path.String() != "items[0]" {
t.Fatalf("Path = %v, want items[0]", ee.Path)
}
}
@@ -1541,8 +1537,8 @@ func TestMarshalTextErrorCarriesPath(t *testing.T) {
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "inner.f" {
t.Fatalf("Path = %q, want %q", ee.Path, "inner.f")
if ee.Path.String() != "inner.f" {
t.Fatalf("Path = %v, want inner.f", ee.Path)
}
}
@@ -1710,7 +1706,7 @@ func TestEncoderInlineTables(t *testing.T) {
// With the option both fit the threshold and become inline tables, nested
// ones included.
out, err := NewEncoder().InlineTables(60).Marshal(cfg)
out, err := Marshal(cfg, InlineTables(60))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -1720,7 +1716,7 @@ func TestEncoderInlineTables(t *testing.T) {
}
// A threshold below the rendering keeps the header form.
out, err = NewEncoder().InlineTables(10).Marshal(cfg)
out, err = Marshal(cfg, InlineTables(10))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -1749,7 +1745,7 @@ func TestEncoderInlineTablesOrderAndRoundTrip(t *testing.T) {
if err != nil {
t.Fatalf("marshal: %v", err)
}
compact, err := NewEncoder().InlineTables(20).Marshal(cfg)
compact, err := Marshal(cfg, InlineTables(20))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -1785,7 +1781,7 @@ func TestEncoderInlineTablesKeepsArraysOfTables(t *testing.T) {
Small inlineTLS `toml:"small"`
}
cfg := Cfg{Items: []Item{{N: 1}}, Small: inlineTLS{On: true}}
out, err := NewEncoder().InlineTables(60).Marshal(cfg)
out, err := Marshal(cfg, InlineTables(60))
if err != nil {
t.Fatalf("marshal: %v", err)
}
@@ -1920,8 +1916,8 @@ func TestMarshalerElementErrorCarriesPath(t *testing.T) {
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "items[1]" {
t.Fatalf("Path = %q, want %q", ee.Path, "items[1]")
if ee.Path.String() != "items[1]" {
t.Fatalf("Path = %v, want items[1]", ee.Path)
}
}
@@ -1946,3 +1942,457 @@ func TestMarshalerResultIsNormalised(t *testing.T) {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalNumber(t *testing.T) {
t.Run("the literal is written as it is", func(t *testing.T) {
out, err := Marshal(map[string]any{
"hex": Number("0x1f"), "sep": Number("1_000"),
"signed": Number("+1.0"), "inf": Number("inf"),
})
if err != nil {
t.Fatal(err)
}
want := "hex = 0x1f\ninf = inf\nsep = 1_000\nsigned = +1.0\n"
if string(out) != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
})
t.Run("a Number field round-trips", func(t *testing.T) {
type Cfg struct {
Rate Number `toml:"rate"`
}
out, err := Marshal(Cfg{Rate: "1_000"})
if err != nil {
t.Fatal(err)
}
if string(out) != "rate = 1_000\n" {
t.Fatalf("output %q", out)
}
var back map[string]any
if err := Unmarshal(out, &back, NumbersAsLiterals(true)); err != nil {
t.Fatal(err)
}
if got, ok := back["rate"].(Number); !ok || got != "1_000" {
t.Errorf("round trip = %#v, want Number(\"1_000\")", back["rate"])
}
})
t.Run("a Number inside a value array", func(t *testing.T) {
out, err := Marshal(map[string]any{"vals": []any{Number("0x1f"), "s", int64(2)}})
if err != nil {
t.Fatal(err)
}
want := "vals = [0x1f, \"s\", 2]\n"
if string(out) != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
})
t.Run("an invalid literal is an error", func(t *testing.T) {
for _, lit := range []Number{"01", "1__0", "abc", "1.2.3"} {
if _, err := Marshal(map[string]any{"n": lit}); err == nil {
t.Errorf("Number(%q) encoded without an error", lit)
}
}
})
}
func TestMarshalAppend(t *testing.T) {
buf := []byte("preamble\n")
out, err := MarshalAppend(buf, map[string]any{"a": int64(1)})
if err != nil {
t.Fatal(err)
}
want := "preamble\na = 1\n"
if string(out) != want {
t.Errorf("output %q, want %q", out, want)
}
if &out[0] != &buf[0] {
t.Log("append reallocated; capacity differed")
}
out2, err := MarshalAppend(out, map[string]any{"b": true})
if err != nil {
t.Fatal(err)
}
if string(out2) != want+"b = true\n" {
t.Errorf("second append %q", out2)
}
buf = []byte("keep\n")
if out3, err := MarshalAppend(buf, Document{}); err == nil {
t.Errorf("MarshalAppend with an unencodable value = %q, want an error", out3)
}
}
func TestMarshalCyclicData(t *testing.T) {
t.Run("a cyclic struct is an error, not a crash", func(t *testing.T) {
type Node struct {
Name string `toml:"name"`
Next *Node `toml:"next"`
}
a := &Node{Name: "a"}
b := &Node{Name: "b"}
a.Next = b
b.Next = a
_, err := Marshal(a)
if err == nil {
t.Fatal("Marshal(cyclic) succeeded, want an error")
}
if !strings.Contains(err.Error(), "may be cyclic") {
t.Errorf("err = %v, want it to name the cycle", err)
}
})
t.Run("a cyclic map is an error", func(t *testing.T) {
m := map[string]any{}
m["self"] = m
if _, err := Marshal(m); err == nil {
t.Fatal("Marshal(cyclic map) succeeded, want an error")
}
})
t.Run("a cyclic value array is an error", func(t *testing.T) {
m := map[string]any{}
m["items"] = []any{int64(1), m}
if _, err := Marshal(map[string]any{"outer": m}); err == nil {
t.Fatal("Marshal(cyclic array) succeeded, want an error")
}
})
t.Run("a deeply nested but finite value encodes", func(t *testing.T) {
type Node struct {
Next *Node `toml:"next"`
}
root := &Node{}
cur := root
for range 5000 {
cur.Next = &Node{}
cur = cur.Next
}
if _, err := Marshal(root); err != nil {
t.Errorf("Marshal(deep) = %v, want nil", err)
}
})
}
func TestUnmarshalOptionsShape(t *testing.T) {
data := []byte("host = \"db\"\nextra = 1\n")
type Config struct {
Host string `toml:"host,required"`
}
t.Run("the zero value takes the defaults", func(t *testing.T) {
var cfg struct {
Host string `toml:"host"`
Extra int `toml:"extra"`
}
if err := Unmarshal(data, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Host != "db" || cfg.Extra != 1 {
t.Errorf("decoded %+v", cfg)
}
})
t.Run("strict and required work in one call", func(t *testing.T) {
err := Unmarshal(data, &Config{}, RejectUnknownFields(true))
want := `interpres: unknown field "extra" for interpres.Config`
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("UseNumber keeps the literal", func(t *testing.T) {
var tree map[string]any
in := []byte("n = 1_000\n")
if err := Unmarshal(in, &tree, NumbersAsLiterals(true)); err != nil {
t.Fatal(err)
}
if got, ok := tree["n"].(Number); !ok || got != "1_000" {
t.Errorf("n = %#v, want Number(\"1_000\")", tree["n"])
}
})
t.Run("the limits apply", func(t *testing.T) {
var nested strings.Builder
nested.WriteString("x = ")
for range 20 {
nested.WriteString("[")
}
nested.WriteString("1")
for range 20 {
nested.WriteString("]")
}
var tree map[string]any
if err := Unmarshal([]byte(nested.String()), &tree, MaxNestingDepth(10)); err == nil {
t.Error("a document over MaxDepth decoded, want an error")
}
if err := Unmarshal([]byte("a = 1\n"), &tree, MaxInputSize(2)); err == nil {
t.Error("a document over MaxInputSize decoded, want an error")
}
})
}
func TestInlineTag(t *testing.T) {
type Inner struct {
A int `toml:"a"`
B int `toml:"b"`
}
t.Run("a struct field writes inline", func(t *testing.T) {
type Cfg struct {
Inner Inner `toml:"inner,inline"`
}
out, err := Marshal(Cfg{Inner: Inner{1, 2}})
if err != nil {
t.Fatal(err)
}
if string(out) != "inner = {a = 1, b = 2}\n" {
t.Errorf("output %q", out)
}
})
t.Run("a map field writes inline", func(t *testing.T) {
type Cfg struct {
Opts map[string]int `toml:"opts,inline"`
}
out, err := Marshal(Cfg{Opts: map[string]int{"x": 1}})
if err != nil {
t.Fatal(err)
}
if string(out) != "opts = {x = 1}\n" {
t.Errorf("output %q", out)
}
})
t.Run("a named embedded struct writes inline", func(t *testing.T) {
type Cfg struct {
Inner `toml:"inner,inline"`
}
out, err := Marshal(Cfg{Inner: Inner{1, 2}})
if err != nil {
t.Fatal(err)
}
if string(out) != "inner = {a = 1, b = 2}\n" {
t.Errorf("output %q", out)
}
})
t.Run("an inline field decodes back", func(t *testing.T) {
type Cfg struct {
Inner Inner `toml:"inner,inline"`
}
var cfg Cfg
if err := Unmarshal([]byte("inner = {a = 3, b = 4}\n"), &cfg); err != nil {
t.Fatal(err)
}
if cfg.Inner != (Inner{3, 4}) {
t.Errorf("decoded %+v", cfg.Inner)
}
})
t.Run("a forced inline of an array of tables is an error", func(t *testing.T) {
type Item struct {
N int `toml:"n"`
}
type Cfg struct {
Items []Item `toml:"items,inline"`
}
if _, err := Marshal(Cfg{Items: []Item{{1}}}); err == nil {
t.Error("forced inline of an array of tables succeeded, want an error")
}
})
t.Run("without the tag the header form stands", func(t *testing.T) {
type Cfg struct {
Inner Inner `toml:"inner"`
}
out, err := Marshal(Cfg{Inner: Inner{1, 2}})
if err != nil {
t.Fatal(err)
}
if string(out) != "[inner]\na = 1\nb = 2\n" {
t.Errorf("output %q", out)
}
})
}
func TestOmitEmptyJSONSemantics(t *testing.T) {
type Cfg struct {
Empty string `toml:"empty,omitempty"`
Full string `toml:"full,omitempty"`
Zero int `toml:"zero,omitempty"`
One int `toml:"one,omitempty"`
Off bool `toml:"off,omitempty"`
On bool `toml:"on,omitempty"`
Nil *string `toml:"nil,omitempty"`
Set *string `toml:"set,omitempty"`
Nothing map[string]string `toml:"nothing,omitempty"`
Somethg map[string]string `toml:"somethg,omitempty"`
}
s := "x"
out, err := Marshal(Cfg{
Full: "y",
One: 1,
On: true,
Set: &s,
Somethg: map[string]string{"k": "v"},
})
if err != nil {
t.Fatal(err)
}
want := "full = \"y\"\none = 1\non = true\nset = \"x\"\n\n[somethg]\nk = \"v\"\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
}
func TestEmitFieldComments(t *testing.T) {
type Cfg struct {
Host string `toml:"host,comment=The host to dial"`
Port int `toml:"port,comment=The port to listen on.\nThe default is 8080."`
User string `toml:"user"`
}
cfg := Cfg{Host: "db", Port: 5432, User: "admin"}
t.Run("off by default", func(t *testing.T) {
out, err := Marshal(cfg)
if err != nil {
t.Fatal(err)
}
want := "host = \"db\"\nport = 5432\nuser = \"admin\"\n"
if string(out) != want {
t.Errorf("output:\n%q", out)
}
})
t.Run("on, the comments print above their lines", func(t *testing.T) {
out, err := Marshal(cfg, EmitFieldComments(true))
if err != nil {
t.Fatal(err)
}
want := "# The host to dial\nhost = \"db\"\n" +
"# The port to listen on.\n# The default is 8080.\nport = 5432\n" +
"user = \"admin\"\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
var back Cfg
if err := Unmarshal(out, &back); err != nil {
t.Fatalf("the output does not re-parse: %v", err)
}
if back != cfg {
t.Errorf("round trip = %+v", back)
}
})
t.Run("a table header carries its comment", func(t *testing.T) {
type Inner struct {
A int `toml:"a,comment=The a"`
}
type Nested struct {
Inner Inner `toml:"inner,comment=The inner table"`
}
out, err := Marshal(Nested{Inner: Inner{1}}, EmitFieldComments(true))
if err != nil {
t.Fatal(err)
}
want := "# The inner table\n[inner]\n# The a\na = 1\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
}
// errWriter fails every write with a fixed error.
type errWriter struct{ err error }
func (w errWriter) Write([]byte) (int, error) { return 0, w.err }
// TestMarshalWrite covers the streaming entry: the happy path with options
// and a failing writer.
func TestMarshalWrite(t *testing.T) {
var buf bytes.Buffer
err := MarshalWrite(&buf, map[string]any{"b": 2, "a": 1})
if err != nil {
t.Fatalf("MarshalWrite: %v", err)
}
// A map carries no order, so the writer uses the sorted one.
if buf.String() != "a = 1\nb = 2\n" {
t.Errorf("output = %q", buf.String())
}
writeErr := errors.New("boom")
if err := MarshalWrite(errWriter{writeErr}, map[string]any{"a": 1}); !errors.Is(err, writeErr) {
t.Errorf("err = %v, want the write error wrapped", err)
}
}
// TestMarshalRejectsUnsupportedKinds pins the clear error a field of a kind
// TOML cannot carry raises, through the struct walk.
func TestMarshalRejectsUnsupportedKinds(t *testing.T) {
tests := []struct {
name string
value any
}{
{"func", struct {
F func() `toml:"f"`
}{}},
{"chan", struct {
C chan int `toml:"c"`
}{}},
{"complex", struct {
Z complex128 `toml:"z"`
}{}},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
_, err := Marshal(tt.value)
if err == nil {
t.Fatalf("Marshal accepted %#v", tt.value)
}
if !strings.Contains(err.Error(), "cannot encode") {
t.Errorf("err = %v, want the cannot-encode complaint", err)
}
})
}
}
// TestMarshalOmitsEmptyPointerTableSlice pins that an empty slice of pointer
// tables is omitted, the rule its non-pointer form already follows.
func TestMarshalOmitsEmptyPointerTableSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
out, err := Marshal(struct {
Items []*item `toml:"items"`
}{})
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if len(out) != 0 {
t.Errorf("output = %q, want the empty array of tables omitted", out)
}
}
// TestMarshalRejectsNonWholeMinuteOffset pins that a zone offset carrying
// seconds is refused instead of silently losing them.
func TestMarshalRejectsNonWholeMinuteOffset(t *testing.T) {
z := time.FixedZone("", 57*60+44)
_, err := Marshal(struct {
Stamp time.Time `toml:"stamp"`
}{Stamp: time.Date(1890, 1, 1, 12, 0, 0, 0, z)})
if err == nil || !strings.Contains(err.Error(), "not a whole number of minutes") {
t.Errorf("err = %v, want the whole-minute offset complaint", err)
}
_, err = Marshal(struct {
Stamp OffsetDateTime `toml:"stamp"`
}{Stamp: OffsetDateTime{time.Date(1890, 1, 1, 12, 0, 0, 0, z)}})
if err == nil || !strings.Contains(err.Error(), "not a whole number of minutes") {
t.Errorf("err = %v, want the whole-minute offset complaint for the wrapper", err)
}
}
// cancelOnMarshal cancels the context the encode runs under, the moment its
// method is called, so the emission that follows is already past the walk's
// own checks.
type cancelOnMarshal struct {
cancel context.CancelFunc
}
func (c cancelOnMarshal) MarshalTOML() (any, error) {
c.cancel()
return int64(1), nil
}
// TestMarshalContextCancelsDuringEmission pins that a context cancelled
// between the walk and the emission stops the encode instead of writing the
// whole document out.
func TestMarshalContextCancelsDuringEmission(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
value := map[string]any{"k": cancelOnMarshal{cancel}}
if _, err := MarshalContext(ctx, value); !errors.Is(err, context.Canceled) {
t.Errorf("err = %v, want the cancellation", err)
}
}
+101
View File
@@ -0,0 +1,101 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres_test
import (
"fmt"
"log"
"sourcedock.dev/petrbalvin/interpres/v2"
)
func ExampleParse() {
const doc = `
title = "interpres"
[server]
host = "127.0.0.1"
port = 9090
`
d, err := interpres.Parse([]byte(doc))
if err != nil {
log.Fatal(err)
}
for _, key := range d.Root().Keys() { // written order, not sorted
entry, _ := d.Root().Get(key)
fmt.Println(key, "=", entry.Value())
}
// Output:
// title = interpres
// server = map[host:127.0.0.1 port:9090]
}
func ExampleUnmarshal() {
type Config struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
var cfg Config
err := interpres.Unmarshal([]byte("host = \"db\"\nport = 5432\n"), &cfg)
if err != nil {
log.Fatal(err)
}
fmt.Println(cfg.Host, cfg.Port)
// Output: db 5432
}
func ExampleMarshal() {
type Server struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
type Config struct {
Title string `toml:"title"`
Server Server `toml:"server"`
}
out, err := interpres.Marshal(Config{
Title: "demo",
Server: Server{Host: "127.0.0.1", Port: 9090},
})
if err != nil {
log.Fatal(err)
}
fmt.Printf("%s", out)
// Output:
// title = "demo"
//
// [server]
// host = "127.0.0.1"
// port = 9090
}
func ExampleNumbersAsLiterals() {
var tree map[string]any
err := interpres.Unmarshal([]byte("rate = 1_000\n"), &tree,
interpres.RejectUnknownFields(true),
interpres.NumbersAsLiterals(true))
if err != nil {
log.Fatal(err)
}
fmt.Println(tree["rate"], string(tree["rate"].(interpres.Number)))
// Output: 1_000 1_000
}
func ExampleInlineTables() {
type Config struct {
Title string `toml:"title"`
Extras map[string]string `toml:"extras,inline"`
}
out, err := interpres.Marshal(Config{Title: "demo", Extras: map[string]string{"b": "two", "a": "one"}},
interpres.Layout(interpres.LayoutKindDeclaration),
interpres.LiteralMultiline(80),
interpres.InlineTables(40))
if err != nil {
log.Fatal(err)
}
fmt.Printf("%s", out)
// Output:
// title = "demo"
// extras = {a = "one", b = "two"}
}
+48 -15
View File
@@ -3,11 +3,12 @@
// Command basic demonstrates decoding and encoding a TOML document with
// interpres. It exercises struct mapping, arrays of tables, Marshaler
// customisation, the Decoder's strict mode, and the Encoder's policy
// options, covering every feature a regular user would reach for.
// customisation, and the Encoder's policy options, covering every feature a
// regular user would reach for.
package main
import (
"errors"
"fmt"
"io"
"os"
@@ -17,8 +18,7 @@ import (
)
// document is a small but realistic configuration: it has scalars, a
// sub-table, an array of tables, and a date-time. We pick a 32-bit port so
// the demonstration also covers overflow-safe integer conversion.
// sub-table, an array of tables, and a date-time.
const document = `
title = "interpres demo"
launched = 2024-11-04T09:00:00Z
@@ -39,13 +39,16 @@ admin = false
// Config mirrors the document above. The Server field is a named struct so
// the reader sees explicit subtable boundaries; Users is a slice of named
// structs so the array-of-tables path is exercised.
// structs so the array-of-tables path is exercised. Retries carries the
// `omitzero` tag option: a zero value of the field's type drops from the
// output, and a `time.Duration` zero is zero nanoseconds.
type Config struct {
Title string `toml:"title"`
Launched time.Time `toml:"launched"`
Debug bool `toml:"debug"`
Server Server `toml:"server"`
Users []User `toml:"users"`
Title string `toml:"title"`
Launched time.Time `toml:"launched"`
Debug bool `toml:"debug"`
Server Server `toml:"server"`
Users []User `toml:"users"`
Retries time.Duration `toml:"retries,omitzero"`
}
type Server struct {
@@ -58,9 +61,11 @@ type User struct {
Admin bool `toml:"admin"`
}
// Port is a typed alias that controls how its value appears in TOML. The
// MarshalTOML hook returns a string, so a Port field is rendered as
// "host:port" instead of the raw integer.
// Port is a typed string alias that carries a Marshaler. The MarshalTOML
// hook returns the string unchanged, so a Port field is rendered as the
// string it holds, a string the encoding would print the same way without
// the hook; the demonstration that a Marshaler reshapes a value is
// Endpoint's below.
type Port string
func (p Port) MarshalTOML() (any, error) {
@@ -130,12 +135,12 @@ func Run(stdout, stderr io.Writer) int {
}
fmt.Fprintf(stdout, "\n--- marshal (group by kind, default) ---\n%s", out)
out2, err := interpres.NewEncoder().GroupByKind(false).Marshal(cfg)
out2, err := interpres.Marshal(cfg, interpres.Layout(interpres.LayoutKindDeclaration))
if err != nil {
fmt.Fprintln(stderr, "marshal:", err)
return 1
}
fmt.Fprintf(stdout, "\n--- marshal (GroupByKind=false) ---\n%s", out2)
fmt.Fprintf(stdout, "\n--- marshal (LayoutKindDeclaration) ---\n%s", out2)
// Demonstrate Unmarshaler-style mutation: re-decode the second output to
// prove it round-trips back into the same Go value.
@@ -147,5 +152,33 @@ func Run(stdout, stderr io.Writer) int {
fmt.Fprintf(stdout, "\n--- round-trip --- ok (title=%q, users=%d)\n",
roundTripped.Title, len(roundTripped.Users))
// Typed errors: a decode failure names the key path it failed at, and
// errors.AsType reaches the DecodeError to read the path and the cause
// separately, without parsing the message text.
bad := []byte("[[users]]\nname = \"x\"\nadmin = \"not-a-bool\"\n")
var badCfg Config
err = interpres.Unmarshal(bad, &badCfg)
if err == nil {
fmt.Fprintln(stderr, "expected a decode error")
return 1
}
if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
fmt.Fprintf(stdout, "\n--- typed error --- path %s: %v\n", de.Path.String(), de.Err)
} else {
fmt.Fprintln(stderr, "expected a DecodeError")
return 1
}
// omitzero: the retries field carries the tag option and a zero duration,
// so the re-encoded config above simply has no retries line. Give it a
// value and the line appears.
cfg.Retries = 30 * time.Second
out3, err := interpres.Marshal(cfg)
if err != nil {
fmt.Fprintln(stderr, "marshal:", err)
return 1
}
fmt.Fprintf(stdout, "\n--- omitzero ---\n%s", out3)
return 0
}
+1 -1
View File
@@ -25,7 +25,7 @@ func TestRunPrintsConfigAndMarshal(t *testing.T) {
"admin=false",
"--- marshal (group by kind, default) ---",
`title = "interpres demo"`,
"--- marshal (GroupByKind=false) ---",
"--- marshal (LayoutKindDeclaration) ---",
"[server]",
"port = 9090",
"[[users]]",
+39
View File
@@ -0,0 +1,39 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
// Command statements walks the top-level statements of a TOML document with
// interpres.Statements, the shape a configuration tool uses to read the
// sections it cares about and skip the rest.
package main
import (
"fmt"
"io"
"os"
"sourcedock.dev/petrbalvin/interpres/v2"
)
func main() {
if err := run(os.Stdin, os.Stdout); err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
}
func run(stdin io.Reader, stdout io.Writer) error {
for stmt, err := range interpres.Statements(stdin) {
if err != nil {
return err
}
switch {
case stmt.Index >= 0:
fmt.Fprintf(stdout, "[[%s]] #%d\n", stmt.Key, stmt.Index)
case stmt.Table != nil:
fmt.Fprintf(stdout, "[%s] keys: %v\n", stmt.Key, stmt.Table.Keys())
default:
fmt.Fprintf(stdout, "%s = %v\n", stmt.Key, stmt.Value)
}
}
return nil
}
+39
View File
@@ -0,0 +1,39 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package main
import (
"strings"
"testing"
)
func TestStatementsExample(t *testing.T) {
in := strings.NewReader(`title = "demo"
port = 8080
[server]
host = "127.0.0.1"
[[items]]
name = "a"
[[items]]
name = "b"
`)
var out strings.Builder
if err := run(in, &out); err != nil {
t.Fatal(err)
}
for _, want := range []string{
"title = demo",
"port = 8080",
"[server] keys: [host]",
"[[items]] #0",
"[[items]] #1",
} {
if !strings.Contains(out.String(), want) {
t.Errorf("output missing %q:\n%s", want, out.String())
}
}
}
+142
View File
@@ -0,0 +1,142 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"context"
"errors"
"reflect"
"strings"
"testing"
"time"
)
type fuzzNested struct {
X int `toml:"x"`
Y string `toml:"y"`
}
type fuzzDoc struct {
Num int `toml:"num"`
Flt float64 `toml:"flt"`
Str string `toml:"str"`
Flag bool `toml:"flag"`
Small uint8 `toml:"small"`
When time.Time `toml:"when"`
Tags []string `toml:"tags"`
Lims map[string]any `toml:"lims"`
Tab fuzzNested `toml:"tab"`
Arr []fuzzNested `toml:"arr"`
Other string `toml:"other"`
}
// fuzzStmts is the statement pool the generated documents draw from: every
// destination kind the targeted parse handles, beside the shapes that make
// it fall back (overflow, unknown tables, duplicate keys).
var fuzzStmts = []string{
`num = 1`, `num = 300`, `small = 300`, `small = 7`,
`flt = 2.5`, `str = "x"`, `flag = true`,
`when = 1979-05-27T07:32:00Z`,
`tags = ["a", "b"]`, `tags = []`, `lims = { k = 1 }`,
`[tab]`, `tab.x = 1`, `tab.y = "s"`, `x = 2`, `y = "t"`,
`[[arr]]`, `x = 3`, `y = "u"`,
`[tab.nested]`, `x = 4`,
`other = "o"`, `zz = 1`, `[zz]`, `k = 1`,
`num = 2`,
}
func fuzzDocument(data []byte) []byte {
var b strings.Builder
for i, by := range data {
if i > 0 {
b.WriteByte('\n')
}
b.WriteString(fuzzStmts[int(by)%len(fuzzStmts)])
}
return []byte(b.String())
}
// treeDecodeInto is the reference decode: the ordinary tree path, non-strict
// like the fuzz decode; the strict contracts have their own deterministic
// tests.
func treeDecodeInto(data []byte, v any) error {
dec := newDecoder()
tree, _, err := parseWithOptions(context.Background(), data, parseOptions{}, false)
if err != nil {
return err
}
return dec.decode(tree, v)
}
// decodeFinding normalises an error for the comparison. Decode-stage
// findings several tables may produce (an unknown field, a missing required
// key) compare as their class alone: the tree decode picks the reporting
// table by map order and so does not promise one. Everything else compares
// as its exact text.
func decodeFinding(err error) string {
if err == nil {
return ""
}
if de, ok := errors.AsType[*DecodeError](err); ok {
if strings.Contains(de.Err.Error(), "unknown field") {
return "unknown"
}
if strings.Contains(de.Err.Error(), "missing required key") {
return "required"
}
return de.Path.String() + ": " + de.Err.Error()
}
return err.Error()
}
// FuzzTargetedDecode holds the targeted parse to the tree decode as its
// reference: for every generated document the two paths must agree on the
// error class and on the decoded value.
func FuzzTargetedDecode(f *testing.F) {
seeds := []string{
"num = 1\nstr = \"x\"\n[tab]\nx = 2\n[[arr]]\nx = 3\n",
"small = 300\n",
"[tab]\ntab.x = 1\n",
"lims = { k = 1 }\ntags = [\"a\"]\n",
"[[arr]]\ny = \"u\"\n[zz]\nk = 1\n",
"when = 07:32:00\n[tab.nested]\n",
"small = 300\n[[arr]]\nflt = 2.5\n",
}
for _, s := range seeds {
f.Add([]byte(s))
}
f.Fuzz(func(t *testing.T, data []byte) {
doc := fuzzDocument(data)
var tgt fuzzDoc
tgtErr := Unmarshal(doc, &tgt)
if tgtErr != nil {
// A document with several decode-stage findings reports a different
// one per run (the tree decode walks its maps in random order), so the
// reference gets a few chances to produce the finding the targeted
// side carries. The targeted error is either the tree's own or the
// fallback already reran the tree.
for i := range 8 {
var ref fuzzDoc
refErr := treeDecodeInto(doc, &ref)
if refErr == nil {
t.Fatalf("reference succeeded on retry %d, targeted failed: %v\ndoc:\n%s", i, tgtErr, doc)
}
if decodeFinding(refErr) == decodeFinding(tgtErr) {
return
}
if i == 7 {
t.Fatalf("errors disagree after retries:\ntargeted: %v\nlast tree: %v\ndoc:\n%s", tgtErr, refErr, doc)
}
}
}
var ref fuzzDoc
refErr := treeDecodeInto(doc, &ref)
if refErr != nil {
t.Fatalf("reference failed, targeted succeeded: %v\ndoc:\n%s", refErr, doc)
}
if !reflect.DeepEqual(ref, tgt) {
t.Fatalf("values disagree:\ntree: %#v\ntargeted: %#v\ndoc:\n%s", ref, tgt, doc)
}
})
}
+68
View File
@@ -121,3 +121,71 @@ func tomlEqual(a, b any) bool {
return reflect.DeepEqual(a, b)
}
}
// FuzzMarshal drives the encoder with generated Go values and holds it to
// the same round-trip invariant FuzzParse holds the parser to: a value built
// only of encodable kinds must marshal, the document must re-parse, and the
// tree must equal the value it came from.
func FuzzMarshal(f *testing.F) {
seeds := [][]byte{
{},
{0, 0, 1, 2},
{1, 1, 2, 3, 2, 2, 3, 4},
{0, 3, 1, 9, 3, 3, 2, 8, 1, 0, 1, 7},
}
for _, s := range seeds {
f.Add(s)
}
f.Fuzz(func(t *testing.T, data []byte) {
v := fuzzValue(data)
out, err := Marshal(v)
if err != nil {
t.Fatalf("marshal of an encodable value failed: %v\nvalue: %#v", err, v)
}
tree, err := ParseMap(out)
if err != nil {
t.Fatalf("re-parse of the emitted document failed: %v\ndoc:\n%s", err, out)
}
if !tomlEqual(v, tree) {
t.Fatalf("round-trip changed the value\nvalue: %#v\ndoc:\n%s\ntree: %#v", v, out, tree)
}
})
}
// fuzzKeys is the fixed key pool the generated values draw from, so keys are
// always valid bare keys and repeat often.
var fuzzKeys = []string{"alpha", "beta", "gamma", "delta"}
// fuzzValue builds a map[string]any of encodable kinds from data: integers,
// positive floats, short strings, nested tables and scalar arrays. The bytes
// decide the shape deterministically.
func fuzzValue(data []byte) map[string]any {
root := map[string]any{}
cur := root
depth := 0
for i := 0; i+3 < len(data); i += 4 {
key := fuzzKeys[int(data[i])%len(fuzzKeys)]
switch data[i+1] % 5 {
case 0:
cur[key] = int64(data[i+2])<<8 | int64(data[i+3])
case 1:
cur[key] = float64(int(data[i+2])%1000)/8.0 + 0.125
case 2:
cur[key] = string(rune('a' + int(data[i+2])%26))
case 3:
cur[key] = []any{
int64(data[i+2]),
float64(int(data[i+3])%100)/4.0 + 0.25,
string(rune('a' + int(data[i+3])%26)),
}
case 4:
if depth < 6 {
next := map[string]any{}
cur[key] = next
cur = next
depth++
}
}
}
return root
}
+482 -162
View File
@@ -16,59 +16,122 @@
// doc, err := interpres.Parse(data)
// tree := doc.Map()
//
// A Decoder allows strict decoding that rejects keys without a matching
// struct field, mirroring (*json.Decoder).DisallowUnknownFields.
// Strict decoding that rejects keys without a matching struct field is an
// option, mirroring the RejectUnknownMembers option of encoding/json/v2:
//
// err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
package interpres
import (
"bytes"
"context"
"errors"
"fmt"
"io"
"iter"
"os"
"reflect"
"slices"
"strings"
"time"
)
// A SyntaxError describes a malformed TOML document, including the 1-based
// line on which the problem was detected.
// A SyntaxError describes a malformed TOML document. Line is the 1-based line
// the problem was detected on. Offset is the byte offset in the input the scan
// stopped at, and Column is the 1-based column on that line; both are new in
// 2.0 and a struct literal that names Line and Msg alone still builds.
type SyntaxError struct {
Line int
Msg string
Line int
Offset int
Column int
Msg string
}
func (e *SyntaxError) Error() string {
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
}
// SourceLine returns the source line the error points at, rendered from src,
// followed by a caret line marking the column. It is meant for a message the
// reader sees under the input:
//
// port = = 8080
// ^
//
// The caret sits at Offset when it falls inside src, and at the start of the
// line when the error carries no position.
func (e *SyntaxError) SourceLine(src []byte) string {
off := min(e.Offset, len(src))
start := 0
if i := bytes.LastIndexByte(src[:off], '\n'); i >= 0 {
start = i + 1
}
end := len(src)
if i := bytes.IndexByte(src[start:], '\n'); i >= 0 {
end = start + i
}
return string(src[start:end]) + "\n" + strings.Repeat(" ", off-start) + "^"
}
// A Path names a value in a document, one segment per level from the root:
// a key contributes its name and an array element its bracketed index, so the
// path of the weight field of the first item is the segments
// ["items", "[0]", "weight"]. String renders the TOML notation,
// "items[0].weight".
type Path []string
// String renders the path the way a TOML document writes it: keys join with
// dots and an index attaches to the previous segment in brackets.
func (p Path) String() string {
var b strings.Builder
for _, s := range p {
if strings.HasPrefix(s, "[") {
b.WriteString(s)
continue
}
if b.Len() > 0 {
b.WriteByte('.')
}
b.WriteString(s)
}
return b.String()
}
// A DecodeError wraps a decoding failure with the key path at which it
// happened. Path lists one segment per level from the document root, the
// outermost key first: a key contributes its name and an array element its
// bracketed index, so the path of the weight field in the first item reads
// ["items", "[0]", "weight"]. The rendered message is unchanged by the type;
// read it programmatically with errors.AsType:
// happened. Read the path programmatically with errors.AsType:
//
// if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
// fmt.Println(de.Path, de.Err)
// fmt.Println(de.Path.String(), de.Err)
// }
type DecodeError struct {
// Path is the key path from the document root, outermost key first.
Path []string
Path Path
// Err is the failure at that path.
Err error
}
func (e *DecodeError) Error() string { return e.Path[0] + ": " + e.Err.Error() }
func (e *DecodeError) Error() string {
msg := strings.TrimPrefix(e.Err.Error(), "interpres: ")
if p := e.Path.String(); p != "" {
return "interpres: " + p + ": " + msg
}
return "interpres: " + msg
}
// Unwrap returns the failure the path points at.
func (e *DecodeError) Unwrap() error { return e.Err }
// newDecodeError wraps err with one path segment. The rest of the path comes
// from the DecodeError err already carries, if any: the decoder wraps each
// key and index on its way down, so the innermost wrap holds the deepest
// segments and each outer wrap prepends one.
// key and index on its way down, so the wrap flattens that inner error's
// segments onto the front and keeps the failure it pointed at, leaving one
// path and one failure to render.
func newDecodeError(key string, err error) *DecodeError {
path := make([]string, 0, 4)
path := make(Path, 0, 4)
path = append(path, key)
if de, ok := errors.AsType[*DecodeError](err); ok {
path = append(path, de.Path...)
err = de.Err
}
return &DecodeError{Path: path, Err: err}
}
@@ -80,12 +143,18 @@ func newDecodeError(key string, err error) *DecodeError {
// unchanged by the type; read it programmatically with errors.AsType.
type EncodeError struct {
// Path is the key path of the failing value.
Path string
Path Path
// Err is the failure at that path.
Err error
}
func (e *EncodeError) Error() string { return "interpres: " + e.Path + ": " + e.Err.Error() }
func (e *EncodeError) Error() string {
msg := strings.TrimPrefix(e.Err.Error(), "interpres: ")
if p := e.Path.String(); p != "" {
return "interpres: " + p + ": " + msg
}
return "interpres: " + msg
}
// Unwrap returns the failure the path points at.
func (e *EncodeError) Unwrap() error { return e.Err }
@@ -127,11 +196,37 @@ func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error) {
return tree, err
}
// parseOptions bound the work one parse may do. A zero field takes the
// default.
// ParseFile reads the TOML document at path and parses it into a Document,
// the shape Parse gives. Every error names the file it came from: a read
// failure and a parse failure alike carry the path as their first words,
// wrapped so errors.AsType still reaches the SyntaxError inside.
func ParseFile(path string) (*Document, error) {
data, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
doc, err := Parse(data)
if err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
return doc, nil
}
// Valid reports whether data is a valid TOML document: nil when the parser
// accepts it, and the parse error when it does not. It is the library call
// the --validate mode of interpres-decode is built on, and it reads nothing
// but the bytes it is given.
func Valid(data []byte) error {
_, err := ParseMapContext(context.Background(), data)
return err
}
// parseOptions bound the work one parse may do and the shape it produces. A
// zero field takes the default.
type parseOptions struct {
maxDepth int
maxInputSize int
useNumber bool
}
// parseWithOptions parses data, building the node tree of a Document when
@@ -152,7 +247,7 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
}
// The parser scans data in place; it only reads the buffer, and every
// string it stores in the tree is copied out of it.
p := &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, wantDoc: wantDoc}
p := &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, wantDoc: wantDoc, useNumber: opts.useNumber}
tree, err := p.parse()
if err != nil {
return nil, nil, err
@@ -165,10 +260,14 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
// Unmarshal parses a TOML document and stores the result in the value pointed
// to by v. v is typically a pointer to a struct or to a map[string]any.
// Options tune the call; with none, unknown keys are ignored, numbers are
// evaluated, and the nesting default applies.
//
// Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a
// case-insensitive match on the field name when no tag is present. A tag of
// "-" skips the field.
// "-" skips the field. Two document keys that differ only in case and both
// match one field resolve deterministically: the lexicographically greater
// one wins, the same key winning every run.
//
// A destination implementing Unmarshaler receives the parsed value as it is,
// a TOML string fills a destination implementing encoding.TextUnmarshaler, and
@@ -176,81 +275,176 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
// bare integer as its nanosecond count.
//
// Unmarshal is equivalent to UnmarshalContext with context.Background.
func Unmarshal(data []byte, v any) error {
return UnmarshalContext(context.Background(), data, v)
func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error {
return UnmarshalContext(context.Background(), data, v, opts...)
}
// ParseAs decodes a TOML document into T in one call, the generic shorthand
// for Unmarshal with a destination variable:
//
// cfg, err := interpres.ParseAs[Config](data, interpres.RejectUnknownFields(true))
//
// The options are Unmarshal's. The zero T comes back with the error.
func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error) {
var v T
err := Unmarshal(data, &v, opts...)
return v, err
}
// NewSchema precompiles the codec for T: the struct schema both directions
// walk and the interface flags the decoder and the encoder resolve through
// are built once and cached, so the first document pays the cost instead of
// the hot path. A T that is not a struct warms nothing; there is nothing to
// precompute for a map or a slice.
func NewSchema[T any]() {
t := reflect.TypeFor[T]()
if t.Kind() != reflect.Struct {
return
}
cachedStructSchema(t)
_ = typeFlags(t)
_ = encTypeFlags(t)
pt := reflect.PointerTo(t)
_ = typeFlags(pt)
_ = encTypeFlags(pt)
}
// UnmarshalContext is the cancellable variant of Unmarshal.
func UnmarshalContext(ctx context.Context, data []byte, v any) error {
tree, err := ParseMapContext(ctx, data)
if err != nil {
return err
}
return newDecoder().decode(tree, v)
func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...UnmarshalOption) error {
return settingsFor(opts).decode(ctx, data, v)
}
// A Decoder decodes a TOML document into a Go value with configurable
// strictness and configurable limits on the parse it performs.
type Decoder struct {
// UnmarshalRead reads the document from r and decodes it into v, the
// streaming-shaped entry the json/v2 vocabulary uses. The reader is
// consumed in full, because the parser scans its source in place; with
// MaxInputSize set, reading stops one byte past the limit so the size the
// option bounds is the memory held, not what a reader is drained into first.
// The options and the behaviour are Unmarshal's.
func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error {
s := settingsFor(opts)
var data []byte
var err error
if s.maxInputSize > 0 {
data, err = io.ReadAll(io.LimitReader(r, int64(s.maxInputSize)+1))
} else {
data, err = io.ReadAll(r)
}
if err != nil {
return fmt.Errorf("interpres: read: %w", err)
}
return Unmarshal(data, v, opts...)
}
// An UnmarshalOption configures one Unmarshal, UnmarshalContext,
// UnmarshalRead or ParseAs call. Options are function values over the
// private decode settings, the shape encoding/json/v2 uses for its own
// options, and compose by simple listing:
//
// err := interpres.Unmarshal(data, &cfg,
// interpres.RejectUnknownFields(true),
// interpres.NumbersAsLiterals(true))
//
// A destination that the direct skeleton cannot model falls back to the
// tree path, so every option means the same thing on every document.
type UnmarshalOption func(*decodeSettings)
// decodeSettings is the option carrier of one decode call. The context is
// not one: it arrives as its own argument, because every entry point names it
// explicitly.
type decodeSettings struct {
disallowUnknown bool
useNumber bool
maxDepth int
maxInputSize int
localLoc *time.Location
}
// NewDecoder returns a Decoder.
func NewDecoder() *Decoder { return &Decoder{} }
// DisallowUnknownFields causes Decode to return an error when the document
// contains a key with no matching destination struct field.
func (d *Decoder) DisallowUnknownFields() *Decoder {
d.disallowUnknown = true
return d
func settingsFor(opts []UnmarshalOption) *decodeSettings {
s := &decodeSettings{}
for _, opt := range opts {
opt(s)
}
return s
}
// MaxDepth bounds how deeply arrays and inline tables may nest in a document
// this decoder accepts. The parser is a recursive descent, so a document that
// nests without bound would exhaust the stack; one that nests deeper than the
// limit is rejected with a SyntaxError naming it instead. Use 0 or any
// negative value for the default of 10000, which no hand-written document
// approaches.
func (d *Decoder) MaxDepth(depth int) *Decoder {
d.maxDepth = depth
return d
}
// MaxInputSize bounds the size of a document this decoder accepts, in bytes; a
// larger one is rejected before parsing starts. Use 0 or any negative value for
// no limit, which is the default: the caller already holds the bytes, so the
// size is a policy the caller sets rather than a protection the library
// imposes on its own. Parse and ParseContext take no limit beyond the nesting
// default.
func (d *Decoder) MaxInputSize(size int) *Decoder {
d.maxInputSize = size
return d
}
// Decode parses data and stores the result in the value pointed to by v,
// honouring the decoder's strictness settings.
//
// Decode is equivalent to DecodeContext with context.Background.
func (d *Decoder) Decode(data []byte, v any) error {
return d.DecodeContext(context.Background(), data, v)
}
// DecodeContext is the cancellable variant of Decode.
func (d *Decoder) DecodeContext(ctx context.Context, data []byte, v any) error {
tree, _, err := parseWithOptions(ctx, data, parseOptions{
maxDepth: d.maxDepth,
maxInputSize: d.maxInputSize,
}, false)
// decode runs the decode the settings describe: the targeted parse when the
// destination takes it, the tree path otherwise or on fallback.
func (s *decodeSettings) decode(ctx context.Context, data []byte, v any) error {
dec := newDecoder()
dec.disallowUnknown = s.disallowUnknown
dec.ctx = ctx
dec.loc = s.localLoc
if canTargetDecode(v) {
// The targeted parse fills struct destinations without the
// intermediate tree; a document or destination it cannot model falls
// back to the tree path, whose contracts it keeps. The size limit is
// checked here, the targeted parse being the parse itself.
if s.maxInputSize > 0 && len(data) > s.maxInputSize {
return fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), s.maxInputSize)
}
if err := parseIntoTargeted(ctx, data, dec, s.useNumber, s.maxDepth, v); err != errTargetFallback {
return err
}
}
opts := parseOptions{
maxDepth: s.maxDepth,
maxInputSize: s.maxInputSize,
useNumber: s.useNumber,
}
tree, doc, err := parseWithOptions(ctx, data, opts, typeWantsOrder(reflect.TypeOf(v)))
if err != nil {
return err
}
dec := newDecoder()
dec.disallowUnknown = d.disallowUnknown
dec.nodes = indexNodes(doc.Root())
return dec.decode(tree, v)
}
// RejectUnknownFields makes the decode fail when the document contains a
// key with no matching destination struct field. Off by default: unknown
// keys are ignored.
func RejectUnknownFields(v bool) UnmarshalOption {
return func(s *decodeSettings) { s.disallowUnknown = v }
}
// NumbersAsLiterals keeps the numbers of the document as a Number carrying
// the literal the document wrote, so 0x1f, 1_000, +1.0 and inf survive a
// round trip with their spelling intact. A destination of a concrete numeric
// kind still takes the evaluated value; the literal is kept only where a
// Number, or an any, receives it. Off by default: numbers evaluate to
// int64 and float64.
func NumbersAsLiterals(v bool) UnmarshalOption {
return func(s *decodeSettings) { s.useNumber = v }
}
// LocalTimeLocation sets the zone a local date-time is placed in when it
// decodes into a time.Time destination. Without the option a local date-time
// fills only its own wrapper type (LocalDateTime, LocalDate, LocalTime),
// whose embedded time.Time is UTC; with the option, a time.Time destination
// takes the value too, carried in the location given. A nil location restores
// the default.
func LocalTimeLocation(loc *time.Location) UnmarshalOption {
return func(s *decodeSettings) { s.localLoc = loc }
}
// MaxNestingDepth bounds how deeply arrays and inline tables may nest in a
// document the decode accepts. The parser is a recursive descent, so a
// document that nests without bound would exhaust the stack; one that nests
// deeper than the limit is rejected with a SyntaxError naming it instead.
// Use 0 or any negative value for the default of 10000, which no
// hand-written document approaches.
func MaxNestingDepth(depth int) UnmarshalOption {
return func(s *decodeSettings) { s.maxDepth = depth }
}
// MaxInputSize bounds the size of a document the decode accepts, in bytes; a
// larger one is rejected before parsing starts. Use 0 or any negative value
// for no limit, which is the default: the caller already holds the bytes, so
// the size is a policy the caller sets rather than a protection the library
// imposes on its own.
func MaxInputSize(size int) UnmarshalOption {
return func(s *decodeSettings) { s.maxInputSize = size }
}
// Marshaler is the interface implemented by types that can produce a custom
// TOML representation of themselves. MarshalTOML returns a value that Marshal
// then encodes as if the returned value had been passed in its place, which
@@ -269,7 +463,8 @@ type Marshaler interface {
// argument is whatever the parser produced for that key: one of string,
// bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime,
// []any, or map[string]any. A tree built by hand may carry a plain time.Time
// where the parser would put an OffsetDateTime.
// where the parser would put an OffsetDateTime, and NumbersAsLiterals a
// Number.
//
// UnmarshalTOML may parse, inspect, or transform the value however it likes,
// then store the result by mutating its receiver through the standard
@@ -277,7 +472,7 @@ type Marshaler interface {
// reflect.Value.Set or by reassigning fields through a pointer the receiver
// holds).
//
// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the
// UnmarshalTOML is invoked from Unmarshal and its siblings when the
// destination type implements the interface. The decoder does not need to
// consult the concrete return value; whatever the receiver stores is kept.
//
@@ -288,17 +483,31 @@ type Unmarshaler interface {
UnmarshalTOML(data any) error
}
// UnmarshalerContext is Unmarshaler with the decode's context handed in. A
// type that implements both interfaces gets UnmarshalTOMLContext, so a long
// custom decode can abort on cancellation instead of running to completion.
// The context a non-cancellable entry point carries is context.Background,
// never nil.
type UnmarshalerContext interface {
UnmarshalTOMLContext(ctx context.Context, data any) error
}
// Marshal returns the TOML encoding of v. The output is valid TOML 1.1.
// Options tune the emission; with none, the layout groups entries by kind,
// empty arrays emit and sub-tables take the header form.
//
// Marshal traverses v using reflection and applies the following rules:
//
// - The top-level value must be a struct or a map[string]V. Pointers are
// followed; a nil top-level pointer is an error.
// - The top-level value must be a struct, a map[string]V or an OrderedMap
// (or a non-nil pointer to one). A Document writes itself back, and a nil
// one is an error.
// - Struct fields are matched by `toml:"name"` tag (case-insensitive
// fallback to field name; `-` skips). The tag options `omitzero` (skip
// the zero value of the field's type) and `omitempty` (skip an empty
// slice, array, or map) drop a field from the output on encode; the
// decoder ignores them. Anonymous (embedded) fields without a tag are
// fallback to field name; `-` skips). The tag option `omitzero` skips a
// field holding the zero value of its type (a type with an IsZero method
// decides through it), and `omitempty` skips a value that is empty in
// the encoding/json sense: an empty string, a zero number, false, a nil
// pointer or interface, and an empty slice, array or map. The decoder
// ignores both options. Anonymous (embedded) fields without a tag are
// inlined.
// - Maps use sorted keys for deterministic output.
// - Slices and arrays of structs or maps become TOML arrays of tables; a
@@ -309,10 +518,12 @@ type Unmarshaler interface {
// inline table.
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
// and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/
// LocalTime (local variants). A date-time writes its seconds only when the value carries
// them, and drops the trailing zeros of a fractional second.
// - A table element of a value array, and a sub-table inlined by
// Encoder.InlineTables, is written as an inline table, across lines when it
// LocalTime (local variants). A date-time writes its seconds only when
// the value carries them, and drops the trailing zeros of a fractional
// second. A zone offset that is not a whole number of minutes is refused,
// because TOML has no form that carries its seconds.
// - A table element of a value array, and a sub-table the InlineTables
// option inlines, is written as an inline table, across lines when it
// does not fit one.
// - Values implementing Marshaler are encoded by calling MarshalTOML and
// using its result.
@@ -321,76 +532,193 @@ type Unmarshaler interface {
// returns. time.Duration is written in its canonical Go form, `1h30m0s`.
// - nil pointer fields are omitted.
//
// Marshal cannot encode cyclic data structures; passing one will loop until
// the stack overflows. The output is not guaranteed to be byte-identical to
// the input that produced v: comments, whitespace, key order (for maps),
// string quoting style, and the choice between `[table]` headers and inline
// tables are not preserved.
// Marshal rejects a value that nests deeper than 10000 levels with an error
// naming the limit, so cyclic data is reported instead of running the stack
// out. The output is not guaranteed to be byte-identical to the input that
// produced v: comments, whitespace, key order (for maps), string quoting
// style, and the choice between `[table]` headers and inline tables are not
// preserved.
//
// Marshal is equivalent to MarshalContext with context.Background.
func Marshal(v any) ([]byte, error) {
return MarshalContext(context.Background(), v)
func Marshal(v any, opts ...MarshalOption) ([]byte, error) {
return MarshalContext(context.Background(), v, opts...)
}
// A Statement is one top-level statement of a document, what Statements
// yields: a key with its value, a table with its node, or one element of an
// array of tables with its node.
type Statement struct {
// Key is the key as the document wrote it.
Key string
// Value is the value of a key/value statement, and the value map of a
// table statement.
Value any
// Table is the node of a table or array-of-tables statement, carrying the
// written key order and the comments; nil for a plain key/value.
Table *Table
// Index is the element's position when the statement is one element of an
// array of tables, and -1 otherwise.
Index int
}
// Statements reads a TOML document from r and returns an iterator over its
// top-level statements in written order: key/value statements, including a
// value that is an array or an inline table, a [table] header as one
// statement carrying its Table node, and an [[array of tables]] as one
// statement per element, each with the element's node and its Index.
// Iteration stops at the first error, which arrives as the second value, and
// at a false yield: a caller that breaks after the statement it wanted reads
// no further ones.
//
// The reader is consumed in full before the first statement is yielded,
// because the parser scans the source in place; processing the yielded
// statements one at a time is what bounds what the caller holds, and a
// later direct-to-target parse removes the whole-source hold.
func Statements(r io.Reader) iter.Seq2[Statement, error] {
return func(yield func(Statement, error) bool) {
data, err := io.ReadAll(r)
if err != nil {
yield(Statement{Index: -1}, err)
return
}
doc, err := Parse(data)
if err != nil {
yield(Statement{Index: -1}, err)
return
}
for _, e := range doc.Root().Entries() {
// Only an array of tables yields per element, the branch the
// write side takes too: a value array is one statement whatever
// its elements, and an emptied array of tables holds no element
// to yield.
if _, isTables := e.Value().([]map[string]any); isTables && len(e.Elements()) > 0 {
for i, el := range e.Elements() {
if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: el, Index: i}, nil) {
return
}
}
continue
}
if child := e.Table(); child != nil {
if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: child, Index: -1}, nil) {
return
}
continue
}
if !yield(Statement{Key: e.Key(), Value: e.Value(), Index: -1}, nil) {
return
}
}
}
}
// MarshalAppend appends the TOML encoding of v to buf and returns the extended
// buffer, the shape json/v2's MarshalAppendTo and json's MarshalAppend have.
// A failed encoding leaves buf untouched and comes back with a nil slice.
func MarshalAppend(buf []byte, v any, opts ...MarshalOption) ([]byte, error) {
out, err := Marshal(v, opts...)
if err != nil {
return nil, err
}
return append(buf, out...), nil
}
// MarshalContext is the cancellable variant of Marshal.
func MarshalContext(ctx context.Context, v any) ([]byte, error) {
func MarshalContext(ctx context.Context, v any, opts ...MarshalOption) ([]byte, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
return NewEncoder().MarshalContext(ctx, v)
return settingsForEncode(opts).marshal(ctx, v)
}
// An Encoder encodes Go values into TOML.
//
// All options default to the behaviour that passes the toml-test compliance
// suite in both directions:
//
// GroupByKind: true (scalars first, then tables, then arrays of tables)
// OmitEmptyArrays: false (a nil/empty []string slice emits [] as a value;
// a nil/empty []Item struct slice is still skipped)
// LiteralMultilineAt: 0 (always emit the escaped basic form, never a
// literal one)
// InlineTablesAt: 0 (always emit a table header, never an inline
// table)
//
// Use the chainable option methods to opt out. The option state is private;
// callers that need the underlying knobs reach for the methods rather than
// reading or mutating fields.
type Encoder struct {
groupByKind bool // default true; set via (*Encoder).GroupByKind
omitEmptyArrays bool // default false; set via (*Encoder).OmitEmptyArrays
literalMultilineAt int // default 0; set via (*Encoder).UseLiteralMultiline
inlineTablesAt int // default 0; set via (*Encoder).InlineTables
// MarshalWrite encodes v and writes the document to w, the streaming-shaped
// entry the json/v2 vocabulary uses. The options and the behaviour are
// Marshal's.
func MarshalWrite(w io.Writer, v any, opts ...MarshalOption) error {
out, err := Marshal(v, opts...)
if err != nil {
return err
}
if _, err := w.Write(out); err != nil {
return fmt.Errorf("interpres: write: %w", err)
}
return nil
}
// NewEncoder returns an Encoder with default options.
func NewEncoder() *Encoder { return &Encoder{groupByKind: true} }
// A LayoutKind names the layout the encoder writes a document's entries in.
type LayoutKind int
// GroupByKind toggles whether fields at the same TOML level are reordered
// into the group-by-kind layout (scalars first, then tables, then arrays of
// tables). When set to false, the emitter preserves the source declaration
// order (struct field order, or sorted key order for maps).
func (e *Encoder) GroupByKind(v bool) *Encoder {
e.groupByKind = v
return e
const (
// LayoutKindGrouped reorders entries at one level: scalars first, then
// sub-tables, then arrays of tables. The default.
LayoutKindGrouped LayoutKind = iota
// LayoutKindDeclaration preserves the declaration order: struct field
// order, or sorted key order for maps.
LayoutKindDeclaration
)
// A MarshalOption configures one Marshal, MarshalContext, MarshalAppend or
// MarshalWrite call. Options are function values over the private encode
// settings, the shape encoding/json/v2 uses for its own, and compose by
// simple listing:
//
// out, err := interpres.Marshal(cfg,
// interpres.Layout(interpres.LayoutKindDeclaration),
// interpres.InlineTables(60))
type MarshalOption func(*encodeSettings)
// encodeSettings is the option carrier of one encode call. As on the decode
// side, the context arrives as its own argument.
type encodeSettings struct {
cfg encodeConfig
}
func settingsForEncode(opts []MarshalOption) *encodeSettings {
s := &encodeSettings{cfg: encodeConfig{layout: LayoutKindGrouped}}
for _, opt := range opts {
opt(s)
}
return s
}
// marshal runs the encode the settings describe.
func (s *encodeSettings) marshal(ctx context.Context, v any) ([]byte, error) {
enc := newEncoder()
enc.ctx = ctx
enc.opts = s.cfg
if err := enc.encode(v); err != nil {
enc.release()
return nil, err
}
// The output leaves the pooled buffer as a copy, so the next Marshal
// reuses the buffer without touching what the caller holds.
out := slices.Clone(enc.buf.Bytes())
enc.release()
return out, nil
}
// Layout sets the layout the encoder writes a document's entries in:
// LayoutKindGrouped, the default, reorders them scalars first, then tables,
// then arrays of tables; LayoutKindDeclaration preserves declaration order.
// A value the two constants do not name behaves as LayoutKindGrouped.
func Layout(kind LayoutKind) MarshalOption {
return func(s *encodeSettings) { s.cfg.layout = kind }
}
// OmitEmptyArrays opts in to skipping empty (non-nil, length 0) TOML arrays
// of scalars. The default emits them as "key = []". Nil slices and empty
// arrays of tables are already always omitted.
func (e *Encoder) OmitEmptyArrays() *Encoder {
e.omitEmptyArrays = true
return e
func OmitEmptyArrays(v bool) MarshalOption {
return func(s *encodeSettings) { s.cfg.omitEmptyArrays = v }
}
// UseLiteralMultiline sets the length threshold at which a multi-line string
// LiteralMultiline sets the length threshold at which a multi-line string
// is emitted as a literal triple-quoted string instead of the escaped form.
// Use 0 or any negative value to disable (always escaped). The literal form
// is selected only when the value contains an internal newline; otherwise the
// single-line basic form is used regardless of this setting.
func (e *Encoder) UseLiteralMultiline(threshold int) *Encoder {
e.literalMultilineAt = threshold
return e
func LiteralMultiline(threshold int) MarshalOption {
return func(s *encodeSettings) { s.cfg.literalMultilineAt = threshold }
}
// InlineTables sets the size limit, in bytes of the single-line rendering, at
@@ -403,33 +731,25 @@ func (e *Encoder) UseLiteralMultiline(threshold int) *Encoder {
// value array. An inlined table that does not fit the line is written across
// lines, which TOML 1.1 allows.
//
// With GroupByKind(false) the layout is already for presentation only, and an
// With LayoutKindDeclaration the layout is already for presentation only, and an
// inlined table follows the same rule as any other value line: it lands in the
// section of the header that precedes it.
func (e *Encoder) InlineTables(threshold int) *Encoder {
e.inlineTablesAt = threshold
return e
func InlineTables(threshold int) MarshalOption {
return func(s *encodeSettings) { s.cfg.inlineTablesAt = threshold }
}
// Marshal encodes v to TOML bytes. It is equivalent to calling Marshal with v.
// EmitFieldComments turns on printing the comment a field's `toml` tag
// carries in a `comment=` option, above the field's line or header, the
// comments a round trip through the Go type would otherwise drop:
//
// Marshal is equivalent to MarshalContext with context.Background.
func (e *Encoder) Marshal(v any) ([]byte, error) {
return e.MarshalContext(context.Background(), v)
}
// MarshalContext is the cancellable variant of Marshal.
func (e *Encoder) MarshalContext(ctx context.Context, v any) ([]byte, error) {
enc := newEncoder()
enc.ctx = ctx
enc.opts = *e
if err := enc.encode(v); err != nil {
enc.release()
return nil, err
}
// The output leaves the pooled buffer as a copy, so the next Marshal
// reuses the buffer without touching what the caller holds.
out := slices.Clone(enc.buf.Bytes())
enc.release()
return out, nil
// Port int `toml:"port,comment=The port to listen on"`
//
// Go doc comments are not visible to reflection, so the tag is the channel
// that carries the text. Off by default, and a field without a `comment=`
// option prints none. Multi-line comments carry newlines in the tag, each
// line printed with its own "# " marker. The tag's options separate with
// commas, so the comment text itself cannot carry one; the first comma ends
// it.
func EmitFieldComments(v bool) MarshalOption {
return func(s *encodeSettings) { s.cfg.emitFieldComments = v }
}
+279 -2
View File
@@ -5,7 +5,12 @@ package interpres
import (
"errors"
"fmt"
"math"
"os"
"path/filepath"
"reflect"
"slices"
"strings"
"testing"
"time"
@@ -312,7 +317,7 @@ func TestDisallowUnknownFields(t *testing.T) {
}
var strict C
err := NewDecoder().DisallowUnknownFields().Decode(data, &strict)
err := Unmarshal(data, &strict, RejectUnknownFields(true))
if err == nil {
t.Fatal("expected error for unknown field, got nil")
}
@@ -328,7 +333,7 @@ func TestDisallowUnknownFieldsReportsSmallestKey(t *testing.T) {
data := []byte("known = \"x\"\nzeta = 1\nalpha = 2\nmu = 3\n")
for range 20 {
var c C
err := NewDecoder().DisallowUnknownFields().Decode(data, &c)
err := Unmarshal(data, &c, RejectUnknownFields(true))
if err == nil {
t.Fatal("expected error for unknown fields")
}
@@ -726,3 +731,275 @@ func TestParseNestingLimit(t *testing.T) {
t.Errorf("Msg = %q, want it to name the nesting limit", se.Msg)
}
}
func TestParseFile(t *testing.T) {
path := filepath.Join(t.TempDir(), "config.toml")
if err := os.WriteFile(path, []byte("port = 8080\n"), 0o644); err != nil {
t.Fatal(err)
}
doc, err := ParseFile(path)
if err != nil {
t.Fatal(err)
}
if got := doc.Map()["port"]; got != int64(8080) {
t.Errorf("port = %v, want 8080", got)
}
_, err = ParseFile(filepath.Join(t.TempDir(), "missing.toml"))
if err == nil || !strings.Contains(err.Error(), "missing.toml") {
t.Errorf("read error = %v, want it to name the file", err)
}
bad := filepath.Join(t.TempDir(), "broken.toml")
if err := os.WriteFile(bad, []byte("port =\n"), 0o644); err != nil {
t.Fatal(err)
}
_, err = ParseFile(bad)
if err == nil || !strings.Contains(err.Error(), "broken.toml") {
t.Errorf("parse error = %v, want it to name the file", err)
}
s, ok := errors.AsType[*SyntaxError](err)
if !ok || s.Line != 1 {
t.Errorf("parse error = %v, want a SyntaxError with line 1 inside", err)
}
}
func TestValid(t *testing.T) {
if err := Valid([]byte("a = 1\n[t]\nb = 2\n")); err != nil {
t.Errorf("Valid(valid) = %v, want nil", err)
}
err := Valid([]byte("a = \n"))
if err == nil {
t.Fatal("Valid(invalid) = nil, want an error")
}
if _, ok := errors.AsType[*SyntaxError](err); !ok {
t.Errorf("Valid(invalid) = %v, want a SyntaxError", err)
}
}
func TestParseAsAndNewSchema(t *testing.T) {
type Config struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
cfg, err := ParseAs[Config]([]byte("host = \"db\"\nport = 5432\n"))
if err != nil {
t.Fatal(err)
}
if cfg.Host != "db" || cfg.Port != 5432 {
t.Errorf("decoded %+v", cfg)
}
if _, err := ParseAs[Config]([]byte("port =\n")); err == nil {
t.Error("ParseAs(invalid) succeeded, want an error and the zero value")
}
NewSchema[Config]()
if _, ok := structSchemaCache.Load(reflect.TypeFor[Config]()); !ok {
t.Error("NewSchema left no schema in the cache")
}
NewSchema[map[string]any]() // must not panic
}
func TestZeroOffsetRoundTrip(t *testing.T) {
// A document may write a zero offset as +00:00; the tree must hold the
// same value after a round trip, because the written form is "Z" either
// way.
src := []byte("a = 1979-05-27T07:32:00+00:00\n")
tree, err := ParseMap(src)
if err != nil {
t.Fatal(err)
}
out, err := Marshal(tree)
if err != nil {
t.Fatal(err)
}
re, err := ParseMap(out)
if err != nil {
t.Fatal(err)
}
if !reflect.DeepEqual(tree, re) {
t.Errorf("round trip changed the tree: %#v vs %#v", tree, re)
}
if got := tree["a"].(OffsetDateTime).String(); got != "1979-05-27T07:32Z" {
t.Errorf("a = %q, want 1979-05-27T07:32Z", got)
}
}
func TestStatements(t *testing.T) {
src := strings.NewReader(`title = "demo"
port = 8080
[server]
host = "127.0.0.1"
[[items]]
name = "a"
[[items]]
name = "b"
`)
var lines []string
for stmt, err := range Statements(src) {
if err != nil {
t.Fatal(err)
}
switch {
case stmt.Index >= 0:
lines = append(lines, fmt.Sprintf("%s #%d", stmt.Key, stmt.Index))
case stmt.Table != nil:
lines = append(lines, fmt.Sprintf("[%s] %v", stmt.Key, stmt.Table.Keys()))
default:
lines = append(lines, fmt.Sprintf("%s = %v", stmt.Key, stmt.Value))
}
}
want := []string{
`title = demo`,
`port = 8080`,
`[server] [host]`,
`items #0`,
`items #1`,
}
if !slices.Equal(lines, want) {
t.Errorf("statements =\n%v\nwant:\n%v", lines, want)
}
t.Run("breaking stops the iteration", func(t *testing.T) {
src := strings.NewReader("a = 1\nb = 2\nc = 3\n")
count := 0
for range Statements(src) {
count++
break
}
if count != 1 {
t.Errorf("iterated %d statements after break, want 1", count)
}
})
t.Run("a parse error arrives as the second value", func(t *testing.T) {
for stmt, err := range Statements(strings.NewReader("broken =\n")) {
if err == nil {
t.Fatalf("statement %+v without an error", stmt)
}
if _, ok := errors.AsType[*SyntaxError](err); !ok {
t.Errorf("err = %v, want a SyntaxError", err)
}
break
}
})
}
func TestParseCRLFDocument(t *testing.T) {
tree, err := ParseMap([]byte("a = 1\r\nb = 2\r\n[t]\r\nc = \"x\"\r\n"))
if err != nil {
t.Fatal(err)
}
if tree["a"] != int64(1) || tree["b"] != int64(2) {
t.Errorf("tree = %v", tree)
}
}
// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U:
// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the
// greatest scalar value parses.
func TestParseUnicodeEscapeBoundaries(t *testing.T) {
bad := []struct {
name string
in string
}{
{"high surrogate", `a = "\ud800"`},
{"low surrogate", `a = "\udfff"`},
{"past the greatest scalar", `a = "\U00110000"`},
{"signed short escape", `a = "\u+041"`},
{"negative long escape", `a = "\U-0000001"`},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := Parse([]byte(tt.in))
if err == nil {
t.Fatalf("Parse accepted %q", tt.in)
}
})
}
tree, err := ParseMap([]byte("a = \"\\U0010FFFF\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "􏿿" {
t.Errorf("a = %q", tree["a"])
}
}
// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a
// date-time with a component out of range is rejected as a date-time, not
// left to the number decoder's complaint.
func TestParseRejectsOutOfRangeDateTimes(t *testing.T) {
bad := []struct {
name string
in string
}{
{"hour 24", "a = 1979-05-27T24:00:00Z"},
{"minute 60", "a = 1979-05-27T07:60:00Z"},
{"second 60", "a = 1979-05-27T07:32:60Z"},
{"month 13", "a = 1979-13-27T07:32:00Z"},
{"day 32", "a = 1979-05-32T07:32:00Z"},
{"february the thirtieth", "a = 1979-02-30"},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseMap([]byte(tt.in))
if err == nil {
t.Fatalf("ParseMap accepted %q", tt.in)
}
if !strings.Contains(err.Error(), "invalid date-time") {
t.Errorf("err = %v, want the date-time complaint", err)
}
})
}
}
// TestParseMultilineStringEdges pins the carriage-return and delimiter rules
// of multi-line strings: a bare CR right after the opening delimiter is the
// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the
// content survives.
func TestParseMultilineStringEdges(t *testing.T) {
_, err := ParseMap([]byte("a = \"\"\"\rX\"\"\""))
if err == nil || !strings.Contains(err.Error(), "bare carriage return") {
t.Errorf("err = %v, want the bare-CR error after the delimiter", err)
}
tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "X\r\nY" {
t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"])
}
}
// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes
// before the closing delimiter of a basic multi-line string are content, and
// more than five are the error.
func TestParseMultilineBasicDelimiterRuns(t *testing.T) {
tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != `end"` {
t.Errorf("a = %q", tree["a"])
}
_, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\""))
if err == nil || !strings.Contains(err.Error(), "too many") {
t.Errorf("err = %v, want the too-many-delimiters error", err)
}
}
// TestParseLineEndingBackslashEdges pins the line-ending backslash at the
// very end of the input and before a bare CR.
func TestParseLineEndingBackslashEdges(t *testing.T) {
bad := []string{
"a = \"\"\"x \\\\",
"a = \"\"\"x \\\\\rZ\"\"\"",
}
for _, in := range bad {
if _, err := ParseMap([]byte(in)); err == nil {
t.Errorf("ParseMap accepted %q", in)
}
}
}
+42 -3
View File
@@ -1,7 +1,7 @@
# interpres.
#
# Everything below the variable block is the standard recipe set from the `justfile`
# skill, identical in every repository; project values live in the variable block only.
# Everything below the variable block is the standard recipe set, identical in
# every repository; project values live in the variable block only.
binary := "interpres-decode"
package := "./cmd/interpres-decode"
@@ -94,12 +94,51 @@ dev:
# Runs the official toml-test compliance suite in both directions, decoder and encoder, against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
toml-test: build
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
coverage-html: test
go tool cover -html=coverage.out -o coverage.html
# Cross-compile smoke: the library and the command build for the foreign architectures and the browser and edge runtimes; not a gate, it is a hand-run convenience and runs std-lib only.
cross:
GOARCH=arm64 go build ./...
GOARCH=loong64 go build ./...
GOARCH=riscv64 go build ./...
GOOS=js GOARCH=wasm go build ./...
GOOS=wasip1 GOARCH=wasm go build ./...
GOARCH=arm64 CGO_ENABLED=0 go build -o /dev/null {{package}}
GOARCH=loong64 CGO_ENABLED=0 go build -o /dev/null {{package}}
GOARCH=riscv64 CGO_ENABLED=0 go build -o /dev/null {{package}}
# Runs the example program under examples/basic; not standard because `run` runs the adapter, and an example is documentation, not the product.
example:
go run ./examples/basic
# The release pre-flight, in one command: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release. Not a gate, it is the checklist before a release may even be discussed.
release-check version:
#!/usr/bin/env perl
# The version arrives through the recipe interpolation: just does not hand
# positional arguments to a shebang script's @ARGV.
my $version = "{{version}}";
$version =~ m{\Av?\d+\.\d+\.\d+\z} or die qq{usage: just release-check X.Y.Z\n};
my $branch = qx{git rev-parse --abbrev-ref HEAD};
chomp $branch;
$branch eq q{development} or die qq{release-check: on '$branch', cut releases from development\n};
my $dirty = qx{git status --porcelain};
$dirty eq q{} or die qq{release-check: the working tree is dirty\n};
system(qw{git fetch origin}) == 0 or die qq{release-check: git fetch failed\n};
my $local = qx{git rev-parse development};
my $remote = qx{git rev-parse origin/development};
$local eq $remote or die qq{release-check: development is out of sync with origin\n};
my $changelog = do { open(my $fh, q{<}, q{CHANGELOG.md}) or die qq{release-check: cannot read CHANGELOG.md: $!\n}; local $/; <$fh> };
$changelog =~ m{## \[development\]\n\n### \w+} or die qq{release-check: the [development] section of CHANGELOG.md is missing or empty\n};
print qq{branch, tree, sync and changelog verified; running the gates\n};
system(qw{just gates}) == 0 or die qq{release-check: the gates failed\n};
print qq{release-check: ready to release $version\n};
print qq{after tagging, verify the /v2 module resolves through the proxy:\n};
print qq{ cd \$(mktemp -d) && go mod init t && GOPRIVATE= GOPROXY=https://proxy.golang.org go get sourcedock.dev/petrbalvin/interpres/v2\@$version\n};
# Compares the toml-test counts the documentation names with the live suite run; not standard, it exists because a corpus change used to be corrected by hand.
docs-drift:
perl scripts/docs-drift.pl
+140
View File
@@ -0,0 +1,140 @@
.TH INTERPRES-DECODE 1 "2026-09-22" "interpres 2.0.0" "User Commands"
.SH NAME
interpres-decode \- TOML validator and toml-test harness adapter
.SH SYNOPSIS
.B interpres-decode
[\fIFLAGS\fR]
.br
.B interpres-decode
.B \-\-encode
.br
.B interpres-decode
.B \-\-validate
[\fIFILE\fR...]
.br
.B interpres-decode
.B \-\-validate
[\fIDIRECTORY\fR...]
.br
.B interpres-decode
.B \-\-json
.br
.B interpres-decode
.B \-\-struct
.br
.B interpres-decode
.B \-\-schema
\fITYPE\fR
\fIFILE.go\fR
.br
.B interpres-decode
.B \-\-version
.SH DESCRIPTION
.B interpres-decode
is the toml-test harness adapter in both directions and a TOML validator.
Without a mode flag it reads one TOML document from standard input and writes
the toml-test tagged-JSON representation to standard output.
.B \-\-encode
reads a tagged-JSON description from standard input and writes the TOML
document it describes.
.B \-\-validate
parses each named file, or standard input when none are named, and prints one
line per invalid document to standard error; a named directory is walked for
.B .toml
files, every one validated, and the walk closes with a summary on standard
error naming the counts. The name
.B \-
means standard input.
.B \-\-json
prints plain indented JSON instead of the tagged form; it shapes the decoding
output only, so it is rejected together with the mode flags.
.B \-\-struct
prints a Go struct definition inferred from the document on standard input.
.B \-\-schema
writes a TOML template for the struct type
\fITYPE\fR
declared in the Go source file
\fIFILE.go\fR,
taking the key names, comments and defaults from the fields' tags.
.B \-\-version
prints the binary's version and exits.
.PP
The mode flags
.BR \-\-validate ,
.BR \-\-encode ,
.B \-\-struct
and
.B \-\-schema
cannot be combined.
.SH OPTIONS
.TP
.B \-\-validate
Validate the documents instead of emitting tagged JSON.
.TP
.B \-\-encode
Read tagged JSON from standard input and write TOML instead.
.TP
.B \-\-json
With the default mode, print plain indented JSON instead of tagged JSON.
.TP
.B \-\-struct
Infer a Go struct definition from the document on standard input and print it.
.TP
.BI \-\-schema " TYPE"
Write a TOML template for the struct type \fITYPE\fR; the Go source file
follows as the first argument.
.TP
.B \-\-version
Print the version and exit.
.TP
.B \-\-help
Print the usage.
.SH EXIT STATUS
.TP
.B 0
The document parsed and the output was written; in validate mode, every
document parsed.
.TP
.B 1
Adapter: a parse error. Validate: at least one document is invalid. Struct:
the document on standard input failed to parse.
.TP
.B 2
A usage error, a read or write failure, malformed tagged JSON, or a value
with no TOML representation.
.SH EXAMPLES
Decode a document into tagged JSON:
.PP
.nf
.RS
echo 'title = "hello"' | interpres-decode
.RE
.fi
.PP
Validate a directory of configuration, with the summary:
.PP
.nf
.RS
interpres-decode \-\-validate configs/
.RE
.fi
.PP
Infer a Go type from a document:
.PP
.nf
.RS
interpres-decode \-\-struct < config.toml > config.go
.RE
.fi
.PP
Write the template back from the type:
.PP
.nf
.RS
interpres-decode \-\-schema Config config.go
.RE
.fi
.SH SEE ALSO
The repository's
.B docs/CLI.md
carries the full reference, including the tagged-JSON wire format.
+45
View File
@@ -10,6 +10,51 @@ import (
"strings"
)
// A Number holds a TOML number as the literal the document wrote it with:
// 0x1f, 1_000, +1.0, inf. The NumbersAsLiterals option decodes integers and
// floats into
// it, so a round trip through the value tree keeps the spelling instead of a
// normalised one, and Marshal writes the literal back as it is.
//
// Number is a string type, the shape encoding/json.Number has: the literal is
// carried, not evaluated. Float64 and Int64 evaluate it on demand, and a
// destination of another numeric kind takes the evaluated value through the
// ordinary conversion rules.
type Number string
// Float64 returns the value as a float64. An integer or radix literal
// converts; a literal that is not a valid TOML number is an error.
func (n Number) Float64() (float64, error) {
v, err := decodeNumber(string(n))
if err != nil {
return 0, fmt.Errorf("interpres: %w", err)
}
switch v := v.(type) {
case float64:
return v, nil
case int64:
return float64(v), nil
}
return 0, fmt.Errorf("interpres: %q is not a number", n)
}
// Int64 returns the value as an int64. A float literal is an error, however
// whole its value, and so is a literal that is not a valid TOML number.
func (n Number) Int64() (int64, error) {
v, err := decodeNumber(string(n))
if err != nil {
return 0, fmt.Errorf("interpres: %w", err)
}
i, ok := v.(int64)
if !ok {
return 0, fmt.Errorf("interpres: %q is not an integer", n)
}
return i, nil
}
// String returns the literal itself.
func (n Number) String() string { return string(n) }
// decodeNumber parses a bare numeric token under strict TOML rules: no leading
// zeros, underscores only between digits, prefixed radixes without a sign, and
// floats with explicit fraction/exponent digits.
+199
View File
@@ -0,0 +1,199 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"fmt"
"maps"
"reflect"
"slices"
"sync"
)
// An OrderedMap is a string-keyed table that remembers the order its keys
// were set in, the shape a map[string]any cannot carry. Marshal writes a
// table of its own kind in that order, and decoding a document into one
// fills it in the order the document wrote the keys, where a map
// destination carries no order at all. The values are untyped, the shape
// the parser produces, so a nested table inside an OrderedMap is a plain
// map[string]any; the order is kept at the level the OrderedMap sits at.
//
// The zero value is an empty table ready for use.
type OrderedMap struct {
keys []string
values map[string]any
}
var orderedMapType = reflect.TypeFor[OrderedMap]()
// NewOrderedMap returns an empty OrderedMap.
func NewOrderedMap() *OrderedMap { return &OrderedMap{} }
// Set stores value under key. A key the table already has keeps its position
// and takes the new value; a new one joins the end.
func (m *OrderedMap) Set(key string, value any) {
if m.values == nil {
m.values = make(map[string]any, 4)
}
if _, ok := m.values[key]; !ok {
m.keys = append(m.keys, key)
}
m.values[key] = value
}
// Get returns the value under key, and whether the table has one.
func (m *OrderedMap) Get(key string) (any, bool) {
v, ok := m.values[key]
return v, ok
}
// Delete removes key. A later Set of the same key appends it to the end
// again.
func (m *OrderedMap) Delete(key string) {
if _, ok := m.values[key]; !ok {
return
}
delete(m.values, key)
m.keys = slices.DeleteFunc(m.keys, func(k string) bool { return k == key })
}
// Keys returns the keys in the order they were set.
func (m *OrderedMap) Keys() []string { return m.keys }
// Len returns the number of keys.
func (m *OrderedMap) Len() int { return len(m.keys) }
// Range calls f for every key in order, stopping when f returns false.
func (m *OrderedMap) Range(f func(key string, value any) bool) {
for _, k := range m.keys {
if !f(k, m.values[k]) {
return
}
}
}
// Map returns the values as a plain map, which carries no order. It is the
// view Marshal's Document-free callers need.
func (m *OrderedMap) Map() map[string]any { return m.values }
// --- decode: the order the document wrote ----------------------------------
// wantsOrderCache holds whether a destination type mentions OrderedMap
// anywhere a decode can reach. One computed answer per type, the same
// trade-off structSchemaCache makes.
var wantsOrderCache sync.Map // reflect.Type -> bool
// typeWantsOrder reports whether decoding into t can reach an OrderedMap, in
// which case the parse has to build the node tree the key order is read
// from. Structs walk their exported fields, and pointers, slices, arrays and
// maps walk their element; anything else holds no OrderedMap.
func typeWantsOrder(t reflect.Type) bool {
if t == nil {
return false
}
if v, ok := wantsOrderCache.Load(t); ok {
return v.(bool)
}
r := scanWantsOrder(t, make(map[reflect.Type]bool))
v, _ := wantsOrderCache.LoadOrStore(t, r)
return v.(bool)
}
func scanWantsOrder(t reflect.Type, seen map[reflect.Type]bool) bool {
for {
if t == orderedMapType {
return true
}
if seen[t] {
return false
}
seen[t] = true
switch t.Kind() {
case reflect.Pointer, reflect.Slice, reflect.Array, reflect.Map:
t = t.Elem()
case reflect.Struct:
for f := range t.Fields() {
if f.PkgPath != "" {
continue
}
if scanWantsOrder(f.Type, seen) {
return true
}
}
return false
default:
return false
}
}
}
// nodes maps a table's value map to its node, the index the decoder reads
// the written key order from. The key is the map header's runtime pointer,
// the one identity a map value offers; the nodes share their maps with the
// value tree, so one lookup per table is exact.
type nodeIndex map[uintptr]*Table
// indexNodeIndex walks a document's node tree into an index. A nil tree
// gives a nil index, which every lookup answers with nil.
func indexNodes(t *Table) nodeIndex {
if t == nil {
return nil
}
idx := nodeIndex{}
var walk func(t *Table)
walk = func(t *Table) {
idx[reflect.ValueOf(t.values).Pointer()] = t
for _, e := range t.entries {
if e.child != nil {
walk(e.child)
}
// The elements of a value array carry a node only where an element
// is an inline table; the rest are nil.
for _, el := range e.elements {
if el != nil {
walk(el)
}
}
}
}
walk(t)
return idx
}
// nodeOf returns the node a value table was parsed into, or nil when the
// parse built no node tree, which is the ordinary decode's shape. A tree
// built by hand carries no nodes either.
func (d *decoder) nodeOf(tbl map[string]any) *Table {
return d.nodes[reflect.ValueOf(tbl).Pointer()]
}
// fillOrderedMap decodes a parsed table into an OrderedMap destination,
// taking the keys in the order the document wrote them. A table with no
// node, which is what a hand-built tree or a ParseMap result offers, fills
// in sorted key order, the deterministic order a map can offer.
func (d *decoder) fillOrderedMap(tbl map[string]any, dst reflect.Value) error {
if !dst.CanAddr() {
return fmt.Errorf("interpres: cannot decode into an OrderedMap that is not addressable")
}
om := dst.Addr().Interface().(*OrderedMap)
if om.values == nil {
om.values = make(map[string]any, len(tbl))
}
keys := slices.Sorted(maps.Keys(tbl))
if node := d.nodeOf(tbl); node != nil {
keys = node.Keys()
}
for _, key := range keys {
val, ok := tbl[key]
if !ok {
continue
}
elem := reflect.New(reflect.TypeFor[any]()).Elem()
if err := d.assign(val, elem); err != nil {
return newDecodeError(key, err)
}
om.Set(key, elem.Interface())
}
return nil
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"context"
"slices"
"testing"
)
func TestOrderedMapBasics(t *testing.T) {
m := NewOrderedMap()
if m.Len() != 0 {
t.Fatalf("fresh map holds %d keys", m.Len())
}
m.Set("b", 1)
m.Set("a", 2)
m.Set("c", 3)
if got := m.Keys(); !slices.Equal(got, []string{"b", "a", "c"}) {
t.Errorf("keys = %v, want [b a c]", got)
}
if v, ok := m.Get("a"); !ok || v != 2 {
t.Errorf("a = %v, %v", v, ok)
}
m.Set("a", 9)
if got := m.Keys(); !slices.Equal(got, []string{"b", "a", "c"}) {
t.Errorf("keys after replace = %v, want the position kept", got)
}
if v, _ := m.Get("a"); v != 9 {
t.Errorf("a = %v, want 9", v)
}
seen := ""
m.Range(func(key string, value any) bool {
seen += key
return key != "a"
})
if seen != "ba" {
t.Errorf("range visited %q, want \"ba\"", seen)
}
m.Delete("b")
m.Delete("missing")
if got := m.Keys(); !slices.Equal(got, []string{"a", "c"}) {
t.Errorf("keys after delete = %v, want [a c]", got)
}
m.Delete("c")
m.Set("c", 3)
if got := m.Keys(); !slices.Equal(got, []string{"a", "c"}) {
t.Errorf("re-set key = %v, want it appended as [a c]", got)
}
}
func TestMarshalOrderedMap(t *testing.T) {
t.Run("top level keeps the order", func(t *testing.T) {
m := NewOrderedMap()
m.Set("zebra", int64(1))
m.Set("alpha", "x")
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
want := "zebra = 1\nalpha = \"x\"\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
t.Run("a pointer top level does the same", func(t *testing.T) {
m := &OrderedMap{}
m.Set("second", true)
m.Set("first", int64(2))
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
if string(out) != "second = true\nfirst = 2\n" {
t.Errorf("output %q", out)
}
})
t.Run("a struct field keeps the order as a table", func(t *testing.T) {
type Cfg struct {
Title string `toml:"title"`
Extra *OrderedMap `toml:"extra"`
}
m := &OrderedMap{}
m.Set("late", int64(1))
m.Set("early", int64(2))
out, err := Marshal(Cfg{Title: "t", Extra: m})
if err != nil {
t.Fatal(err)
}
want := "title = \"t\"\n\n[extra]\nlate = 1\nearly = 2\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
t.Run("inline form keeps the order too", func(t *testing.T) {
m := NewOrderedMap()
m.Set("zebra", int64(1))
m.Set("alpha", int64(2))
out, err := Marshal(map[string]any{"t": m}, InlineTables(60))
if err != nil {
t.Fatal(err)
}
if string(out) != "t = {zebra = 1, alpha = 2}\n" {
t.Errorf("output %q", out)
}
})
t.Run("an array of tables keeps each element's order", func(t *testing.T) {
type Cfg struct {
Items []*OrderedMap `toml:"items"`
}
a, b := NewOrderedMap(), NewOrderedMap()
a.Set("y", int64(1))
a.Set("x", int64(2))
b.Set("n", int64(3))
out, err := Marshal(Cfg{Items: []*OrderedMap{a, b}})
if err != nil {
t.Fatal(err)
}
want := "[[items]]\ny = 1\nx = 2\n\n[[items]]\nn = 3\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
t.Run("a nil value is skipped", func(t *testing.T) {
m := NewOrderedMap()
m.Set("gone", nil)
m.Set("here", int64(1))
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
if string(out) != "here = 1\n" {
t.Errorf("output %q", out)
}
})
}
func TestDecodeOrderedMap(t *testing.T) {
t.Run("keys come back in written order", func(t *testing.T) {
doc := []byte("zebra = 1\nmiddle = \"m\"\nalpha = true\n")
var m OrderedMap
if err := Unmarshal(doc, &m); err != nil {
t.Fatal(err)
}
if got := m.Keys(); !slices.Equal(got, []string{"zebra", "middle", "alpha"}) {
t.Fatalf("keys = %v", got)
}
if v, _ := m.Get("middle"); v != "m" {
t.Errorf("middle = %#v", v)
}
})
t.Run("a nested table keeps the table order", func(t *testing.T) {
type Cfg struct {
Ports []int `toml:"ports"`
DB *OrderedMap `toml:"db"`
}
doc := []byte("ports = [1, 2]\n\n[db]\nslow = 1\nfast = 2\n")
var cfg Cfg
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if got := cfg.DB.Keys(); !slices.Equal(got, []string{"slow", "fast"}) {
t.Errorf("db keys = %v", got)
}
})
t.Run("an array of tables fills in order", func(t *testing.T) {
var m OrderedMap
doc := []byte("b = 1\n[[items]]\nname = \"x\"\n[[items]]\nname = \"y\"\na = 2\n")
if err := Unmarshal(doc, &m); err != nil {
t.Fatal(err)
}
if got := m.Keys(); !slices.Equal(got, []string{"b", "items"}) {
t.Errorf("keys = %v, want [b items]", got)
}
elems, ok := m.values["items"].([]map[string]any)
if !ok || len(elems) != 2 {
t.Fatalf("items = %#v", m.values["items"])
}
if elems[1]["name"] != "y" {
t.Errorf("second element = %#v", elems[1])
}
})
t.Run("the sorted fallback needs a tree without nodes", func(t *testing.T) {
// Unmarshal and Decode build the node tree whenever the destination can
// reach an OrderedMap, so the sorted fallback is only reachable from a
// tree that never had one.
tree, _, err := parseWithOptions(context.Background(), []byte("b = 1\na = 2\n"), parseOptions{}, false)
if err != nil {
t.Fatal(err)
}
var m OrderedMap
if err := newDecoder().decode(tree, &m); err != nil {
t.Fatal(err)
}
if got := m.Keys(); !slices.Equal(got, []string{"a", "b"}) {
t.Errorf("keys = %v, want the sorted [a b]", got)
}
})
t.Run("the order survives a round trip", func(t *testing.T) {
doc := []byte("z = 1\na = 2\nm = 3\n")
var m OrderedMap
if err := Unmarshal(doc, &m); err != nil {
t.Fatal(err)
}
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
if string(out) != "z = 1\na = 2\nm = 3\n" {
t.Errorf("output:\n%q", out)
}
})
}
+110 -28
View File
@@ -4,6 +4,7 @@
package interpres
import (
"bytes"
"context"
"fmt"
"strconv"
@@ -36,6 +37,10 @@ type parser struct {
maxDepth int
depth int
// useNumber leaves the numbers a Number carries the literal, instead of
// the evaluated int64 or float64 the tree holds by default.
useNumber bool
root map[string]any
current map[string]any
headers map[string]bool
@@ -43,6 +48,13 @@ type parser struct {
dotted map[string]bool
arrays map[string]bool
// scopeMarks records the definition-map entries added under an array of
// tables, keyed by that array's path, so a new element's reset drops
// exactly what the previous element added. Without it the reset scans
// every map for the prefix, which a document with many elements and many
// definitions outside them turns quadratic.
scopeMarks map[string][]string
currentPath []string
// keys interns key strings: a document that repeats a key across
@@ -191,6 +203,7 @@ func (p *parser) markHeader(pk string) {
p.headers = make(map[string]bool, 4)
}
p.headers[pk] = true
p.trackScope(pk)
}
func (p *parser) markFrozen(pk string) {
@@ -198,6 +211,7 @@ func (p *parser) markFrozen(pk string) {
p.frozen = make(map[string]bool, 4)
}
p.frozen[pk] = true
p.trackScope(pk)
}
func (p *parser) markDotted(pk string) {
@@ -205,6 +219,7 @@ func (p *parser) markDotted(pk string) {
p.dotted = make(map[string]bool, 4)
}
p.dotted[pk] = true
p.trackScope(pk)
}
func (p *parser) markArray(pk string) {
@@ -212,6 +227,22 @@ func (p *parser) markArray(pk string) {
p.arrays = make(map[string]bool, 2)
}
p.arrays[pk] = true
p.trackScope(pk)
}
// trackScope records a definition entry under every array of tables it falls
// inside, so resetScopeUnder can drop it when a later element opens. An entry
// under no array, such as every definition before the first header, needs no
// record: no reset can ever name it.
func (p *parser) trackScope(pk string) {
for arr := range p.arrays {
if strings.HasPrefix(pk, arr+"\x00") {
if p.scopeMarks == nil {
p.scopeMarks = make(map[string][]string, 2)
}
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
}
}
}
// internKey returns the shared string for key bytes. The lookup works on the
@@ -477,9 +508,14 @@ func (p *parser) parseKeyValue() error {
if p.doc != nil {
node := p.currentNode
if len(rest) > 0 {
// The tables a dotted key builds hold the position of a line, so
// the write side marks them and gives each leaf back as a dotted
// key rather than a header that would swallow the lines after it.
node = node.addTable(first, dests[0])
node.dotted = true
for i, k := range rest[:len(rest)-1] {
node = node.addTable(k, dests[i+1])
node.dotted = true
}
}
_, inline := val.(map[string]any)
@@ -565,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) {
// resetScopeUnder forgets the definition records nested under key, which
// belong to the previous element of an array of tables: headers, frozen
// inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element.
// fresh in the new element. The records to drop are the ones the element
// added, which scopeMarks holds; the array's own entry, and everything
// outside it, keep their place.
func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00"
p.resetMapUnder(p.headers, prefix)
p.resetMapUnder(p.frozen, prefix)
p.resetMapUnder(p.dotted, prefix)
p.resetMapUnder(p.arrays, prefix)
}
// resetMapUnder deletes the entries m holds under prefix. An empty or
// unallocated map holds none, so the common case walks nothing.
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
if len(m) == 0 {
return
pk := pathKey(key)
for _, k := range p.scopeMarks[pk] {
delete(p.headers, k)
delete(p.frozen, k)
delete(p.dotted, k)
delete(p.arrays, k)
}
for k := range m {
if strings.HasPrefix(k, prefix) {
delete(m, k)
}
if p.scopeMarks != nil {
p.scopeMarks[pk] = nil
}
}
@@ -648,7 +678,7 @@ func (p *parser) parseKeyComponent() (string, error) {
// does not decode names that, before any grammar message can.
if !p.eof() && p.peek() >= utf8.RuneSelf {
if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in key")
return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos)
}
}
if p.pos == start {
@@ -703,7 +733,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value")
}
if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value")
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
}
// A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -714,13 +744,22 @@ func (p *parser) parseAtom() (any, error) {
tok = tok + " " + string(p.src[timeStart:p.pos])
}
}
if v, ok := parseDateTime(tok); ok {
v, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
return v, nil
}
v, err := decodeNumber(tok)
if err != nil {
return nil, p.errf("%s", err)
}
// The token's shape is validated either way; NumbersAsLiterals only keeps the
// literal instead of the evaluated value.
if p.useNumber {
return Number(tok), nil
}
return v, nil
}
@@ -748,6 +787,20 @@ func hasHighByte(s string) bool {
return false
}
// invalidUTF8Offset returns the offset of the first byte in s that does not
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
// that is invalid rather than the end of the token around it.
func invalidUTF8Offset(s string) int {
for i := 0; i < len(s); {
r, size := utf8.DecodeRuneInString(s[i:])
if r == utf8.RuneError && size == 1 {
return i
}
i += size
}
return -1
}
// --- strings ---------------------------------------------------------------
func (p *parser) parseBasicString() (string, error) {
@@ -876,7 +929,7 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 in string")
return p.errf("invalid UTF-8 in string at byte offset %d", p.pos)
}
p.pos += size
b.WriteRune(r)
@@ -885,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed.
// A newline immediately after the opening delimiter is trimmed, and it is
// a newline: a bare CR here is the bare-CR error like anywhere else, not
// a newline to trim.
if !p.eof() && p.peek() == '\r' {
if next, ok := p.peekAt(1); !ok || next != '\n' {
return "", p.errf("bare carriage return is not allowed in a string")
}
p.pos++
}
if !p.eof() && p.peek() == '\n' {
@@ -1037,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) {
}
hex := string(p.src[p.pos : p.pos+n])
p.pos += n
v, err := strconv.ParseInt(hex, 16, 64)
// ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
// read would let "\U-0000001" through the range checks below only to
// write U+FFFD for a document the grammar rejects.
v, err := strconv.ParseUint(hex, 16, 32)
if err != nil {
return 0, p.errf("invalid unicode escape \\%s", hex)
}
@@ -1069,7 +1130,15 @@ func (p *parser) parseArray() (val any, err error) {
}
}()
}
for {
for i := 0; ; i++ {
// A container the size of memory should answer cancellation inside the
// value, not only between statements, so the element loops check the
// context on their own cadence.
if i%ctxCheckInterval == 0 {
if err := p.checkCtx(); err != nil {
return nil, err
}
}
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
@@ -1138,7 +1207,12 @@ func (p *parser) parseInlineTable() (val any, err error) {
p.pos++
return tbl, nil
}
for {
for i := 0; ; i++ {
if i%ctxCheckInterval == 0 {
if err := p.checkCtx(); err != nil {
return nil, err
}
}
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
@@ -1384,7 +1458,7 @@ func (p *parser) skipComment() (string, error) {
default:
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in comment")
return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos)
}
p.pos += size
}
@@ -1442,13 +1516,21 @@ func (p *parser) expectLineEnd() error {
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 after value")
return p.errf("invalid UTF-8 after value at byte offset %d", p.pos)
}
return p.errf("unexpected %q after value", string(r))
}
// errf builds the SyntaxError with the position the scan stopped at: the line,
// the byte offset in the input, and the 1-based column on that line. The
// offset is the cursor, which on an escape or a delimiter run sits just after
// the bytes that caused the complaint; SourceLine renders the caret there.
func (p *parser) errf(format string, args ...any) error {
return &SyntaxError{Line: p.line, Msg: fmt.Sprintf(format, args...)}
col := p.pos + 1
if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 {
col = p.pos - start
}
return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)}
}
// pathKey joins key components with a NUL separator so a dotted path can be
+47
View File
@@ -0,0 +1,47 @@
#!/usr/bin/env perl
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
# SPDX-License-Identifier: MIT
# docs-drift compares the toml-test suite counts the documentation names with
# the run this repository produces now. A corpus change moves the counts, and
# README.md and docs/ARCHITECTURE.md quote them; this is the check that keeps
# the quotation honest. Builtins only, and the toml-test binary on PATH.
use v5.40;
my $out = qx{toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1 2>&1};
die "docs-drift: toml-test failed to run; build the adapter first (just build)\n"
if !defined $out || $out =~ /not found|No such file/;
# A red run is a failure to answer, not a drift: the counts it prints describe
# a suite that did not pass, and sending the maintainer to correct counts that
# are correct would be the wrong diagnosis.
die "docs-drift: the toml-test run failed; fix the suite before comparing counts\n"
if $? != 0;
my %live;
for my $kind (qw(valid invalid encoder)) {
my ($passed) = $out =~ /\b$kind tests:\s+(\d+) passed/;
die "docs-drift: could not read the $kind count from the toml-test output\n"
unless defined $passed;
my ($failed) = $out =~ /\b$kind tests:\s+\d+ passed, (\d+) failed/;
die "docs-drift: the $kind run has $failed failures; fix the suite first\n"
if defined $failed && $failed != 0;
$live{$kind} = $passed;
}
print "docs-drift: the suite now stands at $live{valid} valid, $live{invalid} invalid and $live{encoder} encoder cases\n";
my $drift = 0;
for my $file ('README.md', 'docs/ARCHITECTURE.md') {
open(my $fh, '<', $file) or die "docs-drift: cannot read $file: $!\n";
my $text = do { local $/; <$fh> };
close($fh);
while ($text =~ /(\d+)\s+(valid|invalid|encoder)/g) {
my ($quoted, $kind) = ($1, $2);
if ($quoted != $live{$kind}) {
print "docs-drift: $file quotes $quoted $kind cases, the suite says $live{$kind}\n";
$drift = 1;
}
}
}
die "docs-drift: the documentation has drifted from the suite\n" if $drift;
print "docs-drift: the documentation matches the suite\n";
+1396
View File
File diff suppressed because it is too large Load Diff
+874
View File
@@ -0,0 +1,874 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"errors"
"maps"
"net"
"reflect"
"strings"
"sync/atomic"
"testing"
"time"
)
type targetNested struct {
X int `toml:"x"`
Y string `toml:"y"`
}
type targetCfg struct {
Num int `toml:"num"`
Small uint8 `toml:"small"`
Tags []string `toml:"tags"`
Lims map[string]any `toml:"lims"`
Tab targetNested `toml:"tab"`
Arr []targetNested `toml:"arr"`
Other string `toml:"other"`
}
// TestTargetedStrictFindings pins the strict findings of the targeted parse
// to the tree decode's own texts, paths included. Every case here was first
// surfaced by FuzzTargetedDecode.
func TestTargetedStrictFindings(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "unknown key in a header table",
doc: "[tab]\nother = \"o\"\n",
want: `interpres: tab: unknown field "other" for interpres.targetNested`,
},
{
name: "unknown nested header without the parent header",
doc: "[tab.nested]\nx = 1\n",
want: `interpres: tab: unknown field "nested" for interpres.targetNested`,
},
{
name: "unknown key in an array-of-tables element",
doc: "[[arr]]\nother = \"o\"\n",
want: `interpres: arr[0]: unknown field "other" for interpres.targetNested`,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg, RejectUnknownFields(true))
if err == nil {
t.Fatalf("no error, want %q", tt.want)
}
if err.Error() != tt.want {
t.Errorf("message = %q, want %q", err.Error(), tt.want)
}
})
}
}
// TestTargetedParseErrors pins the parse-stage errors the targeted skeleton
// raises, whose texts and lines are the tree parser's own.
func TestTargetedParseErrors(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "header on an assigned scalar",
doc: "zz = 1\n[zz]\nx = 4\n",
want: "interpres: line 2: key \"zz\" is not a table",
},
{
name: "dotted key on an assigned scalar",
doc: "zz = 1\nzz.x = 2\n",
want: "interpres: line 2: key \"zz\" is not a table",
},
{
name: "duplicate unknown keys",
doc: "zz = 1\nzz = 2\n",
want: "interpres: line 2: duplicate key \"zz\"",
},
{
name: "duplicate inside an unknown table",
doc: "[zz]\nk = 1\nk = 2\n",
want: "interpres: line 3: duplicate key \"k\"",
},
{
name: "duplicate across a sink's dotted keys",
doc: "[zz]\na.b = 1\na.b = 2\n",
want: "interpres: line 3: duplicate key \"b\"",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg)
if err == nil {
t.Fatalf("no error, want %q", tt.want)
}
if err.Error() != tt.want {
t.Errorf("message = %q, want %q", err.Error(), tt.want)
}
})
}
}
// TestTargetedSilentShapes covers the documents the targeted parse accepts
// with the values the tree decode gives.
func TestTargetedSilentShapes(t *testing.T) {
t.Run("dotted key after an unknown nested header", func(t *testing.T) {
// [tab.nested] is unknown and sinks; tab.x then lands in tab, and the
// sink's own x is a different key, the tree's shape exactly.
var cfg, ref targetCfg
in := []byte("[tab]\nx = 1\n[tab.nested]\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if err := treeDecodeInto(in, &ref); err != nil {
t.Fatalf("reference: %v", err)
}
if !reflect.DeepEqual(cfg, ref) {
t.Errorf("values disagree: targeted %+v, tree %+v", cfg, ref)
}
if cfg.Tab.X != 1 {
t.Errorf("tab.x = %d, want 1", cfg.Tab.X)
}
})
t.Run("unknown keys are ignored without strict", func(t *testing.T) {
var cfg, ref targetCfg
in := []byte("num = 5\nz1 = 1\n[zz]\nk = 1\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if err := treeDecodeInto(in, &ref); err != nil {
t.Fatalf("reference: %v", err)
}
if !reflect.DeepEqual(cfg, ref) {
t.Errorf("values disagree: targeted %+v, tree %+v", cfg, ref)
}
if cfg.Num != 5 {
t.Errorf("num = %d, want 5", cfg.Num)
}
})
t.Run("an inline table into a map field", func(t *testing.T) {
var cfg targetCfg
in := []byte("lims = { cpu = 4, deep = { a = true } }\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if cfg.Lims["cpu"] != int64(4) {
t.Errorf("lims = %v", cfg.Lims)
}
})
t.Run("an overflow falls back to the decode error", func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("small = 300\n"), &cfg)
want := "interpres: small: integer 300 overflows uint8"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("too many array-of-tables elements falls back", func(t *testing.T) {
type Item struct {
N int `toml:"n"`
}
var cfg struct {
Items [2]Item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n[[items]]\nn = 2\n[[items]]\nn = 3\n"), &cfg)
want := "interpres: items: cannot assign 3 elements to [2]interpres.Item"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("a UseNumber tree keeps literals in the targeted path", func(t *testing.T) {
var cfg struct {
Rate Number `toml:"rate"`
}
if err := Unmarshal([]byte("rate = 1_000\n"), &cfg, NumbersAsLiterals(true)); err != nil {
t.Fatal(err)
}
if cfg.Rate != "1_000" {
t.Errorf("rate = %q, want 1_000", cfg.Rate)
}
})
t.Run("dotted keys fill a map field", func(t *testing.T) {
var cfg targetCfg
in := []byte("lims.a.b = true\nlims.c = 3\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if cfg.Lims["c"] != int64(3) {
t.Errorf("lims = %v", cfg.Lims)
}
})
t.Run("an inline table cannot be extended", func(t *testing.T) {
var cfg targetCfg
in := []byte("lims = { a = 1 }\n[lims.deep]\nb = 2\n")
err := Unmarshal(in, &cfg)
if err == nil || !strings.Contains(err.Error(), "cannot extend inline table") {
t.Errorf("err = %v, want the inline-table extension error", err)
}
})
}
// TestTargetedShapesMatrix walks a document per destination shape, both
// through the targeted path and the tree reference, so the two agree on
// every branch the skeleton carries.
func TestTargetedShapesMatrix(t *testing.T) {
docs := []string{
// Scalars of every kind, arrays, maps, tables, arrays of tables.
"num = 7\nflt = 1.25\nstr = \"s\"\nflag = false\nsmall = 9\ntags = [\"a\"]\nlims = { a = 1 }\n\n[tab]\nx = 1\ny = \"t\"\n\n[[arr]]\nx = 2\ny = \"u\"\n\n[[arr]]\nx = 3\ny = \"v\"\n",
// Dotted keys through nested tables and maps.
"tab.x = 1\ntab.y = \"s\"\nlims.a.b = true\nlims.c = 3\nnum = 2\n",
// Inline tables nested in arrays, mixed value arrays.
"lims = { a = { b = 1 } }\ntags = []\nother = \"o\"\n",
// A sub-table of an array of tables, then a second element.
"[[arr]]\nx = 1\n[arr.nested]\ny = \"n\"\n[[arr]]\ny = \"m\"\n",
// Negative and signed numbers, exponents, radix forms into floats.
"flt = -3.5e2\nnum = -42\nflt = +1.0\n",
// A quoted key and a defined-string-shaped value.
"\"quoted key\" = 1\nstr = \"multi\"\n",
}
for i, doc := range docs {
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
tgtErr := Unmarshal([]byte(doc), &tgt)
if (refErr == nil) != (tgtErr == nil) {
t.Errorf("doc %d: error presence disagrees: tree %v, targeted %v", i, refErr, tgtErr)
continue
}
if refErr != nil {
continue
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("doc %d: values disagree:\ntree: %#v\ntargeted: %#v", i, ref, tgt)
}
}
}
// TestTargetedFallbackContracts pins the documents that must fall back and
// produce the tree decode's exact error.
func TestTargetedFallbackContracts(t *testing.T) {
type Item struct {
N int `toml:"n"`
}
tests := []struct {
name string
doc string
want string
}{
{
name: "uint8 overflow",
doc: "small = 300\n",
want: "interpres: small: integer 300 overflows uint8",
},
{
name: "negative into uint",
doc: "small = -1\n",
want: "interpres: small: cannot assign negative -1 to uint8",
},
{
name: "a table into a scalar",
doc: "num = { a = 1 }\n",
want: "interpres: num: cannot assign table to int",
},
{
name: "an integer into a string field",
doc: "other = 5\n",
want: "interpres: other: cannot assign integer to string",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg)
if err == nil || err.Error() != tt.want {
t.Errorf("err = %v, want %q", err, tt.want)
}
})
}
_ = Item{}
}
// TestTargetedHeaderOnAssignedScalar pins the parse error a header raises
// when the key already holds a scalar, before any fallback can happen.
func TestTargetedHeaderOnAssignedScalarArray(t *testing.T) {
var cfg targetCfg
in := []byte("arr = []\n[[arr]]\nx = 1\n")
err := Unmarshal(in, &cfg)
want := "interpres: line 2: key \"arr\" is not an array of tables"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
}
// TestTargetedBranchParity walks the fallback branches of the targeted
// skeleton: every document here takes the tree path on a rerun, and must
// carry the tree decode's exact error text.
func TestTargetedBranchParity(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "a header over a value array",
doc: "tags = [\"x\"]\n[tags]\na = 1\n",
want: "interpres: line 2: key \"tags\" is not a table",
},
{
name: "an array header over a value array",
doc: "tags = [\"x\"]\n[[tags]]\na = 1\n",
want: "interpres: line 2: key \"tags\" is not an array of tables",
},
{
name: "an array header over a datetime field",
doc: "when = 1979-05-27T07:32:00Z\n[[when]]\nx = 1\n",
want: "interpres: line 2: key \"when\" is not an array of tables",
},
{
name: "a boolean into a string field",
doc: "other = true\n",
want: "interpres: other: cannot assign bool to string",
},
{
name: "a leading-zero integer",
doc: "num = 01\n",
want: "interpres: line 1: leading zeros are not allowed in numbers",
},
{
name: "an int64-overflowing integer",
doc: "num = 99999999999999999999\n",
want: "interpres: line 1: integer \"99999999999999999999\" out of range",
},
{
name: "a malformed boolean",
doc: "flag = tru\n",
want: "interpres: line 1: invalid value",
},
{
name: "a negative number into an unsigned field",
doc: "small = -5\n",
want: "interpres: small: cannot assign negative -5 to uint8",
},
{
name: "an integer into a string field via the generic path",
doc: "other = 5\n",
want: "interpres: other: cannot assign integer to string",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg, RejectUnknownFields(true))
if tt.want == "" {
if err != nil {
t.Fatalf("err = %v, want nil", err)
}
return
}
if err == nil || err.Error() != tt.want {
t.Errorf("err = %v, want %q", err, tt.want)
}
})
}
}
// TestTargetedDecodeHookFields keeps the custom decode hooks of scalar-typed
// fields working in the targeted path.
func TestTargetedDecodeHookFields(t *testing.T) {
type Cfg struct {
IP net.IP `toml:"ip"`
Dur time.Duration `toml:"dur"`
Unm *scalarUnmarshaler `toml:"unm"`
}
var cfg Cfg
in := []byte("ip = \"192.0.2.1\"\ndur = \"1h30m\"\nunm = \"hello\"\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatal(err)
}
if cfg.IP.String() != "192.0.2.1" {
t.Errorf("ip = %v", cfg.IP)
}
if cfg.Dur != 90*time.Minute {
t.Errorf("dur = %v", cfg.Dur)
}
if cfg.Unm == nil || cfg.Unm.val != "hello" {
t.Errorf("unm = %+v", cfg.Unm)
}
}
// TestTargetedOddShapes pins the fallback and value shapes the matrix does
// not reach: space-separated date-times, non-string map keys and repeated
// dotted map keys.
func TestTargetedOddShapes(t *testing.T) {
t.Run("a space-separated date-time", func(t *testing.T) {
type Cfg struct {
When time.Time `toml:"when"`
}
var cfg, ref Cfg
doc := []byte("when = 1979-05-27 07:32:00Z\n")
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatal(err)
}
if !cfg.When.Equal(ref.When) {
t.Errorf("when = %v, want %v", cfg.When, ref.When)
}
})
t.Run("a map with a non-string key falls back", func(t *testing.T) {
type Cfg struct {
M map[int]string `toml:"m"`
}
var cfg, ref Cfg
doc := []byte("m = { a = 1 }\n")
err := Unmarshal(doc, &cfg)
refErr := treeDecodeInto(doc, &ref)
if err == nil || refErr == nil {
t.Fatalf("err = %v, refErr = %v, want both to fail", err, refErr)
}
if err.Error() != refErr.Error() {
t.Errorf("errors disagree: targeted %q, tree %q", err, refErr)
}
})
t.Run("a repeated dotted map key is a duplicate", func(t *testing.T) {
var cfg targetCfg
doc := []byte("lims.a.b = 1\nlims.a.b = 2\n")
err := Unmarshal(doc, &cfg)
want := "interpres: line 2: duplicate key \"b\""
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("an underscored integer takes the token path", func(t *testing.T) {
var cfg targetCfg
doc := []byte("num = 1_000\n")
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Num != 1000 {
t.Errorf("num = %d, want 1000", cfg.Num)
}
})
t.Run("an 18-digit integer takes the fast path", func(t *testing.T) {
var cfg struct {
Big int64 `toml:"big"`
}
doc := []byte("big = 999999999999999999\n")
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Big != 999999999999999999 {
t.Errorf("big = %d", cfg.Big)
}
})
}
// TestTargetedMapTableShapes covers the map-entry branches of the targeted
// skeleton: entries that become tables, entries that refuse them, and the
// duplicate checks across them.
func TestTargetedMapTableShapes(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "a header opens a map entry table",
doc: "lims.c = 1\n[lims.d]\nk = 1\n",
want: "",
},
{
name: "a header over an assigned map entry",
doc: "lims.a = 1\n[lims.a]\nk = 1\n",
want: "interpres: line 2: key \"a\" is not a table",
},
{
name: "a dotted key over an assigned map entry",
doc: "lims.a = 1\nlims.a.b = 2\n",
want: "interpres: line 2: key \"a\" is not a table",
},
{
name: "a duplicate plain map entry",
doc: "lims.a = 1\nlims.a = 2\n",
want: "interpres: line 2: duplicate key \"a\"",
},
{
name: "an array of tables inside a map entry",
doc: "lims.c = 1\n[[lims.items]]\nk = 1\n",
want: "",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg, ref targetCfg
err := Unmarshal([]byte(tt.doc), &cfg)
refErr := treeDecodeInto([]byte(tt.doc), &ref)
if (err == nil) != (refErr == nil) {
t.Fatalf("error presence disagrees: tree %v, targeted %v", refErr, err)
}
if err != nil {
if err.Error() != refErr.Error() {
t.Fatalf("errors disagree:\ntree: %v\ntargeted: %v", refErr, err)
}
return
}
if !reflect.DeepEqual(cfg, ref) {
t.Errorf("values disagree: targeted %+v, tree %+v", cfg, ref)
}
})
}
}
// TestTargetedNestedMapDescents pins the descents into a map of maps that
// meet entries the document built earlier: a dotted key twice through the
// same sub-table, a header into a dotted-built sub-table, and a typed array
// under a map key. Each shape once panicked on a reflect Elem of a map.
func TestTargetedNestedMapDescents(t *testing.T) {
t.Run("dotted key through one sub-table twice", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\nm.a.c = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["c"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("header under a dotted-built sub-table", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\n[m.a.deep]\nx = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["deep"].(map[string]any)["x"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("typed array under a map key", func(t *testing.T) {
var cfg struct {
M map[string][]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("[[m.arr]]\nx = 1\n\n[[m.arr]]\ny = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.M["arr"]) != 2 || cfg.M["arr"][1]["y"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
}
// TestTargetedPointerElementSlice pins that an array of tables over a slice
// of pointer elements fills the pointed-to structs.
func TestTargetedPointerElementSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items []*item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n\n[[items]]\nn = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 2 || cfg.Items[0] == nil || cfg.Items[1].N != 2 {
t.Errorf("items = %#v", cfg.Items)
}
}
// TestTargetedArrayScopeResets pins that a new element of an array of tables
// starts a fresh definition scope, the contract the changelog documents.
func TestTargetedArrayScopeResets(t *testing.T) {
doc := "[[a]]\nb.c = 1\n\n[[a]]\n\n[a.b]\nx = 1\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
if refErr != nil {
t.Fatalf("tree decode: %v", refErr)
}
if err := Unmarshal([]byte(doc), &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
}
// TestTargetedUnknownArrayElements pins that every element of an unknown
// array of tables is a fresh namespace, and a sub-table header reaches the
// last element the way the tree parser's does.
func TestTargetedUnknownArrayElements(t *testing.T) {
doc := "[[zz]]\nk = 1\n\n[[zz]]\nk = 2\n\n[zz.sub]\nx = 3\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
tgtErr := Unmarshal([]byte(doc), &tgt)
if (refErr == nil) != (tgtErr == nil) {
t.Fatalf("error presence disagrees: tree %v, targeted %v", refErr, tgtErr)
}
if refErr != nil {
return
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
// A dotted key may not enter the array: the tree's own rule.
var dotted targetCfg
dErr := Unmarshal([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
refDotted := treeDecodeInto([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
if (dErr == nil) != (refDotted == nil) {
t.Errorf("dotted into an array: targeted %v, tree %v", dErr, refDotted)
}
}
// TestTargetedFixedArrayUnderFill pins that a fixed-size array the document
// under-fills is the length mismatch the tree decode raises, with the
// field's path.
func TestTargetedFixedArrayUnderFill(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items [2]item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n"), &cfg)
if err == nil {
t.Fatal("unmarshal accepted an under-filled array")
}
want := `interpres: items: cannot assign 1 elements to [2]interpres.item`
if err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedPrefilledSliceReplaced pins that a prefilled slice is replaced
// by the document's elements on both paths, not appended to.
func TestTargetedPrefilledSliceReplaced(t *testing.T) {
type item struct {
N int `toml:"n"`
}
doc := []byte("[[items]]\nn = 1\n")
var ref struct {
Items []item `toml:"items"`
}
ref.Items = []item{{N: 9}}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
var tgt struct {
Items []item `toml:"items"`
}
tgt.Items = []item{{N: 9}}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if len(tgt.Items) != 1 || tgt.Items[0].N != 1 {
t.Errorf("items = %#v, want the prefilled element replaced", tgt.Items)
}
}
// TestTargetedHeaderOverValueArrayKeepsCase pins that a value array assigned
// under a differently cased key than the field's name still blocks the
// array-of-tables header over it, the tree parse error.
func TestTargetedHeaderOverValueArrayKeepsCase(t *testing.T) {
var cfg struct {
Arr []targetNested `toml:"arr"`
}
err := Unmarshal([]byte("Arr = [{x = 1}]\n[[Arr]]\nx = 2\n"), &cfg)
if err == nil || err.Error() != `interpres: line 2: key "Arr" is not an array of tables` {
t.Errorf("err = %v, want the parse error over the assigned field", err)
}
}
// TestTargetedDottedInlineFreezePath pins that an inline table assigned by a
// dotted key freezes the whole path the statement wrote: a later header
// under that path is the extension error, and a key outside it stays free.
func TestTargetedDottedInlineFreezePath(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("m.a.b = {x = 1}\nb.y = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
err = Unmarshal([]byte("m.a.b = {x = 1}\n[m.a.b]\ny = 2\n"), &cfg)
want := `interpres: line 2: cannot extend inline table "m.a.b"`
if err == nil || err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedStrictThroughDottedKeys pins that strict and required findings
// survive the transient tables a dotted descent builds.
func TestTargetedStrictThroughDottedKeys(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("tab.zz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the strict failure through the dotted key", err)
}
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the path through the dotted key", err)
}
}
// TestTargetedRequiredThroughDottedKeys pins that a required tag is honoured
// when the table is reached only through dotted keys.
func TestTargetedRequiredThroughDottedKeys(t *testing.T) {
type nested struct {
X int `toml:"x,required"`
Y int `toml:"y"`
}
var cfg struct {
Tab nested `toml:"tab"`
}
err := Unmarshal([]byte("tab.y = 1\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), `missing required key "x"`) {
t.Errorf("err = %v, want the missing required key through the dotted key", err)
}
}
// TestTargetedOrderedMapSliceFallsBack pins that a slice of OrderedMap
// elements takes the tree path, whose fill keeps the written order.
func TestTargetedOrderedMapSliceFallsBack(t *testing.T) {
var cfg struct {
Items []OrderedMap `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nk = \"v\"\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 1 || cfg.Items[0].Keys()[0] != "k" {
t.Errorf("items = %#v, want the element filled in written order", cfg.Items)
}
}
// hookMap is a named map type whose decode hook counts its calls.
type hookMap map[string]any
var hookMapCalls atomic.Int32
func (h *hookMap) UnmarshalTOML(data any) error {
hookMapCalls.Add(1)
m, _ := data.(map[string]any)
if *h == nil {
*h = hookMap{}
}
maps.Copy((*h), m)
return nil
}
// TestTargetedMapFieldHookGetsWholeTable pins that a named map field with a
// decode hook receives the whole parsed table, even in its header form.
func TestTargetedMapFieldHookGetsWholeTable(t *testing.T) {
type cfg struct {
M hookMap `toml:"m"`
}
var c cfg
hookMapCalls.Store(0)
err := Unmarshal([]byte("[m]\na = 1\nb = 2\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if hookMapCalls.Load() != 1 {
t.Errorf("hook calls = %d, want exactly one with the whole table", hookMapCalls.Load())
}
if c.M["a"] != int64(1) || c.M["b"] != int64(2) {
t.Errorf("m = %#v", c.M)
}
}
// errHook fails every decode with a fixed error and counts its calls.
type errHook struct{ calls *int }
func (e *errHook) UnmarshalTOML(any) error {
if e.calls != nil {
*e.calls++
}
return errors.New("boom")
}
// TestTargetedHookErrorRunsOnce pins that a failing hook's error is the
// tree path's own, wrapped with the key, and that the hook is not run a
// second time by a fallback.
func TestTargetedHookErrorRunsOnce(t *testing.T) {
calls := 0
cfg := struct {
F errHook `toml:"f"`
}{F: errHook{calls: &calls}}
err := Unmarshal([]byte("f = 1\n"), &cfg)
if err == nil || err.Error() != "interpres: f: unmarshal: boom" {
t.Errorf("err = %v, want the wrapped hook failure", err)
}
if calls != 1 {
t.Errorf("hook calls = %d, want one", calls)
}
}
// TestTargetedUnknownBeforeRequired pins the report order the tree decode
// produces: an unknown key wins over a missing required one.
func TestTargetedUnknownBeforeRequired(t *testing.T) {
type inner struct {
X int `toml:"x,required"`
}
var cfg struct {
Tab inner `toml:"tab"`
}
err := Unmarshal([]byte("[tab]\nzz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the unknown key reported before the required one", err)
}
}
// TestTargetedStrictPathStableAcrossHeaders pins that the path a strict
// finding wraps does not alias the parser's key buffer: the table that owns
// the unknown key keeps its name after a later header.
func TestTargetedStrictPathStableAcrossHeaders(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("[tab]\nzz = 1\n\n[lims]\nx = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the finding on tab, not the later header", err)
}
}
// TestTargetedPrefilledMapFieldMergesUnderHeader pins that a prefilled map
// field merges the document's header-form table into it on both paths, the
// rule the root map has always followed.
func TestTargetedPrefilledMapFieldMergesUnderHeader(t *testing.T) {
doc := []byte("[lims]\nnew = 3\n")
var ref, tgt targetCfg
ref.Lims = map[string]any{"keep": "yes"}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
tgt.Lims = map[string]any{"keep": "yes"}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if tgt.Lims["keep"] != "yes" || tgt.Lims["new"] != int64(3) {
t.Errorf("lims = %#v, want the merge", tgt.Lims)
}
}
// TestTargetedNumberTokenValidatesUTF8 pins that the token route the
// targeted parse takes reports invalid UTF-8 with the scanner's own message
// and position.
func TestTargetedNumberTokenValidatesUTF8(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("num = 12\xff\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), "invalid UTF-8 in value at byte offset 8") {
t.Errorf("err = %v, want the UTF-8 complaint on the invalid byte", err)
}
}
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("0=00:00\n1=0000-01-01 00:00:00.0+00:00#000000000000")
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("e = \"\\\\e[0m\\\\x41\\\\x7f\\\\x00\"\n")
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("m = {\n\ttitle = \"one\",\n\tnums = [1, 2,],\n\tinner = { deep = true }, # trailing\n}\n")
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("t = 13:37\nbig = 1979-05-27 07:32:00.5+01:00\nshort = 1979-05-27 07:32\n")