68 Commits
Author SHA1 Message Date
petrbalvin e3dda5e115 fix: take the release-check version from the recipe argument
Test / test (push) Successful in 1m46s
Release / gates (push) Successful in 1m36s
Release / release (push) Successful in 39s
2026-09-22 21:49:05 +02:00
petrbalvin e35b4bb90d fix: take the release-check version from the recipe argument
Test / test (push) Canceled after 44s
2026-09-22 21:48:00 +02:00
petrbalvin 75ade89f34 chore: prepare release v2.0.0
Test / test (push) Successful in 2m3s
2026-09-22 21:45:25 +02:00
petrbalvin 41786c5bc6 chore: drop the internal reference from the justfile header 2026-09-22 21:45:25 +02:00
petrbalvin 010a7b2a1e ci: pin the actions by version tag and cap the test timeout at 10m 2026-09-22 21:45:25 +02:00
petrbalvin eb6ac1ab9d ci: drop the schedule triggers, race and fuzz run on dispatch alone
Test / test (push) Successful in 1m54s
Assisted-by: GLM 5.3
2026-09-22 21:28:30 +02:00
petrbalvin b0f40739c7 chore: keep the coverage floor at the documented 80 percent
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin fba53440c7 ci: state where the nightly race and fuzz sweeps run
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 332cd01d44 chore: check the suite colour before comparing documented counts
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 2fa075de00 docs: align every document with the reviewed behaviour
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin 4900367970 fix(cmd): long-form flags, honest counts and safer inference
Assisted-by: GLM 5.3
2026-09-22 21:15:07 +02:00
petrbalvin b7f39435e1 fix(document): rebuild set nodes and write documents back round-trip
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 0ded34da3c fix(encode): pointer table arrays, whole-minute offsets and emission checks
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin b45f4d65da fix(decode): keep the targeted parse on the tree path's contract
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin f7e427ae3f fix(decode): allocate embedded pointer maps and settle case collisions
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin e677e34508 fix(api): one statement per value array, parseas options and bounded reads
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 2efdb2d059 fix(parse): reject the lenient grammar edges and name out-of-range date-times
Assisted-by: GLM 5.3
2026-09-22 21:15:00 +02:00
petrbalvin 8f2b26bd33 docs: align the remaining Decoder and Encoder mentions with the options API
Test / test (push) Successful in 1m41s
Assisted-by: GLM 5.3 Flash
2026-09-22 19:27:12 +02:00
petrbalvin 2e61ad0ba9 feat!: rework the public API to json/v2-style variadic options
Test / test (push) Successful in 1m49s
Assisted-by: GLM 5.3 Flash
2026-09-22 18:42:18 +02:00
petrbalvin 7ee155d1e9 test(decode): seed the targeted fuzz with regression documents
Test / test (push) Successful in 1m39s
Assisted-by: GLM 5.3 Flash
2026-09-22 16:09:00 +02:00
petrbalvin a7d0041259 perf(decode): parse struct destinations without the value tree
Test / test (push) Successful in 1m50s
Assisted-by: GLM 5.3 Flash
2026-09-22 15:59:03 +02:00
petrbalvin ee32490452 docs: write the 2.0.0 migration guide into the changelog
Test / test (push) Successful in 1m36s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:48:23 +02:00
petrbalvin 80b2bc6e0f perf(encode): write scalars and scalar arrays without boxing
Test / test (push) Successful in 1m34s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:46:26 +02:00
petrbalvin ebaca18093 docs: note how the module resolves and verify it in release-check
Test / test (push) Successful in 1m40s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:31:37 +02:00
petrbalvin 3406955654 docs: add godoc examples, extend the basic example and map encoding/json
Test / test (push) Successful in 1m47s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:29:42 +02:00
petrbalvin f6a96379e6 ci: schedule nightly race and fuzz, pin actions, add release-check and docs drift
Test / test (push) Successful in 1m37s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:26:48 +02:00
petrbalvin 3ffae35a20 feat: add the Document edit pipeline with comment-preserving write
Test / test (push) Successful in 1m35s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:22:46 +02:00
petrbalvin a7a942a8e1 feat: add Statements, the top-level statement iterator
Test / test (push) Successful in 1m36s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:12:04 +02:00
petrbalvin d92bb56853 refactor: rename the encoder layout options
Test / test (push) Successful in 1m35s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:09:00 +02:00
petrbalvin bef1d3fbd9 test: add FuzzMarshal, golden messages, synctest cancellation and cross smoke
Test / test (push) Successful in 1m31s
Assisted-by: GLM 5.3 Flash
2026-09-22 01:03:27 +02:00
petrbalvin 3e741e7790 feat(cmd): add version, plain json, struct inference and schema modes
Test / test (push) Successful in 1m35s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:55:02 +02:00
petrbalvin d18935ebc2 feat: add field comments, local time zone decoding and in-value cancellation
Test / test (push) Successful in 1m32s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:44:23 +02:00
petrbalvin 71bd82a7a5 feat: add ParseAs and NewSchema generics
Test / test (push) Successful in 1m38s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:36:10 +02:00
petrbalvin b02471c09a refactor: unify the error paths in one Path type
Test / test (push) Canceled after 1m31s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:34:48 +02:00
petrbalvin a8a2fcf8c3 feat: add the inline tag and json-style omitempty
Test / test (push) Successful in 1m30s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:28:52 +02:00
petrbalvin 13ac6dd521 test: pin the embedded map and map merge rules
Test / test (push) Successful in 1m34s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:23:41 +02:00
petrbalvin 10391a090f feat: bound the encoder walk and add UnmarshalWithOptions
Test / test (push) Successful in 1m34s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:21:46 +02:00
petrbalvin eaa69dc6f6 feat: add OrderedMap, the table that keeps its key order
Test / test (push) Successful in 1m30s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:15:17 +02:00
petrbalvin aefff80a28 style: gofmt the decode tests
Test / test (push) Successful in 1m29s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:04:49 +02:00
petrbalvin 0ba145ba0c feat: add the required tag option and UnmarshalerContext
Test / test (push) Canceled after 39s
Assisted-by: GLM 5.3 Flash
2026-09-22 00:04:16 +02:00
petrbalvin 2e5dfc54c9 feat: decode arrays into [N]T and add MarshalAppend
Test / test (push) Successful in 2m31s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:58:23 +02:00
petrbalvin 10d49fbe60 feat: carry the byte offset and column in SyntaxError
Test / test (push) Canceled after 2m28s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:55:58 +02:00
petrbalvin 3cc168f39a feat: add ParseFile and Valid
Test / test (push) Successful in 2m33s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:51:58 +02:00
petrbalvin ce0c1ebd9d feat: add Decoder.UseNumber and the Number type
Test / test (push) Successful in 1m52s
Assisted-by: GLM 5.3 Flash
2026-09-21 23:49:39 +02:00
petrbalvin 582222738b perf(encode): emit in place, pool the buffer and cache interface flags
Test / test (push) Successful in 1m29s
2026-09-20 22:15:38 +02:00
petrbalvin 3a4bd74bf0 perf(decode): resolve interfaces through cached type flags 2026-09-20 22:15:23 +02:00
petrbalvin b4d564c682 perf(parse): cut allocations and validate UTF-8 in the scan 2026-09-20 22:15:10 +02:00
petrbalvin adf189aa2c perf(datetime): scan the token shape and render in one pass 2026-09-20 22:15:10 +02:00
petrbalvin 7596754180 test(bench): measure the long document on the typed decode and marshal paths 2026-09-20 22:15:10 +02:00
petrbalvin 72f8b21ac4 feat: parse into a Document that keeps order and comments
Test / test (push) Successful in 1m34s
Assisted-by: DeepSeek V4.1 Flash
2026-09-20 10:40:57 +02:00
petrbalvin a6e3e3fe31 feat: add OffsetDateTime, nesting limits and uniform Marshaler dispatch
Test / test (push) Successful in 2m18s
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 19:36:24 +02:00
petrbalvin b695b69768 docs(api): use an inline table sample that really breaks
Test / test (push) Successful in 1m33s
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 12:18:57 +02:00
petrbalvin 959eaba4b0 feat(encode): add the InlineTables option
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 12:18:30 +02:00
petrbalvin 8f85bb68fa feat(encode): write the TOML 1.1 output form
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 12:18:18 +02:00
petrbalvin bccaf087c8 feat(cmd): add the encoder mode to the toml-test adapter
Test / test (push) Successful in 1m33s
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 11:38:19 +02:00
petrbalvin 0149a5b4d1 docs(encoder): correct the multi-line string claim
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 11:38:17 +02:00
petrbalvin 815141440e feat: honour TextMarshaler and TextUnmarshaler by default
Test / test (push) Successful in 1m35s
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 02:41:09 +02:00
petrbalvin 9023784da3 fix(decode): decode into a defined string or bool type
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 02:40:51 +02:00
petrbalvin 942c4b1489 docs(security): list the newest release as supported
Test / test (push) Successful in 1m33s
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 02:24:24 +02:00
petrbalvin 8f0eae6604 docs: drop the TOML 1.0 compatibility promise
Assisted-by: DeepSeek V4.1 Flash
2026-09-19 02:24:24 +02:00
petrbalvin 1c7329aeea build: move the module path to /v2
Test / test (push) Successful in 1m32s
Assisted-by: GLM 5.3 Flash
2026-09-19 00:14:39 +02:00
petrbalvin ad6c32d0c6 chore: prepare release v1.1.0
Test / test (push) Successful in 1m32s
Release / gates (push) Successful in 1m21s
Release / release (push) Successful in 52s
Assisted-by: GLM 5.3 Flash
2026-09-18 01:10:08 +02:00
petrbalvin 17574a0d15 docs: add the benchmarking document
Test / test (push) Canceled after 1m19s
Assisted-by: GLM 5.3 Flash
2026-09-18 00:20:42 +02:00
petrbalvin 8aa2b1b9c0 docs(api): name the newline a literal string may carry
Assisted-by: GLM 5.3 Flash
2026-09-18 00:20:42 +02:00
petrbalvin 6a043e2824 docs(architecture): scope the no-caching claim to the parser
Assisted-by: GLM 5.3 Flash
2026-09-18 00:20:31 +02:00
petrbalvin 81033bb27c docs(api): document the TOML 1.1 acceptance
Assisted-by: GLM 5.3 Flash
2026-09-18 00:20:31 +02:00
petrbalvin dfd5d240d2 docs(changelog): drop the stale toml-test run mode
Assisted-by: GLM 5.3 Flash
2026-09-18 00:20:31 +02:00
petrbalvin 78946578d1 docs(development): correct the release pipeline gates
Assisted-by: GLM 5.3 Flash
2026-09-18 00:20:31 +02:00
51 changed files with 13587 additions and 996 deletions
+36
View File
@@ -0,0 +1,36 @@
# Fuzz smoke, Go. Dispatched by hand when a change asks for it.
#
# Fuzzing is exploration, so it never belongs to the push pipeline; a 30 second
# smoke per target on a hand dispatch checks a change without holding the
# shared box. The targets run the seeds and whatever the corpus has gathered; a
# failure leaves its crashing input in testdata/fuzz, which the ordinary suite
# then reproduces on every push.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Fuzz
on:
workflow_dispatch:
env:
# One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1
GOMAXPROCS: "2"
jobs:
fuzz:
runs-on: fedora
timeout-minutes: 10
steps:
- uses: actions/checkout@v7
- uses: actions/setup-go@v6
with:
go-version-file: go.mod
cache: true
- name: Fuzz the parser
run: go test -run '^$' -fuzz FuzzParse -fuzztime=30s -timeout 10m .
- name: Fuzz the encoder
run: go test -run '^$' -fuzz FuzzMarshal -fuzztime=30s -timeout 10m .
+6 -4
View File
@@ -1,8 +1,10 @@
# Race, Go. Dispatched by hand, and run as part of the release gates.
# Race, Go. Dispatched by hand.
#
# The race detector roughly doubles both time and memory, which the shared runner box
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once per
# task; here it is an explicit decision rather than a routine.
# cannot afford on every push. Locally it belongs to `just gates`, which runs it once
# per task; here it is an explicit decision rather than a routine, a hand dispatch
# when a change asks for one. Development carries its race gate on every push through
# that local gate.
#
# Every step is one command, so the step that fails is the gate that failed.
name: Race
@@ -32,4 +34,4 @@ jobs:
run: dnf install -y gcc
- name: Race
run: go test -race -count=1 -timeout 30m ./...
run: go test -race -count=1 -timeout 10m ./...
+3 -3
View File
@@ -4,8 +4,8 @@
# carries the CHANGELOG section as its body and nothing else. The gates still run first,
# in their own job and once, minus the race detector: race never runs on a push path or a
# tag, and the local gate raced this tree before the tag was cut. The write permission
# sits on the release job alone, and the version contract these steps implement is in the
# `release` skill.
# sits on the release job alone, and the version the binary reports is the one the
# toolchain records from the tag, with nothing injected.
#
# Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
@@ -77,7 +77,7 @@ jobs:
- name: Tests
# Keep the pattern equal to `packages` in the project's justfile.
run: go test -count=1 -timeout 30m -coverprofile=coverage.out ./...
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./...
- name: Coverage floor
run: |
+9 -6
View File
@@ -1,8 +1,9 @@
# Test, Go. Push and pull request to development. Never on main.
#
# The gates are the ones the justfile's `gates` recipe runs, minus race: the shared
# runner box cannot afford the race detector on every push, so race runs once inside
# the release pipeline instead. The box is one core and 2 GB beside Gitea, so
# runner box cannot afford the race detector on every push. Race has its own
# pipeline, dispatched by hand, and the local `just gates` runs it once per
# task. The box is one core and 2 GB beside Gitea, so
# parallelism is bounded on purpose and everything runs in one job. Extra jobs would
# duplicate the checkout, the Go setup and the dependency download three times without
# buying any parallelism.
@@ -79,7 +80,7 @@ jobs:
- name: Tests
# Scope the pattern to the packages that hold the logic when a thin cmd/ drags the
# total under the floor, and keep it equal to `packages` in the project's justfile.
run: go test -count=1 -timeout 30m -coverprofile=coverage.out ./...
run: go test -count=1 -timeout 10m -coverprofile=coverage.out ./...
- name: Coverage floor
run: |
@@ -105,6 +106,8 @@ jobs:
run: go build -o bin/interpres-decode ./cmd/interpres-decode
- name: Compliance suite
# interpres implements TOML 1.0 and 1.1; the mode is pinned so an upstream
# default change cannot silently move the corpus.
run: bin/toml-test test -decoder=bin/interpres-decode -toml=1.1
# interpres implements TOML 1.1, and the suite runs both directions: the decoder
# on the valid and invalid corpora, the encoder on the tagged JSON of the valid
# one. The mode is pinned so an upstream default change cannot silently move the
# corpus.
run: bin/toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1
+1 -1
View File
@@ -1,7 +1,7 @@
.idea/
.zcode/
# Build artifacts
# Build artefacts
bin/
*.test
*.out
+323 -4
View File
@@ -9,13 +9,332 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Added
-
## [2.0.0] - 2026-09-22
### Added
- `encoding.TextMarshaler` and `encoding.TextUnmarshaler` are honoured by
default, with no option to switch them off. A type that implements them is
encoded as a TOML string and decoded from one: `net.IP` becomes
`"192.0.2.1"`, and a user type with `MarshalText` or `UnmarshalText` follows.
`MarshalTOML` and `UnmarshalTOML` still win over the text methods, and the
four date-time types keep their bare timestamp form instead of becoming a
quoted string. A struct type that implements the interface now encodes as a
string where it was a table before, which is the breaking part of the change.
- `time.Duration` is encoded in its canonical Go form as a TOML string,
`1h30m0s`, because TOML has no duration type; the decoder reads that string
back and still accepts a bare integer as the nanosecond count.
- `interpres-decode --encode`, the adapter's other direction: it reads the
toml-test tagged JSON from stdin and writes the TOML document it describes.
The compliance suite now runs the encoder as well as the decoder, 214
encoder cases against the tagged JSON of the valid corpus.
- `InlineTables(threshold)`: a sub-table whose single-line rendering is
at most `threshold` bytes is written as an inline table instead of a header
section, which shortens a document of small tables. An array of tables keeps
its header form, because its inline form would re-parse as a value array.
- `Document` and `ParseMap`: `Parse` now returns a `*Document`, which holds the
values together with the key order, whether a table was written as an inline
table or under a header, and the comments, with `Keys`, `Entries`, `Get`,
`Comments` and `SetComments` to read and write them. `ParseMap` returns the
plain `map[string]any` tree, the shape `Parse` used to give.
- The edit pipeline on a `Document`: typed getters on `Document` and `Table`
(`GetString`, `GetInt`, `GetFloat`, `GetBool`, `GetArray`, `GetTable`),
`Set` and `Delete` that keep the surviving keys' positions and comments,
`UnmarshalDocument`, which decodes the document into a typed destination
without parsing again, and `Marshal` of a `Document`, which writes the keys
in written order, the comments above the lines and headers they belonged
to, and the inline tables inline.
- `OffsetDateTime`, the Go type of the offset date-time kind, so that all four
TOML date-time kinds have one of their own. `Parse` and `Unmarshal` hand it
back where they produced a bare `time.Time` before, and `Marshal` accepts it.
Unmarshalling into a struct field of type `time.Time` keeps working, because
the plain type takes an offset date-time as it always did; code that asserts
the tree's type, and `UnmarshalTOML` implementations that expect a
`time.Time`, need the new type.
- `MaxNestingDepth(depth)` and `MaxInputSize(size)` options bound the parse an
`Unmarshal` performs, and every parse carries a nesting limit in any case
(10000 levels, which no hand-written document approaches): a document that
nests arrays or inline tables deeper used to run the stack out and is now
rejected with a `SyntaxError` naming the limit.
- `Statements(r)`, an iterator over the top-level statements of the document
the reader carries, in written order: key/value statements, a `[table]`
header as one statement with its node, an `[[array of tables]]` as one
statement per element. A caller that breaks after the statement it wanted
reads no further ones. `examples/statements` shows the walk.
- `ParseAs[T](data, opts...)`, the generic one-line decode, and `NewSchema[T]()`,
which precompiles the struct schema and the interface flags for a hot path
before the first document arrives.
- `EmitFieldComments(true)` prints the comment a field's `toml` tag
carries in a `comment=` option above the field's line or header, the
comments a round trip through the Go type would otherwise drop. Go doc
comments are not visible to reflection, so the tag is the channel that
carries the text.
- `LocalTimeLocation(loc)` lets a local date-time fill a plain
`time.Time` destination in the location given, relabelled rather than
shifted: `07:32` in the document is `07:32` in the zone. Without the
option the wrapper types remain the only destinations a local kind fills.
- The parse checks its context inside a value as well as between statements:
an array, an inline table and a multi-line string check every 64 elements
or lines, so one huge value cannot hold the parse past its cancellation.
- `OrderedMap`, the string-keyed table that remembers the order its keys were
set in: decoding into one fills it in the order the document wrote the
keys, and `Marshal` writes one back in that order, where a map carries no
order on decode and sorts on encode. It works as a decode target on its
own, in a struct field, and as the element of an array of tables; its
values are untyped, so a nested table stays a `map[string]any`.
- `Unmarshal(data, v, opts...)` and the other entries take variadic options,
the shape encoding/json/v2 uses: `RejectUnknownFields`,
`NumbersAsLiterals`, `MaxNestingDepth`, `MaxInputSize`,
`LocalTimeLocation`. `MarshalWrite(w, v, opts...)` and
`UnmarshalRead(r, v, opts...)` are the streaming forms.
- `Marshal` carries a nesting limit of 10000 levels, the parser's own figure:
cyclic data, which used to run the stack out, is now rejected with an error
that names the limit and the path it was met at.
- The `toml` tag gained the `required` option: a field tagged
`toml:"host,required"` makes the decode fail with
`missing required key "host"` when the document carries no key that
resolves to it. The option shapes decoding only, and the encoder ignores
it.
- `UnmarshalerContext`, the custom-decode interface that hands the decode's
context to the method, `UnmarshalTOMLContext(ctx, data)`. It wins over
`UnmarshalTOML` when a type implements both, so a long custom decode can
abort on cancellation; a non-cancellable entry point hands in
`context.Background`, never nil.
- A TOML array decodes into a Go fixed-size array, `[N]T`, where only a slice
was accepted before; the encoder could already encode one. A length mismatch
is an error wrapped with the key path.
- `MarshalAppend(buf, v)` appends the TOML encoding of v to buf and returns
the extended buffer, the shape `json.MarshalAppend` has.
- `interpres-decode --version` prints the binary's version, the module version
the toolchain recorded, so a release-built binary names its own tag.
- `interpres-decode --json` prints plain indented JSON instead of the tagged
form, the shape for people and diffs, with the date-time wrappers in their
TOML form.
- `interpres-decode --validate` walks a named directory for `.toml` files and
closes the sweep with a summary naming how many documents were checked and
how many were invalid; single files stay quiet on success as before.
- `interpres-decode --struct` infers a Go struct definition from a document:
one field per key in written order, nested tables as nested struct types,
an array of tables as a slice. The printed type compiles and decodes the
document it came from.
- `interpres-decode --schema TYPE file.go` writes a TOML template for the
named struct type of a Go source, the `comment=` tag option printed as a
comment and the `default=` option as the value. It is the inverse of
`--struct`.
- `ParseFile(path)` reads the file and parses it into a `Document`, with the
file name at the front of every error it returns, read failure and parse
failure alike. `Valid(data)` reports whether a document parses, nil on
success and the parse error on failure, the library call the `--validate`
mode of interpres-decode is built on.
- `SyntaxError` carries the byte `Offset` the scan stopped at and the 1-based
`Column` on the line, beside the line it always had, and `SourceLine(src)`
renders that line with a caret under the position, for messages shown under
the input. An input that is not valid UTF-8 names the offset of the first
invalid byte in its message. The new fields are additive: a `SyntaxError`
built from a line and a message alone is unchanged.
- `NumbersAsLiterals(true)` decodes the integers and floats of the document into
`Number`, which carries the literal the document wrote, so `0x1f`, `1_000`,
`+1.0` and `inf` survive a round trip with their spelling instead of the
normalised `31`, `1000` and `1.0`. Typed destinations take the evaluated
value as before, a `Number` field takes the literal, and `Marshal` writes a
`Number` back as its bare literal, rejecting one that is not a valid TOML
number.
### Changed
- The stateful `Decoder` and `Encoder` of 1.x are replaced by variadic
options on the entries, the shape encoding/json/v2 uses: `Layout(kind)`
with `LayoutKindGrouped` or `LayoutKindDeclaration`, `OmitEmptyArrays`,
`LiteralMultiline(threshold)`, `InlineTables(threshold)`,
`EmitFieldComments`, `RejectUnknownFields`, `NumbersAsLiterals`,
`MaxNestingDepth`, `MaxInputSize`, `LocalTimeLocation`.
- `DecodeError` and `EncodeError` carry one `Path` type, a list of segments
(`"items"`, `"[0]"`, `"weight"`) with a `String()` rendering the TOML
notation, `items[0].weight`. The decode error used to hold a bare
`[]string`, the encode error a plain string. Both messages render the same
way now, `interpres: items[0].weight: ...`, with one `interpres:` prefix
where the composition used to double it.
- `omitempty` follows the encoding/json semantics: the field is skipped when
it holds an empty string, a zero number, `false`, a nil pointer or
interface, or a nil or empty slice, array or map. In 1.x the option covered
only the collections.
- The `toml` tag gained the `inline` option: a struct or map field tagged
`toml:"retry,inline"` emits as `retry = {…}` instead of a header section,
whatever its size, a named embedded struct included. Forcing it on an array
of tables is an error, because the inline form would re-parse as a value
array and change the value's Go type.
- The output takes the TOML 1.1 form. A date-time writes its seconds only when
the value carries them and drops the trailing zeros of a fractional second,
so `07:32:00` is written `07:32` and half a second as `00.5`. Both are the
same value, and a document written without seconds now comes back without
them. `LocalDateTime.String()`, `LocalTime.String()` and the offset date-time
rendering follow the same rule.
- An inline table that would pass the hundredth column is written across lines
with a trailing comma and one tab of indentation per nesting level, the shape
TOML 1.1 allows an inline table to take.
- `MarshalTOML` reaches every array element and every field, whatever the Go
kind, and its result is normalised like any other value: an element rendering
itself as a table keeps the `[[header]]` form, one rendering itself as a
scalar turns the array into a value array, and the method runs once per
element. It is found on the addressable pointer as well, so a
pointer-receiver `MarshalTOML` is called for a field or an element, exactly
as `MarshalText` is.
- TOML 1.1 is the acceptance contract, and TOML 1.0 is not. The compliance
suite runs the 1.1 corpus alone, and the promise that every 1.0 document
parses exactly as before is withdrawn. Nothing that parses today stops
parsing: the 1.0 valid corpus still passes in full. The documents whose
verdict changes are the ones 1.1 relaxed, such as the `\xHH` escape
sequences 1.0 rejected.
- The module path carries the /v2 suffix the Go toolchain requires of
every major version 2 module: imports change to
`sourcedock.dev/petrbalvin/interpres/v2`.
- Input that is not valid UTF-8 is now rejected where the parser's scan
meets the invalid byte, with a `SyntaxError` naming that line, instead of
a whole-input check that always reported line 1. Invalid input is still
rejected; the reported location is now the byte's own.
**Performance**
- Struct destinations decode directly: for a type the direct skeleton can
model, the parser resolves tables and keys against the struct schema while
the document scans and no intermediate value tree is kept. The strict
decode of the representative document drops from 168 to 160 allocations
per call against the tree path in the same process, and the 2000-element
document reaches allocation parity; every document the skeleton cannot
model falls back to the tree path and its exact error contracts. A
differential fuzz target decodes every generated document both ways.
- Marshal writes plain scalars and typed scalar arrays straight from their
reflect cells instead of boxing them into interface values first, and skips
the per-element resolution for arrays that can never take the `[[header]]`
form. The representative document now costs 130 allocations per call
instead of 141, the long array-of-tables document 55 915 instead of
63 660, with byte-identical output.
- Parsing is faster than in 1.1.0 while carrying the new document layer:
the suite's representative document decodes at about 79 MB/s with 104
allocations per call, and the long array-of-tables document at about
106 MB/s against 56 MB/s in 1.1.0, with allocations on that document
halved from 67 664 to 31 765. Date-time tokens are validated by a byte
scan instead of regular expressions, repeated keys share one string
across array-of-tables elements, and per-statement buffers are reused.
- Typed decoding is 12 percent faster than in 1.1.0 on the representative
document (9792 ns against 11 147 ns) with 24 percent fewer allocations
(167 against 220); interface lookups resolve through a cached per-type
flag set instead of boxing every value into an interface to ask.
- `Marshal` runs at the 1.1.0 speed while emitting the new TOML 1.1 output
form, at half the bytes per operation (6170 against 11 348 on the
representative document), and writes through a pooled output buffer with
a 1 MiB retention cap; repeated marshals keep the live heap flat.
- Two benchmarks measure the shapes that drove the work:
`BenchmarkStrictDecodeLong` and `BenchmarkMarshalLong` run the 2000-entry
document at about 3.8 ms and 3.4 ms per call, at 63 772 and 63 660
allocations.
### Fixed
- An offset date-time written with the `+00:00` offset kept the anonymous
location `time.Parse` invents for it, so a round trip through the tree and
`Marshal`, which writes a zero offset as `Z`, changed the value's
reflection-visible location. The zero offset normalises to `time.UTC` at
parse, and the tree is stable across the round trip.
- Decoding into a defined type whose underlying kind is string or bool, such
as `type Name string`, panicked instead of storing the value, because a
value of the predeclared type is not assignable to a defined type and the
decoder assigned it without a conversion.
- A top-level value the encoder could not normalise reported its path with
a leading dot, `interpres: .port: ...`; the message now reads
`interpres: port: ...`, the shape `EncodeError.Path` already used.
- A token shaped like a date-time with a component out of range, such as an
hour of 24 or a February the 30th, fell through to the number decoder and
failed with the number complaint `invalid character "-" in number`; it now
fails as the date-time it visibly is, `invalid date-time "..."`.
- A `time.Time` or `OffsetDateTime` whose zone offset is not a whole number
of minutes wrote only the minutes, silently shifting the instant by the
seconds dropped; the encoder now refuses such an offset, which TOML has no
form for, instead of corrupting the value.
- An empty array of tables over pointer elements, `[]*T{}`, emitted as
`key = []` while its value form was omitted; it is omitted too now, the
rule TOML forces, because an empty `[[a]]` has no valid form.
- Two lenient grammar edges are closed: a sign in a `\u` or `\U` escape,
which is not a hex digit, is rejected instead of evaluating, and a bare
carriage return right after a multi-line string's opening delimiter is
the bare-CR error instead of a newline trimmed silently.
### Migration from 1.x
**The module path.** 2.0 lives at `sourcedock.dev/petrbalvin/interpres/v2`,
the suffix the Go toolchain requires of every major version 2 module. Change
every import and `go get` line:
```sh
go get sourcedock.dev/petrbalvin/interpres/v2
```
**TOML 1.1 only.** The acceptance contract is the TOML 1.1 corpus, and the
promise that every 1.0 document parses exactly as before is withdrawn.
Documents whose verdict changes are the ones 1.1 relaxed: `\e` and `\xHH`
escapes, times without seconds, multi-line inline tables with comments and a
trailing comma. Nothing that parsed in 1.x stops parsing, because the 1.1
grammar contains the 1.0 one.
**The output takes the 1.1 form.** A date-time writes seconds only when the
value carries them, a fraction drops its trailing zeros, and a long inline
table breaks across lines. A document written from the same value can come
out shorter; it re-parses to the same value.
**Text methods on by default.** A type implementing
`encoding.TextMarshaler` or `encoding.TextUnmarshaler` now takes the text
path with no option to switch it off. A struct that implemented the
interface encodes as a string where it was a table before. `MarshalTOML` and
`UnmarshalTOML` still win.
**One Go type per date-time kind.** Offset date-times hand back
`OffsetDateTime`, not a bare `time.Time`. Code that type-asserts the tree or
expects `time.Time` inside `UnmarshalTOML` needs the new wrapper; a
destination field of type `time.Time` keeps working.
**The document carries what the map could not.** `Parse` returns a
`*Document` with the key order, the inline distinction and the comments;
`ParseMap` gives the plain `map[string]any` tree the old `Parse` returned.
The document is writable, and `Marshal` writes it back with its comments.
**Options instead of Decoder and Encoder.** The stateful types of 1.x are
gone; the entries take variadic options, the shape encoding/json/v2 uses.
`NewDecoder().DisallowUnknownFields().Decode(data, &cfg)` becomes
`Unmarshal(data, &cfg, RejectUnknownFields(true))`, and the encoder
methods become options: `Layout(LayoutKindDeclaration)` replaces
`GroupByKind(false)`, `LiteralMultiline` replaces `UseLiteralMultiline`.
**Tag options.** `omitempty` follows encoding/json: it now also drops empty
strings, zero numbers, `false`, nil pointers and nil interfaces. `required`
demands a key at decode. `inline` forces the inline table form at encode.
`comment=text` carries a comment `EmitFieldComments` prints.
**Errors.** `DecodeError.Path` is a `Path` (segments with a `String()`
renderer), `EncodeError.Path` the same type instead of a plain string, and
both messages render `interpres: server.ports[2]: ...` with one prefix.
`SyntaxError` gained `Offset`, `Column` and `SourceLine`. Decode errors into
a `Number`-carrying tree and the fixed-size array decode are new shapes a
match on the old messages would not see.
**Decoding shapes.** `map[string]any` values merge into a non-empty map
destination; untagged embedded maps beyond the first stay empty; numbers can
stay literals under `NumbersAsLiterals`; local date-times can decode into
`time.Time` under `LocalTimeLocation`. All three are opt-in or additive
except where noted above.
## [1.1.0] - 2026-09-18
### Added
- TOML 1.1 support, on by default: date-times and times without seconds
(`07:32`, `1979-05-27T07:32`, normalised to full seconds on output), the
`\e` and `\xHH` escape sequences, and multi-line inline tables with
comments and trailing commas. The compliance suite runs in TOML 1.1 mode:
214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document
parses exactly as before.
- `interpres-decode -validate [file ...]`: a validate mode beside the
- `interpres-decode --validate [file ...]`: a validate mode beside the
toml-test adapter. It parses each named file, or stdin when none are named,
prints one line per invalid document to stderr, and exits 0 when all are
valid, 1 when one is not, and 2 on a usage or read failure. Install it with
@@ -35,8 +354,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Changed
- The compliance suite is [toml-test](https://github.com/toml-lang/toml-test)
v2.2.0, run in TOML 1.0 mode. The new corpus holds 205 valid and 474 invalid
cases (v1.6.0 had 185 and 371), and it caught the two documents the parser
v2.2.0, up from v1.6.0. Its TOML 1.0 corpus holds 205 valid and 474 invalid
cases (185 and 371 before), and it caught the two documents the parser
still accepted, fixed below.
- The flattened struct layout the decoder consults is cached per struct type
and shared with the encoder, which now resolves duplicate field keys with
@@ -150,7 +469,7 @@ uses only the standard library and passes the entire
nested structs, slices, and `map[string]T`.
- `toml:"name"` field tags, case-insensitive name fallback, and `toml:"-"` to
skip a field.
- `Decoder` with `DisallowUnknownFields` for strict decoding that rejects keys
- `RejectUnknownFields(true)` option for strict decoding that rejects keys
without a destination field, at every struct depth.
- `Unmarshaler` interface (`UnmarshalTOML(data any) error`) for types that take
full control of their decode.
+9 -6
View File
@@ -45,17 +45,19 @@ just test
formatting pass are three commits, never one.
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard
gate. Parser and decoder changes must also keep the toml-test suite at zero
failures, checked with `just toml-test`.
gate. Parser, decoder and encoder changes must also keep both directions of
the toml-test suite at zero failures, checked with `just toml-test`.
6. Update the documentation when the public API, the configuration or the
behaviour changes; the documents move in the same commit as the behaviour
they describe.
7. Open a pull request against `development`.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The
release workflow runs the full gate set including the race detector and
publishes the Gitea release with the matching `CHANGELOG.md` section as its
notes.
release workflow validates the tag, runs the static gates and the test suite
with the coverage floor, and publishes the Gitea release with the matching
`CHANGELOG.md` section as its notes. The race detector is not in that set: race
never runs on a push path, and the local `just gates` raced the tree before the
tag was cut.
## Code style
@@ -114,7 +116,8 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
|---|---|---|
| Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite |
| Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs |
| Release | a `v*` tag | the same gates plus the race detector, then the Gitea release created from the `CHANGELOG.md` section |
| Fuzz | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector |
The local equivalent is `just gates`, which is the same set plus the race
detector.
+43 -29
View File
@@ -1,39 +1,51 @@
# interpres
A TOML 1.0 and 1.1 parser and encoder for Go, written with the standard
library alone. `interpres` (Latin for *interpreter*) gives zero-dependency
A TOML 1.1 parser and encoder for Go, written with the standard library
alone. `interpres` (Latin for *interpreter*) gives zero-dependency
programs an `encoding/json`-style API for reading and writing TOML, and passes
the entire official [toml-test](https://github.com/toml-lang/toml-test) suite:
214 valid and 467 invalid cases, zero failures.
214 valid, 467 invalid and 214 encoder cases, zero failures.
## Features
- **Full TOML 1.0 and 1.1**: bare, quoted and dotted keys; tables and arrays of
- **Full TOML 1.1**: bare, quoted and dotted keys; tables and arrays of
tables; basic and literal strings including multiline, with the 1.1 `\e` and
`\xHH` escapes; integers in the four radixes with `_` separators; floats with
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds
optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and
`Marshal` for structs and maps, mirroring `encoding/json`.
- **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that
- **Decoding and encoding**: `Unmarshal` and `Marshal` for structs and maps,
mirroring `encoding/json`; `Parse` and `ParseMap` for the document with its
key order and the plain untyped tree.
- **Strict decoding**: `RejectUnknownFields(true)` rejects keys that
match no destination field, at every struct depth.
- **Custom types**: `Marshaler` and `Unmarshaler` let a type control its own
TOML representation in both directions.
- **Cancellation**: every entry point has a `*Context` sibling that honours a
`context.Context`.
- **Configurable emission**: `Encoder` options for declaration-order output,
omitting empty arrays, and literal multiline strings.
TOML representation in both directions, and `encoding.TextMarshaler` and
`TextUnmarshaler` are honoured by default, so `net.IP`, `time.Duration` and
user types with text methods need no configuration.
- **Cancellation**: the parse, decode and marshal entries have `*Context`
siblings that honour a `context.Context`, checked while the work runs.
- **Ordered documents**: `Parse` gives a `*Document` that keeps the key order,
tells an inline table from a header one, and carries the comments; `ParseMap`
gives the plain `map[string]any` tree.
- **Configurable emission**: `Marshal` options for declaration-order output,
omitting empty arrays, literal multiline strings, and inlining small
sub-tables.
## Install
As a library:
```sh
go get sourcedock.dev/petrbalvin/interpres
go get sourcedock.dev/petrbalvin/interpres/v2
```
Requires Go 1.27.1 or newer. The module imports only the standard library.
The module is public and resolves through proxy.golang.org and sum.golang.org
like any other; no GOPROXY or GOPRIVATE setup is needed to fetch it. A machine
that sets `GOPRIVATE=sourcedock.dev` fetches directly from the forge instead,
which skips the proxy and the checksum database.
## Quick start
```sh
@@ -43,8 +55,7 @@ just example
```
`just example` runs the tour in `examples/basic`: it decodes an embedded
document into a struct, prints it, and re-encodes it under both `Encoder`
layouts.
document into a struct, prints it, and re-encodes it under both layouts.
## Usage
@@ -64,8 +75,9 @@ err := interpres.Unmarshal(data, &cfg)
```
Fields match by the `toml:"name"` tag, or by the lower-cased field name when no
tag is present; `toml:"-"` skips a field. `Parse` returns the untyped
`map[string]any` tree instead, and `UnmarshalContext` accepts a context.
tag is present; `toml:"-"` skips a field. `Parse` returns a `*Document` that
also carries the key order and the comments, `ParseMap` returns the plain
`map[string]any` tree, and `UnmarshalContext` accepts a context.
### Encode from a struct
@@ -89,9 +101,7 @@ which is the layout that re-parses to the same tree.
### Strict decoding
```go
err := interpres.NewDecoder().
DisallowUnknownFields().
Decode(data, &cfg)
err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
```
A key with no matching field becomes an error instead of a silent drop.
@@ -120,16 +130,20 @@ func (ip *IP) UnmarshalTOML(data any) error {
The value `MarshalTOML` returns is encoded in place of the receiver;
`UnmarshalTOML` receives the parsed value verbatim.
### Encoder options
### Options
```go
out, err := interpres.NewEncoder().
GroupByKind(false). // preserve declaration order
OmitEmptyArrays(). // skip empty scalar arrays
UseLiteralMultiline(80). // long multi-line strings as literal blocks
Marshal(cfg)
out, err := interpres.Marshal(cfg,
interpres.Layout(interpres.LayoutKindDeclaration), // preserve declaration order
interpres.OmitEmptyArrays(true), // skip empty scalar arrays
interpres.LiteralMultiline(80), // long multi-line strings as literal blocks
)
```
The decode and encode calls take variadic options, the shape
encoding/json/v2 uses for its own. `UnmarshalRead(r, v, opts...)` and
`MarshalWrite(w, v, opts...)` are the streaming forms.
### Cancellation
```go
@@ -139,8 +153,7 @@ defer cancel()
out, err := interpres.MarshalContext(ctx, cfg)
```
`ParseContext`, `UnmarshalContext`, `(*Decoder).DecodeContext` and
`(*Encoder).MarshalContext` follow the same pattern.
`ParseContext`, `UnmarshalContext` and `MarshalContext` follow the same pattern.
The full rules for field matching, numeric conversion and emission live in
[docs/API.md](docs/API.md).
@@ -161,7 +174,8 @@ See [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the full workflow, and
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/API.md](docs/API.md): the API reference, decoding and encoding rules
- [docs/CLI.md](docs/CLI.md): the interpres-decode toml-test adapter and validator
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator,
also shipped as the manual page `man/interpres-decode.1`
## Licence
+1 -1
View File
@@ -7,7 +7,7 @@ releases do not receive them.
| Version | Supported |
|---|---|
| 1.0.0 | yes |
| 1.1.0 | yes |
| older releases | no |
## Reporting a vulnerability
+89
View File
@@ -0,0 +1,89 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"errors"
"strings"
"testing"
)
// errReader fails every read with a fixed error.
type errReader struct{ err error }
func (r errReader) Read([]byte) (int, error) { return 0, r.err }
// TestUnmarshalRead covers the streaming entry: the happy path with options,
// a failing reader, and MaxInputSize bounding what a reader is drained into.
func TestUnmarshalRead(t *testing.T) {
var got struct {
Name string `toml:"name"`
N int `toml:"n"`
}
err := UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, RejectUnknownFields(true))
if err != nil {
t.Fatalf("UnmarshalRead: %v", err)
}
if got.Name != "x" {
t.Errorf("Name = %q", got.Name)
}
readErr := errors.New("boom")
if err := UnmarshalRead(errReader{readErr}, &got); !errors.Is(err, readErr) {
t.Errorf("err = %v, want the read error wrapped", err)
}
err = UnmarshalRead(strings.NewReader("name = \"x\"\n"), &got, MaxInputSize(4))
if err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit", err)
}
// The limit bounds the read itself: a reader that would supply far more
// than the limit is not drained into memory first.
big := strings.Repeat("x", 1<<20)
if err := UnmarshalRead(strings.NewReader(big), &got, MaxInputSize(16)); err == nil || !strings.Contains(err.Error(), "over the limit") {
t.Errorf("err = %v, want the size limit before the read completes", err)
}
}
// TestParseAsWithOptions covers the generic shorthand carrying options.
func TestParseAsWithOptions(t *testing.T) {
type cfg struct {
Name string `toml:"name"`
}
got, err := ParseAs[cfg]([]byte("name = \"x\"\nrogue = 1\n"), RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), "unknown field") {
t.Errorf("err = %v, want the strict failure", err)
}
// The statements before the failure stay written, the contract the
// targeted path documents and encoding/json follows.
if got.Name != "x" {
t.Errorf("Name = %q, want the statement before the failure kept", got.Name)
}
}
// TestStatementsValueArrays pins that a value array is one statement, a
// scalar array and an array of inline tables alike; only an array of tables
// yields per element.
func TestStatementsValueArrays(t *testing.T) {
src := strings.NewReader("port = [8080, 9090]\nmix = [{y = 1, x = 2}]\n[[items]]\nn = 1\n")
var got []Statement
for stmt, err := range Statements(src) {
if err != nil {
t.Fatal(err)
}
got = append(got, stmt)
}
if len(got) != 3 {
t.Fatalf("got %d statements, want 3", len(got))
}
if got[0].Index != -1 || got[0].Table != nil {
t.Errorf("port statement = %+v, want one plain key/value", got[0])
}
if got[1].Index != -1 || got[1].Table != nil {
t.Errorf("mix statement = %+v, want one plain key/value", got[1])
}
if got[2].Index != 0 || got[2].Table == nil {
t.Errorf("items statement = %+v, want the element with its node", got[2])
}
}
+81 -5
View File
@@ -4,6 +4,7 @@
package interpres
import (
"context"
"fmt"
"strings"
"testing"
@@ -87,14 +88,14 @@ func BenchmarkParse(b *testing.B) {
b.ReportAllocs()
b.SetBytes(int64(len(benchDoc)))
for b.Loop() {
if _, err := Parse(benchDoc); err != nil {
if _, err := ParseMap(benchDoc); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkMarshal(b *testing.B) {
tree, err := Parse(benchDoc)
tree, err := ParseMap(benchDoc)
if err != nil {
b.Fatal(err)
}
@@ -108,11 +109,10 @@ func BenchmarkMarshal(b *testing.B) {
}
func BenchmarkStrictDecode(b *testing.B) {
dec := NewDecoder().DisallowUnknownFields()
b.ReportAllocs()
for b.Loop() {
var cfg benchConfig
if err := dec.Decode(benchDoc, &cfg); err != nil {
if err := Unmarshal(benchDoc, &cfg, RejectUnknownFields(true)); err != nil {
b.Fatal(err)
}
}
@@ -122,7 +122,83 @@ func BenchmarkParseLong(b *testing.B) {
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
if _, err := Parse(longDoc); err != nil {
if _, err := ParseMap(longDoc); err != nil {
b.Fatal(err)
}
}
}
// benchLongEntry mirrors one [[entry]] element of longDoc for the typed
// decode of the long document.
type benchLongEntry struct {
Name string `toml:"name"`
Weight int `toml:"weight"`
When time.Time `toml:"when"`
Ratio float64 `toml:"ratio"`
Tags []string `toml:"tags"`
}
type benchLongDoc struct {
Title string `toml:"title"`
Entry []benchLongEntry `toml:"entry"`
}
func BenchmarkStrictDecodeLong(b *testing.B) {
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
var doc benchLongDoc
if err := Unmarshal(longDoc, &doc, RejectUnknownFields(true)); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkMarshalLong(b *testing.B) {
tree, err := ParseMap(longDoc)
if err != nil {
b.Fatal(err)
}
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
if _, err := Marshal(tree); err != nil {
b.Fatal(err)
}
}
}
// BenchmarkStrictDecodeTree measures the reference path the targeted decode
// is measured against: the full tree parse followed by the reflection walk.
// The pair runs in one process, so the A/B comparison shares the machine.
func BenchmarkStrictDecodeTree(b *testing.B) {
dec := newDecoder()
dec.disallowUnknown = true
b.ReportAllocs()
for b.Loop() {
tree, _, err := parseWithOptions(context.Background(), benchDoc, parseOptions{}, false)
if err != nil {
b.Fatal(err)
}
var cfg benchConfig
if err := dec.decode(tree, &cfg); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkStrictDecodeTreeLong(b *testing.B) {
dec := newDecoder()
dec.disallowUnknown = true
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
tree, _, err := parseWithOptions(context.Background(), longDoc, parseOptions{}, false)
if err != nil {
b.Fatal(err)
}
var doc benchLongDoc
if err := dec.decode(tree, &doc); err != nil {
b.Fatal(err)
}
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package main
import (
"fmt"
"io"
"strconv"
"strings"
"time"
"unicode"
"unicode/utf8"
"sourcedock.dev/petrbalvin/interpres/v2"
)
// inferStruct reads a TOML document and writes a Go struct definition shaped
// like the document: one field per key in written order, nested tables as
// nested struct types, an array of tables as a slice, and the field names
// invented from the keys. It is the onboarding aid: the printed type compiles
// and decodes the document it came from. The definition is built whole and
// written with a single call, so a failing standard output surfaces as one
// error instead of being dropped mid-print.
func inferStruct(data []byte, stdout io.Writer) error {
doc, err := interpres.Parse(data)
if err != nil {
return err
}
body := &strings.Builder{}
fmt.Fprintln(body, "// Generated by interpres-decode --struct; decode with")
fmt.Fprintln(body, "// sourcedock.dev/petrbalvin/interpres/v2.")
fmt.Fprintln(body, "type inferred struct {")
writeInferredFields(body, tableFields(doc.Root()), map[string]bool{})
fmt.Fprintln(body, "}")
_, err = io.WriteString(stdout, body.String())
return err
}
// inferredField is one document key with the entry it is inferred from.
type inferredField struct {
key string
entry *interpres.Entry
}
// tableFields lists a table's entries in written order.
func tableFields(t *interpres.Table) []inferredField {
out := make([]inferredField, 0, len(t.Keys()))
for _, key := range t.Keys() {
entry, _ := t.Get(key)
out = append(out, inferredField{key: key, entry: entry})
}
return out
}
// mergedTableFields merges the key sets of an array's elements in first-seen
// order. An array's type has to cover every element, and a key may appear
// only in a later one, so the first element alone does not decide the shape;
// each key is inferred from the first element that carries it.
func mergedTableFields(tables []*interpres.Table) []inferredField {
var out []inferredField
seen := map[string]bool{}
for _, t := range tables {
for _, f := range tableFields(t) {
if seen[f.key] {
continue
}
seen[f.key] = true
out = append(out, f)
}
}
return out
}
// writeInferredFields writes one field per entry, in the order given.
// invented tracks the field names already used at one level, so two keys
// that clean to the same name do not collide.
func writeInferredFields(w *strings.Builder, fields []inferredField, invented map[string]bool) {
for _, f := range fields {
writeInferredField(w, f, invented)
}
}
// writeInferredField writes one field for one entry: an array of tables as a
// slice of structs, a child table as a nested struct, and everything else as
// the scalar or slice the decoded value names.
func writeInferredField(w *strings.Builder, f inferredField, invented map[string]bool) {
name := goFieldName(f.key, invented)
// An array of tables carries a node per element; the nodes of a value
// array are nil wherever an element is not a table. Every node present
// is what tells the two apart: [1, {x=1}] stays a value array even
// though one of its elements is a table.
elements := f.entry.Elements()
allTables := len(elements) > 0
for _, el := range elements {
if el == nil {
allTables = false
break
}
}
if allTables {
fmt.Fprintf(w, "\t%s []struct {\n", name)
writeInferredFields(w, mergedTableFields(elements), map[string]bool{})
fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
}
if child := f.entry.Table(); child != nil {
fmt.Fprintf(w, "\t%s struct {\n", name)
writeInferredFields(w, tableFields(child), map[string]bool{})
fmt.Fprintf(w, "\t} %s\n", structTag(f.key))
return
}
val := f.entry.Value()
if items, ok := val.([]any); ok {
fmt.Fprintf(w, "\t%s []%s %s\n", name, inferScalarType(items), structTag(f.key))
return
}
fmt.Fprintf(w, "\t%s %s %s\n", name, goTypeOf(val), structTag(f.key))
}
// structTag renders the toml tag of one key as a Go string literal. The raw
// backtick literal is the conventional shape, but a key carrying a backtick
// would end that literal early and the printed definition would not compile,
// so such tags are rendered with strconv.Quote instead.
func structTag(key string) string {
tag := `toml:"` + key + `"`
if !strings.ContainsAny(tag, "`\r") {
return "`" + tag + "`"
}
return strconv.Quote(tag)
}
// goTypeOf names the Go type the decoded value asks for.
func goTypeOf(val any) string {
switch val.(type) {
case string:
return "string"
case bool:
return "bool"
case int64:
return "int64"
case float64:
return "float64"
case interpres.OffsetDateTime:
return "interpres.OffsetDateTime"
case interpres.LocalDateTime:
return "interpres.LocalDateTime"
case interpres.LocalDate:
return "interpres.LocalDate"
case interpres.LocalTime:
return "interpres.LocalTime"
case time.Time:
return "time.Time"
case []any:
return "[]any"
case map[string]any:
return "map[string]any"
}
return "any"
}
// goFieldName cleans a document key into an exported Go identifier: the
// words the punctuation splits become capitalised runs, a leading digit
// gains a Field prefix, because an underscore would leave the field
// unexported and the decoder would skip it, and a collision with an earlier
// name gains a counter.
func goFieldName(key string, invented map[string]bool) string {
var b strings.Builder
nextUpper := true
for _, r := range key {
switch {
case unicode.IsLetter(r) || unicode.IsDigit(r):
if nextUpper {
r = unicode.ToUpper(r)
nextUpper = false
}
b.WriteRune(r)
default:
nextUpper = true
}
}
name := b.String()
if name == "" {
name = "Field"
}
// The first rune is decoded rather than taken as a byte, because a key
// may open with a digit beyond ASCII.
if first, _ := utf8.DecodeRuneInString(name); unicode.IsDigit(first) {
name = "Field" + name
}
for invented[name] {
name += "2"
}
invented[name] = true
return name
}
// inferScalarType names the Go element type of a scalar array when every
// element agrees, and any when they do not.
func inferScalarType(items []any) string {
seen := ""
for i, item := range items {
t := goTypeOf(item)
if i == 0 {
seen = t
} else if t != seen {
return "any"
}
}
if seen == "" {
return "any"
}
return seen
}
+401 -22
View File
@@ -4,14 +4,16 @@
// Command interpres-decode is the toml-test harness adapter and a TOML
// validator. Without flags it reads a TOML document from standard input and
// writes the toml-test "tagged JSON" representation to standard output. With
// -validate it checks the named documents, or standard input when none are
// named, and exits non-zero on the first invalid one:
// --encode it is the reverse: it reads tagged JSON and writes the TOML document
// it describes. With --validate it checks the named documents, or standard
// input when none are named, and exits non-zero when one is invalid:
//
// interpres-decode -validate config.toml
// interpres-decode --validate config.toml
// interpres-decode --encode < case.json
//
// Run the official suite against the adapter with:
// Run the official suite in both directions against the adapter with:
//
// toml-test ./interpres-decode
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode --encode'
package main
import (
@@ -20,12 +22,16 @@ import (
"flag"
"fmt"
"io"
"io/fs"
"math"
"os"
"path/filepath"
"runtime/debug"
"strconv"
"strings"
"time"
"sourcedock.dev/petrbalvin/interpres"
"sourcedock.dev/petrbalvin/interpres/v2"
)
func main() {
@@ -33,58 +39,238 @@ func main() {
}
// Run runs the command line and returns the process exit code: 0 success,
// 1 an invalid document, 2 a usage, reading, encoding, or
// 1 an invalid document, 2 a usage, reading, writing, encoding, or
// unsupported-value error.
func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError)
fs.SetOutput(stderr)
// The flag package's own diagnostics and default usage render flags
// with a single dash, while the command spells every flag in its
// two-dash long form, the form the manpage documents. Its output is
// therefore discarded and the usage below is the only one printed.
fs.SetOutput(io.Discard)
fs.Usage = func() {}
version := fs.Bool("version", false, "print the version and exit")
validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON")
encode := fs.Bool("encode", false, "read tagged JSON from stdin and write TOML instead")
plainJSON := fs.Bool("json", false, "with the default mode, print plain indented JSON instead of tagged JSON")
infer := fs.Bool("struct", false, "infer a Go struct definition from the document on stdin and print it")
schemaType := fs.String("schema", "", "write a TOML template for the named struct type; the source file follows as the first argument")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
usage(stdout)
return 0
}
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
usage(stderr)
return 2
}
if *version {
if _, err := fmt.Fprintf(stdout, "interpres-decode %s\n", versionString()); err != nil {
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0
}
modes := 0
for _, on := range []*bool{validate, encode, infer} {
if *on {
modes++
}
}
if *schemaType != "" {
modes++
}
if modes > 1 {
fmt.Fprintln(stderr, "interpres-decode: --validate, --encode, --struct and --schema cannot be combined")
return 2
}
// --json shapes the decoding output only, so it is rejected with every
// mode uniformly instead of being silently ignored by some of them.
if *plainJSON && modes > 0 {
fmt.Fprintln(stderr, "interpres-decode: --json shapes the decoder output and cannot be combined with --encode, --struct, --validate or --schema")
return 2
}
if *schemaType != "" {
rest := fs.Args()
if len(rest) != 1 {
fmt.Fprintln(stderr, "interpres-decode: --schema needs the type name and exactly one Go source file")
return 2
}
if err := runSchema(*schemaType, rest[0], stdout); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
return 0
}
if *validate {
return validatePaths(fs.Args(), stdin, stderr)
}
if fs.NArg() > 0 {
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with -validate")
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with --validate")
return 2
}
if *encode {
return encodeJSON(stdin, stdout, stderr)
}
data, err := io.ReadAll(stdin)
if err != nil {
fmt.Fprintln(stderr, "read stdin:", err)
fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2
}
tree, err := interpres.Parse(data)
if err != nil {
fmt.Fprintln(stderr, err)
if *infer {
if err := inferStruct(data, stdout); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
// A document that fails to parse keeps the adapter's invalid
// exit; anything else, a failed write among them, is a tool
// failure.
if _, ok := errors.AsType[*interpres.SyntaxError](err); ok {
return 1
}
return 2
}
return 0
}
tree, err := interpres.ParseMap(data)
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 1
}
if *plainJSON {
enc := json.NewEncoder(stdout)
enc.SetEscapeHTML(false)
enc.SetIndent("", " ")
if err := enc.Encode(plainJSONValue(tree)); err != nil {
fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2
}
return 0
}
tagged, err := tag(tree)
if err != nil {
fmt.Fprintln(stderr, err)
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
enc := json.NewEncoder(stdout)
enc.SetEscapeHTML(false)
if err := enc.Encode(tagged); err != nil {
fmt.Fprintln(stderr, "encode:", err)
fmt.Fprintln(stderr, "interpres-decode: encode:", err)
return 2
}
return 0
}
// usage prints the command line summary, with every flag in its two-dash
// long form: the flag package's default usage printer renders a single dash,
// and the manpage and docs/CLI.md spell the flags the way this text does.
func usage(w io.Writer) {
fmt.Fprint(w, `Usage: interpres-decode [flags]
Without a mode flag the command reads one TOML document from standard input
and writes the toml-test tagged JSON representation to standard output.
--encode read tagged JSON from standard input and write TOML
instead
--help print this usage
--json with the default mode, print plain indented JSON
instead of tagged JSON
--schema TYPE write a TOML template for the named struct type; the
Go source file follows as the first argument
--struct infer a Go struct definition from the document on
standard input and print it
--validate validate the documents instead of emitting tagged JSON
--version print the version and exit
`)
}
// versionString names the version the binary was built at: the module
// version the toolchain recorded, which is the tag when the release pipeline
// builds it, and (devel) for an ordinary build from a working tree.
func versionString() string {
if info, ok := debug.ReadBuildInfo(); ok {
if v := info.Main.Version; strings.HasPrefix(v, "v") {
return v
}
}
return "(devel)"
}
// plainJSONValue converts the parsed tree into the values encoding/json
// renders: the date-time wrappers print in their TOML form, which is the
// same text a reader of the document saw.
func plainJSONValue(v any) any {
switch x := v.(type) {
case map[string]any:
for k, val := range x {
x[k] = plainJSONValue(val)
}
return x
case []any:
for i, val := range x {
x[i] = plainJSONValue(val)
}
return x
case []map[string]any:
out := make([]any, len(x))
for i, val := range x {
out[i] = plainJSONValue(val)
}
return out
case time.Time:
return x.Format(time.RFC3339Nano)
case interpres.OffsetDateTime:
return x.String()
case interpres.LocalDateTime:
return x.String()
case interpres.LocalDate:
return x.String()
case interpres.LocalTime:
return x.String()
}
return v
}
// validatePaths parses every named file, or standard input when none are
// named, and reports each invalid document on stderr. It returns 0 when all
// documents parse, 1 when one does not, and 2 on a usage or read failure.
// named, and reports each invalid document on stderr. A named directory is
// walked for .toml files. It returns 0 when all documents parse, 1 when one
// does not, and 2 on a usage or read failure. A summary names the counts.
func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
if len(paths) == 0 {
paths = []string{"-"}
}
valid := true
var files []string
dirs := 0
for _, p := range paths {
if p == "-" {
files = append(files, "-")
continue
}
info, err := os.Stat(p)
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", p, err)
return 2
}
if !info.IsDir() {
files = append(files, p)
continue
}
dirs++
err = filepath.WalkDir(p, func(path string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if !d.IsDir() && strings.EqualFold(filepath.Ext(path), ".toml") {
files = append(files, path)
}
return nil
})
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: walk %s: %v\n", p, err)
return 2
}
}
checked := 0
invalid := 0
for _, p := range files {
name := p
var data []byte
var err error
@@ -98,17 +284,208 @@ func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
return 2
}
if _, err := interpres.Parse(data); err != nil {
fmt.Fprintf(stderr, "%s: %v\n", name, err)
valid = false
checked++
if _, err := interpres.ParseMap(data); err != nil {
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
invalid++
}
}
if !valid {
// The single-document run stays quiet on success, the contract the
// compliance tooling relies on; a directory walk closes with the
// summary that makes the sweep readable.
if dirs > 0 {
fmt.Fprintf(stderr, "checked %d documents, %d invalid\n", checked, invalid)
}
if invalid > 0 {
return 1
}
return 0
}
// encodeJSON reads a toml-test tagged JSON description from standard input and
// writes the TOML document it describes to standard output.
func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
data, err := io.ReadAll(stdin)
if err != nil {
fmt.Fprintln(stderr, "interpres-decode: read stdin:", err)
return 2
}
var desc any
if err := json.Unmarshal(data, &desc); err != nil {
fmt.Fprintln(stderr, "interpres-decode: decode JSON:", err)
return 2
}
tree, err := untag(desc)
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
doc, ok := tree.(map[string]any)
if !ok {
fmt.Fprintln(stderr, "interpres-decode: the description must be a JSON object at the top level")
return 2
}
out, err := interpres.Marshal(doc)
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: %v\n", err)
return 2
}
if _, err := stdout.Write(out); err != nil {
fmt.Fprintln(stderr, "interpres-decode: write stdout:", err)
return 2
}
return 0
}
// untag converts a toml-test JSON description into the value tree Marshal
// expects: a JSON object becomes a map[string]any, a JSON array becomes a
// []any, and an object carrying exactly the keys "type" and "value" becomes
// the Go value for that TOML type.
func untag(v any) (any, error) {
switch x := v.(type) {
case map[string]any:
if typ, val, ok := taggedValue(x); ok {
return decodeTagged(typ, val)
}
out := make(map[string]any, len(x))
for k, e := range x {
u, err := untag(e)
if err != nil {
return nil, fmt.Errorf("%s: %w", k, err)
}
out[k] = u
}
return out, nil
case []any:
out := make([]any, len(x))
for i, e := range x {
u, err := untag(e)
if err != nil {
return nil, fmt.Errorf("[%d]: %w", i, err)
}
out[i] = u
}
return asTables(out), nil
default:
return nil, fmt.Errorf("unsupported JSON value %T", v)
}
}
// asTables returns the elements as a []map[string]any when there is at least
// one and every element is a table, the shape the encoder renders as an array
// of tables. The tagged JSON cannot tell an array of tables from a value array
// of inline tables, and both parse back to the same value, so the header form
// is chosen because it is the one the encoder otherwise never exercises. An
// empty array stays a []any, because TOML has no empty array of tables.
func asTables(items []any) any {
if len(items) == 0 {
return items
}
tbls := make([]map[string]any, len(items))
for i, e := range items {
tbl, ok := e.(map[string]any)
if !ok {
return items
}
tbls[i] = tbl
}
return tbls
}
// taggedValue reports whether m is a toml-test value object: a JSON object of
// exactly the two string keys "type" and "value", carrying a type this adapter
// knows. Any other object is a table.
func taggedValue(m map[string]any) (typ, val string, ok bool) {
if len(m) != 2 {
return "", "", false
}
ts, ok := m["type"].(string)
if !ok || !knownType(ts) {
return "", "", false
}
vs, ok := m["value"].(string)
if !ok {
return "", "", false
}
return ts, vs, true
}
func knownType(typ string) bool {
switch typ {
case "string", "integer", "float", "bool",
"datetime", "datetime-local", "date-local", "time-local":
return true
}
return false
}
// decodeTagged returns the Go value for one tagged JSON value. Every type but
// string is parsed by the library itself, so the adapter and the library agree
// on what an integer, a float or a date-time is.
func decodeTagged(typ, val string) (any, error) {
if typ == "string" {
return val, nil
}
v, err := parseAtom(val)
if err != nil {
return nil, fmt.Errorf("%s %q: %w", typ, val, err)
}
// A float with no fractional part and no exponent is described by a bare
// integer literal, so here the tag decides and not the literal.
if n, ok := v.(int64); ok && typ == "float" {
return float64(n), nil
}
if !typeMatches(typ, v) {
return nil, fmt.Errorf("%s %q parsed as %T", typ, val, v)
}
return v, nil
}
// parseAtom parses one bare TOML value, by handing `v = <val>` to the library's
// parser and requiring the result to hold exactly that one statement, so a
// value carrying a newline or a comment cannot smuggle a second one in.
func parseAtom(val string) (any, error) {
tree, err := interpres.ParseMap([]byte("v = " + val + "\n"))
if err != nil {
return nil, err
}
if len(tree) != 1 {
return nil, errors.New("not a single bare value")
}
return tree["v"], nil
}
// typeMatches reports whether v is the Go value the tagged type names.
func typeMatches(typ string, v any) bool {
switch typ {
case "integer":
_, ok := v.(int64)
return ok
case "float":
_, ok := v.(float64)
return ok
case "bool":
_, ok := v.(bool)
return ok
case "datetime":
switch v.(type) {
case time.Time, interpres.OffsetDateTime:
return true
}
return false
case "datetime-local":
_, ok := v.(interpres.LocalDateTime)
return ok
case "date-local":
_, ok := v.(interpres.LocalDate)
return ok
case "time-local":
_, ok := v.(interpres.LocalTime)
return ok
}
return false
}
// tag converts an interpres value into its toml-test tagged-JSON form. Tables
// become JSON objects and arrays become JSON arrays; scalars are wrapped in a
// {"type", "value"} object. An error is returned for value types the encoder
@@ -155,6 +532,8 @@ func tag(v any) (any, error) {
return tagged("float", formatFloat(x)), nil
case time.Time:
return tagged("datetime", x.Format(time.RFC3339Nano)), nil
case interpres.OffsetDateTime:
return tagged("datetime", x.Format(time.RFC3339Nano)), nil
case interpres.LocalDateTime:
return tagged("datetime-local", x.Format("2006-01-02T15:04:05.999999999")), nil
case interpres.LocalDate:
+724 -9
View File
@@ -7,12 +7,16 @@ import (
"bytes"
"encoding/json"
"errors"
"go/parser"
"go/token"
"os"
"path/filepath"
"reflect"
"strings"
"testing"
"time"
"sourcedock.dev/petrbalvin/interpres"
"sourcedock.dev/petrbalvin/interpres/v2"
)
func TestRunParsesValidTOML(t *testing.T) {
@@ -215,7 +219,7 @@ func TestTaggedHelper(t *testing.T) {
func TestValidateStdinAcceptsValidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"ok\"\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 0 {
if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
@@ -226,7 +230,7 @@ func TestValidateStdinAcceptsValidDocument(t *testing.T) {
func TestValidateStdinRejectsInvalidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"unterminated\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 1 {
if code := Run([]string{"--validate"}, in, &stdout, &stderr); code != 1 {
t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") {
@@ -248,10 +252,10 @@ func TestValidateFiles(t *testing.T) {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", good}, nil, &stdout, &stderr); code != 0 {
if code := Run([]string{"--validate", good}, nil, &stdout, &stderr); code != 0 {
t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String())
}
if code := Run([]string{"-validate", good, bad}, nil, &stdout, &stderr); code != 1 {
if code := Run([]string{"--validate", good, bad}, nil, &stdout, &stderr); code != 1 {
t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") {
@@ -261,7 +265,7 @@ func TestValidateFiles(t *testing.T) {
func TestValidateMissingFileReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
if code := Run([]string{"--validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
@@ -272,14 +276,725 @@ func TestAdapterModeRejectsPositionalArgument(t *testing.T) {
if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "-validate") {
t.Fatalf("stderr = %q, want it to point at -validate", stderr.String())
if !strings.Contains(stderr.String(), "--validate") {
t.Fatalf("stderr = %q, want it to point at --validate", stderr.String())
}
}
func TestUnknownFlagReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-nope"}, nil, &stdout, &stderr); code != 2 {
if code := Run([]string{"--nope"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
// --- encoder mode ----------------------------------------------------------
func TestRunEncoderScalars(t *testing.T) {
in := `{
"s": {"type": "string", "value": "quote \" and backslash \\"},
"nl": {"type": "string", "value": "line1\nline2"},
"i": {"type": "integer", "value": "-9223372036854775808"},
"g": {"type": "float", "value": "1.5"},
"f": {"type": "float", "value": "inf"},
"b": {"type": "bool", "value": "false"},
"dt": {"type": "datetime", "value": "1979-05-27T07:32:00-07:00"},
"ldt": {"type": "datetime-local", "value": "1979-05-27T07:32:00"},
"ld": {"type": "date-local", "value": "1979-05-27"},
"lt": {"type": "time-local", "value": "07:32:00.999"}
}
`
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
want := "b = false\n" +
"dt = 1979-05-27T07:32-07:00\n" +
"f = inf\n" +
"g = 1.5\n" +
"i = -9223372036854775808\n" +
"ld = 1979-05-27\n" +
"ldt = 1979-05-27T07:32\n" +
"lt = 07:32:00.999\n" +
"nl = \"line1\\nline2\"\n" +
"s = \"quote \\\" and backslash \\\\\"\n"
if stdout.String() != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", stdout.String(), want)
}
}
func TestRunEncoderNested(t *testing.T) {
in := `{
"tbl": {"x": {"type": "bool", "value": "true"},
"sub": {"y": {"type": "integer", "value": "1"}}},
"items": [{"n": {"type": "string", "value": "a"}},
{"n": {"type": "string", "value": "b"}}],
"list": [{"type": "integer", "value": "1"}, {"type": "string", "value": "two"}],
"emptyTbl": {},
"emptyArr": []
}
`
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
want := "emptyArr = []\n" +
"list = [1, \"two\"]\n" +
"\n[emptyTbl]\n" +
"\n[tbl]\nx = true\n" +
"\n[tbl.sub]\ny = 1\n" +
"\n[[items]]\nn = \"a\"\n" +
"\n[[items]]\nn = \"b\"\n"
if stdout.String() != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", stdout.String(), want)
}
}
func TestRunEncoderFloatTagDecides(t *testing.T) {
// A float with no fraction is described by a bare integer literal, so the
// tag decides the type; the output must stay a float.
var stdout, stderr bytes.Buffer
in := `{"whole": {"type": "float", "value": "1"}, "exp": {"type": "float", "value": "5e+22"}}`
code := Run([]string{"--encode"}, strings.NewReader(in), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
if want := "exp = 5e+22\nwhole = 1.0\n"; stdout.String() != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", stdout.String(), want)
}
}
func TestRunEncoderRejectsBadInput(t *testing.T) {
cases := []struct {
name string
in string
want string
}{
{"not-json", "not json", "decode JSON"},
{"top-level-array", `[{"type": "integer", "value": "1"}]`, "must be a JSON object"},
{"untagged-scalar", `{"x": 1}`, "unsupported JSON value"},
{"literal-mismatch", `{"x": {"type": "integer", "value": "1.5"}}`, "parsed as float64"},
{"offset-for-local", `{"x": {"type": "datetime-local", "value": "1979-05-27T07:32:00Z"}}`, "parsed as interpres.OffsetDateTime"},
{"bad-literal", `{"x": {"type": "date-local", "value": "nope"}}`, "date-local"},
{"smuggled-statement", `{"x": {"type": "integer", "value": "1\nx = 2"}}`, "not a single bare value"},
}
for _, c := range cases {
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(c.in), &stdout, &stderr)
if code != 2 {
t.Errorf("%s: Run returned %d, want 2; stderr = %q", c.name, code, stderr.String())
continue
}
if !strings.Contains(stderr.String(), c.want) {
t.Errorf("%s: stderr = %q, want it to mention %q", c.name, stderr.String(), c.want)
}
if stdout.Len() != 0 {
t.Errorf("%s: stdout should be empty, got %q", c.name, stdout.String())
}
}
}
func TestRunEncoderFlagConflicts(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"--encode", "--validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for the two modes together", code)
}
if !strings.Contains(stderr.String(), "cannot be combined") {
t.Errorf("stderr = %q, want it to explain the conflict", stderr.String())
}
stdout.Reset()
stderr.Reset()
if code := Run([]string{"--encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
t.Errorf("Run returned %d, want 2 for an argument", code)
}
}
func TestEncodeAfterDecodeRoundTrip(t *testing.T) {
doc := `title = "x"
flt = 1.5
whole = 7.0
big = 9223372036854775807
when = 1979-05-27T07:32:00-07:00
day = 1979-05-27
clock = 07:32:00.999
list = [1, "two"]
multi = "a\nb"
[tbl]
x = true
[[items]]
n = "a"
`
var tagged, stderr bytes.Buffer
if code := Run(nil, strings.NewReader(doc), &tagged, &stderr); code != 0 {
t.Fatalf("decode returned %d, stderr = %q", code, stderr.String())
}
var out bytes.Buffer
if code := Run([]string{"--encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
t.Fatalf("encode returned %d, stderr = %q", code, stderr.String())
}
want, err := interpres.ParseMap([]byte(doc))
if err != nil {
t.Fatalf("parse of the original: %v", err)
}
got, err := interpres.ParseMap(out.Bytes())
if err != nil {
t.Fatalf("parse of the encoder output (%q): %v", out.String(), err)
}
if !reflect.DeepEqual(want, got) {
t.Errorf("round trip changed the document:\noriginal: %#v\nencoded: %#v\noutput: %q", want, got, out.String())
}
}
func TestRunVersion(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--version"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.HasPrefix(out, "interpres-decode ") {
t.Errorf("output = %q, want the version prefix", out)
}
}
func TestRunPlainJSON(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nwhen = 1979-05-27T07:32:00-07:00\nitems = [1, 2]\n")
code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "\"host\": \"db\"") {
t.Errorf("output = %q, want plain JSON keys", out)
}
if strings.Contains(out, "\"type\"") {
t.Errorf("output = %q, want no tags", out)
}
if !strings.Contains(out, "\n \"") {
t.Errorf("output = %q, want indentation", out)
}
}
func TestValidateDirectorySummary(t *testing.T) {
dir := t.TempDir()
if err := os.WriteFile(filepath.Join(dir, "good.toml"), []byte("a = 1\n"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(dir, "bad.toml"), []byte("a =\n"), 0o644); err != nil {
t.Fatal(err)
}
sub := filepath.Join(dir, "nested")
if err := os.Mkdir(sub, 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(sub, "deep.toml"), []byte("b = true\n"), 0o644); err != nil {
t.Fatal(err)
}
// A second invalid document, so the summary's invalid count is
// exercised beyond the single failure the boolean tracked.
if err := os.WriteFile(filepath.Join(sub, "worse.toml"), []byte("c =\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--validate", dir}, strings.NewReader(""), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 for a directory with invalid files", code)
}
if !strings.Contains(stderr.String(), "checked 4 documents, 2 invalid") {
t.Errorf("stderr = %q, want the summary", stderr.String())
}
}
func TestInferStruct(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("host = \"db\"\nport = 5432\ntags = [\"a\"]\n\n[server]\nname = \"edge\"\n\n[[items]]\nn = 1\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"type inferred struct {",
"Host string `toml:\"host\"`",
"Port int64 `toml:\"port\"`",
"Tags []string `toml:\"tags\"`",
"Server struct {",
"Items []struct {",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunSchemaTemplate(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := `package cfg
type Server struct {
Host string ` + "`toml:\"host,comment=The host to dial,default=example.org\"`" + `
Port int ` + "`toml:\"port,default=8080\"`" + `
}
type Config struct {
Name string ` + "`toml:\"name\"`" + `
Rate float64 ` + "`toml:\"rate,default=0.5\"`" + `
On bool ` + "`toml:\"on\"`" + `
Started time.Time ` + "`toml:\"started\"`" + `
Server Server ` + "`toml:\"server,comment=The server section\"`" + `
Items []Item ` + "`toml:\"items\"`" + `
}
type Item struct {
N int ` + "`toml:\"n\"`" + `
}
`
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"# The server section",
"[server]",
"# The host to dial",
"host = \"example.org\"",
"port = 8080",
"rate = 0.5",
"on = false",
"started = 1979-05-27T00:00:00Z",
"[[items]]",
"n = 0",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunPlainJSONShapes(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("when = 1979-05-27T07:32:00-07:00\nd = 1979-05-27\nt = 07:32:00\nwall = 1979-05-27T07:32:00\n" +
"items = [1, \"two\"]\n\n[[tables]]\nx = true\n")
code := Run([]string{"--json"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"\"when\": \"1979-05-27T07:32-07:00\"",
"\"d\": \"1979-05-27\"",
"\"t\": \"07:32\"",
"\"wall\": \"1979-05-27T07:32\"",
"\"items\": [",
"\"x\": true",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestInferStructScalarShapes(t *testing.T) {
var stdout, stderr bytes.Buffer
in := strings.NewReader("f = 1.5\nb = true\nd = 1979-05-27\nldt = 1979-05-27T07:32:00\nlt = 07:32:00\nnums = [1, 2, 3]\nmixed = [1, \"a\"]\nempty = []\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"F float64",
"B bool",
"D interpres.LocalDate",
"Ldt interpres.LocalDateTime",
"Lt interpres.LocalTime",
"Nums []int64",
"Mixed []any",
"Empty []any",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunHelpPrintsLongFlags(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--help"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, flag := range []string{"--encode", "--help", "--json", "--schema", "--struct", "--validate", "--version"} {
if !strings.Contains(out, flag) {
t.Errorf("usage output missing %q:\n%s", flag, out)
}
}
// Every flag line of the list names its flag in the two-dash long form
// only, so no line opens with a single dash.
for line := range strings.SplitSeq(strings.TrimRight(out, "\n"), "\n") {
if after, ok := strings.CutPrefix(line, " -"); ok && !strings.HasPrefix(after, "-") {
t.Errorf("usage line %q lists a flag with one dash", line)
}
}
}
func TestGoFieldName(t *testing.T) {
cases := []struct{ key, want string }{
{"host", "Host"},
{"ab", "Ab"},
{"http-host", "HttpHost"},
{"3d", "Field3d"},
{"", "Field"},
}
for _, c := range cases {
if got := goFieldName(c.key, map[string]bool{}); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestGoFieldNameCollision(t *testing.T) {
// Two keys that clean to the same name must not collide; the counter
// keeps the fields apart and both stay exported.
invented := map[string]bool{}
cases := []struct{ key, want string }{
{"a-b", "AB"},
{"a b", "AB2"},
{"a_b", "AB22"},
}
for _, c := range cases {
if got := goFieldName(c.key, invented); got != c.want {
t.Errorf("goFieldName(%q) = %q, want %q", c.key, got, c.want)
}
}
}
func TestInferStructDigitLeadingKey(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("3d = true\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Field3d bool") {
t.Errorf("output missing the exported Field3d field:\n%s", stdout.String())
}
}
func TestInferStructBacktickKey(t *testing.T) {
// A backtick in the key would end a raw string literal early, so the
// tag has to be rendered as an interpreted literal instead.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("\"a`b\" = 1\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "AB int64 \"toml:\\\"a`b\\\"\"") {
t.Errorf("output missing the quoted tag:\n%s", out)
}
// The printed definition has to compile; parsing it as Go is the
// syntax half of that proof.
if _, err := parser.ParseFile(token.NewFileSet(), "inferred.go", "package p\n\n"+out, 0); err != nil {
t.Errorf("the printed definition does not parse: %v\n%s", err, out)
}
}
func TestInferStructMergesArrayElements(t *testing.T) {
// The second element carries a key the first lacks, so the slice type
// has to be inferred from both.
var stdout, stderr bytes.Buffer
in := strings.NewReader("[[items]]\nn = 1\n\n[[items]]\nextra = \"late\"\n")
code := Run([]string{"--struct"}, in, &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{
"Items []struct {",
"N int64 `toml:\"n\"`",
"Extra string `toml:\"extra\"`",
} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestInferStructMixedArrayStaysValueArray(t *testing.T) {
// One table element does not make the array an array of tables; a
// struct slice would not decode the scalar element.
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("arr = [1, {x = 1}]\n"), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "Arr []any") {
t.Errorf("output = %q, want a value array typed []any", out)
}
if strings.Contains(out, "[]struct") {
t.Errorf("output = %q, a mixed array must not become a struct slice", out)
}
}
func TestRunStructParseError(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a =\n"), &stdout, &stderr)
if code != 1 {
t.Fatalf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 {
t.Errorf("stdout should be empty on parse error, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "line 1") {
t.Errorf("stderr = %q, want the library's line number", stderr.String())
}
}
func TestRunStructWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader("a = 1\n"), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunSchemaNeedsTypeAndExactlyOneFile(t *testing.T) {
for _, args := range [][]string{{"--schema", "Config"}, {"--schema", "Config", "a.go", "b.go"}} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--schema") {
t.Errorf("Run(%v) stderr = %q, want it to name --schema", args, stderr.String())
}
}
}
func TestRunSchemaUnparsableSource(t *testing.T) {
src := filepath.Join(t.TempDir(), "broken.go")
if err := os.WriteFile(src, []byte("this is not Go\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "broken.go") {
t.Errorf("stderr = %q, want it to name the source file", stderr.String())
}
}
func TestRunSchemaUnknownType(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Missing", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), `no struct type "Missing"`) {
t.Errorf("stderr = %q, want it to name the missing type", stderr.String())
}
}
func TestRunSchemaRecursiveType(t *testing.T) {
// A self-referential struct has no finite template; the generator has
// to name the recursion instead of exhausting the stack.
src := filepath.Join(t.TempDir(), "node.go")
body := "package cfg\n\ntype Node struct {\n\tNext *Node `toml:\"next\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Node", src}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
for _, want := range []string{"recursive", "Node"} {
if !strings.Contains(stderr.String(), want) {
t.Errorf("stderr = %q, want it to mention %q", stderr.String(), want)
}
}
}
func TestRunSchemaMultiNameField(t *testing.T) {
// A field list may name several fields of one type; each name is one
// TOML key.
src := filepath.Join(t.TempDir(), "range.go")
if err := os.WriteFile(src, []byte("package cfg\n\ntype Range struct {\n\tMin, Max int\n}\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Range", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
for _, want := range []string{"min = 0", "max = 0"} {
if !strings.Contains(out, want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
}
func TestRunSchemaEmbeddedStructs(t *testing.T) {
// The library inlines only untagged embedded structs; a tagged one
// keeps its own section.
src := filepath.Join(t.TempDir(), "embed.go")
body := `package cfg
type Inner struct {
X int ` + "`toml:\"x\"`" + `
}
type Tagged struct {
Inner ` + "`toml:\"inner\"`" + `
Y int ` + "`toml:\"y\"`" + `
}
type Flat struct {
Inner
Z int ` + "`toml:\"z\"`" + `
}
`
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
code := Run([]string{"--schema", "Tagged", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Tagged: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out := stdout.String()
if !strings.Contains(out, "y = 0") || !strings.Contains(out, "[inner]") {
t.Errorf("Tagged output = %q, want a y scalar and an [inner] section", out)
}
stdout.Reset()
stderr.Reset()
code = Run([]string{"--schema", "Flat", src}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Flat: Run returned %d, want 0; stderr = %q", code, stderr.String())
}
out = stdout.String()
if !strings.Contains(out, "x = 0") || !strings.Contains(out, "z = 0") {
t.Errorf("Flat output = %q, want x and z flattened as scalars", out)
}
if strings.Contains(out, "[inner]") {
t.Errorf("Flat output = %q, an untagged embedded struct must not become a section", out)
}
}
func TestRunVersionWriteFailure(t *testing.T) {
var stderr bytes.Buffer
code := Run([]string{"--version"}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunSchemaWriteFailure(t *testing.T) {
src := filepath.Join(t.TempDir(), "config.go")
body := "package cfg\n\ntype Config struct {\n\tA int `toml:\"a\"`\n}\n"
if err := os.WriteFile(src, []byte(body), 0o644); err != nil {
t.Fatal(err)
}
var stderr bytes.Buffer
code := Run([]string{"--schema", "Config", src}, strings.NewReader(""), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
}
func TestRunEncodeWriteFailure(t *testing.T) {
var stderr bytes.Buffer
in := `{"a": {"type": "integer", "value": "1"}}`
code := Run([]string{"--encode"}, strings.NewReader(in), errorWriter{}, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2 (write error); stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "write stdout") {
t.Errorf("stderr = %q, want it to mention the failed write", stderr.String())
}
}
func TestRunEmptyInput(t *testing.T) {
t.Run("default", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run(nil, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if strings.TrimSpace(stdout.String()) != "{}" {
t.Errorf("stdout = %q, want an empty table", stdout.String())
}
})
t.Run("encode", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--encode"}, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Fatalf("Run returned %d, want 2, empty input is not JSON", code)
}
})
t.Run("struct", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--struct"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "type inferred struct {\n}") {
t.Errorf("stdout = %q, want an empty struct", stdout.String())
}
})
t.Run("validate", func(t *testing.T) {
var stdout, stderr bytes.Buffer
code := Run([]string{"--validate"}, strings.NewReader(""), &stdout, &stderr)
if code != 0 {
t.Fatalf("Run returned %d, want 0; stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
t.Errorf("validate should be quiet, stdout %q stderr %q", stdout.String(), stderr.String())
}
})
}
func TestRunJSONFlagConflicts(t *testing.T) {
for _, args := range [][]string{
{"--encode", "--json"},
{"--struct", "--json"},
{"--validate", "--json"},
{"--json", "--schema", "Config", "config.go"},
} {
var stdout, stderr bytes.Buffer
code := Run(args, strings.NewReader(""), &stdout, &stderr)
if code != 2 {
t.Errorf("Run(%v) returned %d, want 2", args, code)
}
if !strings.Contains(stderr.String(), "--json") {
t.Errorf("Run(%v) stderr = %q, want it to explain the --json conflict", args, stderr.String())
}
}
}
+363
View File
@@ -0,0 +1,363 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package main
import (
"fmt"
"go/ast"
"go/parser"
"go/token"
"io"
"maps"
"path/filepath"
"reflect"
"slices"
"strconv"
"strings"
)
// runSchema writes a TOML template for the named struct type of a Go source
// file: one key per exported field, the comment a `comment=` tag option
// carries printed above it, and a `default=` option as the value, or the
// type's zero value where no default is given. Struct fields resolve into
// [sections], slices of them into [[array of tables]] blocks, and an
// untagged embedded struct flattens into its parent, the way the library
// decodes it.
func runSchema(typeName, sourcePath string, stdout io.Writer) error {
fset := token.NewFileSet()
file, err := parser.ParseFile(fset, sourcePath, nil, parser.ParseComments)
if err != nil {
return fmt.Errorf("%s: %w", filepath.Base(sourcePath), err)
}
types := declaredStructs(file)
st, ok := types[typeName]
if !ok {
return fmt.Errorf("no struct type %q in %s", typeName, filepath.Base(sourcePath))
}
body := &strings.Builder{}
if err := writeSchemaFields(body, st, types, "", nil); err != nil {
return err
}
_, err = io.WriteString(stdout, strings.TrimLeft(body.String(), "\n"))
return err
}
// declaredStructs collects the field lists of the file's top-level struct
// type declarations.
func declaredStructs(file *ast.File) map[string]*ast.StructType {
out := map[string]*ast.StructType{}
for _, decl := range file.Decls {
gd, ok := decl.(*ast.GenDecl)
if !ok {
continue
}
for _, spec := range gd.Specs {
ts, ok := spec.(*ast.TypeSpec)
if !ok {
continue
}
st, ok := ts.Type.(*ast.StructType)
if !ok {
continue
}
out[ts.Name.Name] = st
}
}
return out
}
// fieldMeta is what the generator reads off one struct field.
type fieldMeta struct {
key string
comment string
def string
typ ast.Expr
}
// writeSchemaFields writes the fields of one struct level: the scalar lines
// first, then the sections, so the template re-parses with every value under
// the header it belongs to. prefix is the dotted path the nested headers
// carry. path holds the struct types of the levels currently being written,
// so a type that reaches itself is reported as recursion instead of
// exhausting the stack.
func writeSchemaFields(w *strings.Builder, st *ast.StructType, types map[string]*ast.StructType, prefix string, path []*ast.StructType) error {
if slices.Contains(path, st) {
return recursionError(st, types)
}
path = append(path, st)
metas, err := metasOf(st, types)
if err != nil {
return err
}
for _, m := range metas {
if _, elemSt := elementStruct(m.typ, types); elemSt != nil {
continue
}
if isStructKind(m.typ, types) || isMapKind(m.typ) {
continue
}
writeComment(w, m.comment)
if _, ok := baseType(m.typ).(*ast.ArrayType); ok {
fmt.Fprintf(w, "%s = []\n", m.key)
continue
}
fmt.Fprintf(w, "%s = %s\n", m.key, scalarLiteral(m))
}
for _, m := range metas {
if !isStructKind(m.typ, types) && !isMapKind(m.typ) {
continue
}
writeComment(w, m.comment)
fmt.Fprintf(w, "[%s%s]\n", prefix, m.key)
if sub := structOf(m.typ, types); sub != nil {
if err := writeSchemaFields(w, sub, types, prefix+m.key+".", path); err != nil {
return err
}
}
fmt.Fprintln(w)
}
for _, m := range metas {
_, elemSt := elementStruct(m.typ, types)
if elemSt == nil {
continue
}
writeComment(w, m.comment)
fmt.Fprintf(w, "[[%s%s]]\n", prefix, m.key)
if err := writeSchemaFields(w, elemSt, types, "", path); err != nil {
return err
}
fmt.Fprintln(w)
}
return nil
}
// writeComment writes the comment lines above a binding.
func writeComment(w *strings.Builder, text string) {
if text == "" {
return
}
for line := range strings.SplitSeq(text, "\n") {
fmt.Fprintf(w, "# %s\n", line)
}
}
// metasOf flattens the exported fields of a struct. The key comes from the
// toml tag, or the lower-cased field name; a `-` key drops the field. An
// embedded struct without a tag name flattens into its parent, the way the
// library inlines it, while a tagged one keeps its own section.
func metasOf(st *ast.StructType, types map[string]*ast.StructType) ([]fieldMeta, error) {
return flattenMetas(st, types, nil)
}
// flattenMetas is metasOf with the chain of struct types currently being
// flattened, which stops a struct that embeds itself, directly or through
// another embedded type.
func flattenMetas(st *ast.StructType, types map[string]*ast.StructType, chain map[*ast.StructType]bool) ([]fieldMeta, error) {
if chain[st] {
return nil, recursionError(st, types)
}
// A copy per branch: the chain is the path being flattened now, not the
// set ever visited, so a type embedded in two siblings is not mistaken
// for recursion.
chain = maps.Clone(chain)
if chain == nil {
chain = map[*ast.StructType]bool{}
}
chain[st] = true
var out []fieldMeta
for _, field := range st.Fields.List {
tagText := ""
if field.Tag != nil {
tagText, _ = strconv.Unquote(field.Tag.Value)
}
toml := reflect.StructTag(tagText).Get("toml")
key, opts, _ := strings.Cut(toml, ",")
if len(field.Names) == 0 {
if key == "" {
// An untagged embedded struct flattens into its parent.
if ident, ok := baseType(field.Type).(*ast.Ident); ok {
if inner, ok := types[ident.Name]; ok {
metas, err := flattenMetas(inner, types, chain)
if err != nil {
return nil, err
}
out = append(out, metas...)
}
}
continue
}
if key == "-" {
continue
}
// A tagged embedded struct is a section of its own; the tag
// name is the only name it has.
out = append(out, fieldMeta{
key: key,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
continue
}
if key == "-" {
continue
}
// A field list may name several fields of one type, `Min, Max int`;
// each name is one TOML key.
for _, name := range field.Names {
if !ast.IsExported(name.Name) {
continue
}
fieldKey := key
if fieldKey == "" {
fieldKey = strings.ToLower(name.Name)
}
out = append(out, fieldMeta{
key: fieldKey,
comment: tagOption(opts, "comment="),
def: tagOption(opts, "default="),
typ: field.Type,
})
}
}
return out, nil
}
// recursionError names the struct type that reached itself. Such a type has
// no finite TOML template: every level would nest another copy of the same
// shape.
func recursionError(st *ast.StructType, types map[string]*ast.StructType) error {
return fmt.Errorf("recursive type %s: the struct contains itself, so it has no finite template", typeName(st, types))
}
// typeName names the declared struct type st refers to, and "anonymous
// struct" for a literal one that no declaration names.
func typeName(st *ast.StructType, types map[string]*ast.StructType) string {
for name, t := range types {
if t == st {
return name
}
}
return "anonymous struct"
}
// tagOption returns the text a `name=` option carries in the option part of
// a tag.
func tagOption(opts, name string) string {
for opts != "" {
var opt string
opt, opts, _ = strings.Cut(opts, ",")
if text, ok := strings.CutPrefix(opt, name); ok {
return text
}
}
return ""
}
// baseType unwraps pointers and parentheses.
func baseType(e ast.Expr) ast.Expr {
for {
switch x := e.(type) {
case *ast.StarExpr:
e = x.X
case *ast.ParenExpr:
e = x.X
default:
return e
}
}
}
// structOf returns the struct type an expression denotes when its
// declaration sits in the same file, or when it is an anonymous struct.
func structOf(e ast.Expr, types map[string]*ast.StructType) *ast.StructType {
if ident, ok := baseType(e).(*ast.Ident); ok {
return types[ident.Name]
}
if st, ok := baseType(e).(*ast.StructType); ok {
return st
}
return nil
}
// isStructKind reports whether the type is a struct the generator renders as
// a section.
func isStructKind(e ast.Expr, types map[string]*ast.StructType) bool {
return structOf(e, types) != nil
}
// isMapKind reports whether the type is a map, which renders as an empty
// section.
func isMapKind(e ast.Expr) bool {
_, ok := baseType(e).(*ast.MapType)
return ok
}
// elementStruct returns the struct type a slice's element denotes, for the
// [[array of tables]] blocks.
func elementStruct(e ast.Expr, types map[string]*ast.StructType) (ast.Expr, *ast.StructType) {
arr, ok := baseType(e).(*ast.ArrayType)
if !ok {
return nil, nil
}
return arr.Elt, structOf(arr.Elt, types)
}
// scalarLiteral renders the value line for a scalar field: the default=
// option when it is set, and the type's zero value otherwise.
func scalarLiteral(m fieldMeta) string {
kind := scalarKind(m.typ)
if m.def != "" {
if kind == "string" {
return strconv.Quote(m.def)
}
return m.def
}
switch kind {
case "int":
return "0"
case "float":
return "0.0"
case "bool":
return "false"
case "datetime":
return "1979-05-27T00:00:00Z"
}
return `""`
}
// scalarKind classifies a scalar type for the zero-value rendering.
func scalarKind(e ast.Expr) string {
switch t := baseType(e).(type) {
case *ast.Ident:
switch t.Name {
case "bool":
return "bool"
case "float32", "float64":
return "float"
case "int", "int8", "int16", "int32", "int64",
"uint", "uint8", "uint16", "uint32", "uint64", "uintptr", "byte", "rune":
return "int"
}
if t.Name != "string" {
// A named type in the file may be a scalar alias; the string
// zero value is the safe default for it and everything unknown.
return "unknown"
}
return "string"
case *ast.SelectorExpr:
if pkg, ok := t.X.(*ast.Ident); ok {
if pkg.Name == "time" && t.Sel.Name == "Time" {
return "datetime"
}
if pkg.Name == "interpres" {
switch t.Sel.Name {
case "OffsetDateTime", "LocalDateTime", "LocalDate", "LocalTime":
return "datetime"
}
}
}
}
return "unknown"
}
+257 -80
View File
@@ -5,15 +5,19 @@ package interpres
import (
"fmt"
"regexp"
"strconv"
"strings"
"time"
)
// TOML distinguishes four date-time kinds. interpres decodes an offset
// date-time to a plain time.Time (it carries a zone), and uses the wrapper
// types below for the local variants so callers can tell them apart.
// TOML distinguishes four date-time kinds, and each has its own Go type:
// OffsetDateTime for the offset kind, and the local wrappers below for the
// three that carry no offset. A plain time.Time is accepted wherever an
// offset date-time is, on both the encoding and the decoding side, so a
// timestamp field does not have to name the wrapper.
// OffsetDateTime is a TOML offset date-time, e.g. 1979-05-27T07:32:00-07:00.
// The embedded time.Time is the instant, with the offset the document wrote.
type OffsetDateTime struct{ time.Time }
// LocalDateTime is a TOML local date-time with no offset, e.g.
// 1979-05-27T07:32:00. The embedded time.Time is in UTC.
@@ -27,112 +31,285 @@ type LocalDate struct{ time.Time }
// The embedded time.Time uses the zero date.
type LocalTime struct{ time.Time }
// String returns the TOML-canonical rendering of the offset date-time, e.g.
// "1979-05-27T07:32Z" or "1979-05-27T07:32:00-07:00". The seconds appear only
// when the value carries them, a fractional second drops its trailing zeros,
// and an offset of zero is written "Z".
func (odt OffsetDateTime) String() string { return offsetString(odt.Time) }
// String returns the TOML-canonical rendering of the local date-time, e.g.
// "1979-05-27T07:32:00" or "...:00.000000123" when the time has a fractional
// second. The fractional component is zero-padded to nanosecond precision.
// "1979-05-27T07:32" or "1979-05-27T07:32:00.5" when the time carries a
// fractional second. TOML 1.1 makes the seconds optional, so they appear only
// when they are non-zero, and a fraction drops its trailing zeros.
func (ldt LocalDateTime) String() string {
base := ldt.Format("2006-01-02T15:04:05")
if ns := ldt.Nanosecond(); ns > 0 {
return base + "." + fmt.Sprintf("%09d", ns)
}
return base
buf := ldt.Time.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
return string(appendClock(buf, ldt.Time))
}
// String returns the TOML-canonical rendering of the local date, e.g.
// "1979-05-27".
func (ld LocalDate) String() string { return ld.Format("2006-01-02") }
// String returns the TOML-canonical rendering of the local time, e.g.
// "07:32:00" or "...:00.000000123" when the time has a fractional second.
// The fractional component is zero-padded to nanosecond precision.
func (lt LocalTime) String() string {
base := lt.Format("15:04:05")
if ns := lt.Nanosecond(); ns > 0 {
return base + "." + fmt.Sprintf("%09d", ns)
// String returns the TOML-canonical rendering of the local time, e.g. "07:32"
// or "07:32:00.5" when the time carries a fractional second.
func (lt LocalTime) String() string { return clockString(lt.Time) }
// appendClock appends the clock part of a TOML time to buf: HH:MM, seconds
// only when the value carries them, and a fraction with its trailing zeros
// dropped, so half a second is ".5" and not ".500000000". Both are the same
// value either way; the shorter form is the one TOML 1.1 allows. The whole
// rendering is built in one buffer, because the encoder writes a date-time
// per entry of a large document.
func appendClock(buf []byte, t time.Time) []byte {
buf = t.AppendFormat(buf, "15:04")
if t.Second() != 0 || t.Nanosecond() != 0 {
buf = t.AppendFormat(buf, ":05")
}
return base
if ns := t.Nanosecond(); ns > 0 {
buf = append(buf, '.')
buf = append(buf, strings.TrimRight(fmt.Sprintf("%09d", ns), "0")...)
}
return buf
}
var (
offsetDateTimeLayouts = []string{
"2006-01-02T15:04:05.999999999Z07:00",
"2006-01-02T15:04:05Z07:00",
"2006-01-02 15:04:05.999999999Z07:00",
"2006-01-02 15:04:05Z07:00",
// TOML 1.1 makes the seconds optional.
"2006-01-02T15:04Z07:00",
"2006-01-02 15:04Z07:00",
}
localDateTimeLayouts = []string{
"2006-01-02T15:04:05.999999999",
"2006-01-02T15:04:05",
"2006-01-02 15:04:05.999999999",
"2006-01-02 15:04:05",
"2006-01-02T15:04",
"2006-01-02 15:04",
}
localTimeLayouts = []string{
"15:04:05.999999999",
"15:04:05",
"15:04",
}
// clockString renders a time of day the way TOML writes it.
func clockString(t time.Time) string {
return string(appendClock(make([]byte, 0, 16), t))
}
// offsetString renders an offset date-time, the fourth TOML kind, in the same
// shape: no zero seconds, no trailing zeros in the fraction, and the offset
// written as "Z" when it is zero. A zone offset that is not a whole number of
// minutes loses its seconds to this rendering, which is why Marshal refuses
// such a value rather than writing it.
func offsetString(t time.Time) string {
buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
buf = appendClock(buf, t)
buf = t.AppendFormat(buf, "Z07:00")
return string(buf)
}
// dateTimeKind names the date-time shape a bare token has, as the scanner
// below classifies it.
type dateTimeKind int
const (
dateTimeNone dateTimeKind = iota
dateTimeOffset
dateTimeLocal
dateTimeDate
dateTimeClock
)
// dateTimeShape enforces the strict TOML grammar (two-digit components,
// seconds optional since 1.1, a fraction only after seconds) that time.Parse
// would otherwise accept loosely (e.g. a single-digit hour).
var dateTimeShape = regexp.MustCompile(
`^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}(:\d{2}(\.\d+)?)?([Zz]|[+-]\d{2}:\d{2})?)?$` +
`|^\d{2}:\d{2}(:\d{2}(\.\d+)?)?$`,
// The layouts the time package parses each shape with. Parsing accepts a
// fractional second even when the layout does not carry one, so each shape
// needs a single layout, chosen by whether the token has seconds.
const (
offsetDateTimeLayout = "2006-01-02T15:04:05Z07:00"
offsetClockLayout = "2006-01-02T15:04Z07:00"
localDateTimeLayout = "2006-01-02T15:04:05"
localClockLayout = "2006-01-02T15:04"
localTimeLayout = "15:04:05"
localTimeClockLayout = "15:04"
localDateOnlyLayout = "2006-01-02"
)
// offsetBounds extracts the numeric offset of a date-time. The ABNF bounds it
// to 00:00 through 23:59, but time.Parse accepts values outside that range
// and rolls them over (for example "+00:60" becomes "+01:00"), so the bounds
// are enforced here.
var offsetBounds = regexp.MustCompile(`([+-])(\d{2}):(\d{2})$`)
// scanDateTimeShape validates a bare token against the strict TOML date-time
// grammar and reports which kind it is: two-digit components, seconds
// optional since TOML 1.1, a fraction only after seconds, an offset only
// after a time, and an offset bounded to 00:00 through 23:59. The grammar is
// a fixed byte shape, so the scan is a byte walk; the regular expressions
// this replaced cost the parser measurably per token, and a shape that fails
// the scan is simply not a date-time.
func scanDateTimeShape(tok string) (kind dateTimeKind, seconds bool) {
// A local clock on its own: HH:MM[:SS[.fraction]].
if len(tok) >= 5 && tok[2] == ':' {
n, secs, ok := scanClock(tok, 0)
if !ok || n != len(tok) {
return dateTimeNone, false
}
return dateTimeClock, secs
}
// A date, optionally followed by a time and an offset.
if len(tok) < 10 || tok[4] != '-' || tok[7] != '-' {
return dateTimeNone, false
}
for _, i := range [8]int{0, 1, 2, 3, 5, 6, 8, 9} {
if !isDecDigit(tok[i]) {
return dateTimeNone, false
}
}
if len(tok) == 10 {
return dateTimeDate, false
}
if sep := tok[10]; sep != 'T' && sep != 't' && sep != ' ' {
return dateTimeNone, false
}
n, secs, ok := scanClock(tok, 11)
if !ok {
return dateTimeNone, false
}
if n == len(tok) {
return dateTimeLocal, secs
}
// The offset: Z/z, or a signed HH:MM bounded as the ABNF requires.
switch c := tok[n]; {
case c == 'Z' || c == 'z':
if n+1 != len(tok) {
return dateTimeNone, false
}
case c == '+' || c == '-':
if n+6 != len(tok) || tok[n+3] != ':' ||
!isDecDigit(tok[n+1]) || !isDecDigit(tok[n+2]) ||
!isDecDigit(tok[n+4]) || !isDecDigit(tok[n+5]) ||
tok[n+1] > '2' || (tok[n+1] == '2' && tok[n+2] > '3') ||
tok[n+4] > '5' {
return dateTimeNone, false
}
default:
return dateTimeNone, false
}
return dateTimeOffset, secs
}
// scanClock validates HH:MM[:SS[.fraction]] starting at i and returns the
// position after the clock, whether seconds were present, and whether the
// shape is valid.
func scanClock(tok string, i int) (pos int, seconds bool, ok bool) {
if i+5 > len(tok) || tok[i+2] != ':' ||
!isDecDigit(tok[i]) || !isDecDigit(tok[i+1]) ||
!isDecDigit(tok[i+3]) || !isDecDigit(tok[i+4]) {
return 0, false, false
}
i += 5
if i == len(tok) || tok[i] != ':' {
return i, false, true
}
if i+3 > len(tok) || !isDecDigit(tok[i+1]) || !isDecDigit(tok[i+2]) {
return 0, false, false
}
i += 3
if i == len(tok) || tok[i] != '.' {
return i, true, true
}
i++
digits := i
for i < len(tok) && isDecDigit(tok[i]) {
i++
}
if i == digits {
return 0, false, false
}
return i, true, true
}
// normaliseDateTimeToken rewrites the date/time separator to 'T' and the
// offset marker to 'Z', the characters the layouts above carry. A token that
// already has them is returned as it is, without a copy.
func normaliseDateTimeToken(tok string, kind dateTimeKind) string {
if kind == dateTimeDate || kind == dateTimeClock {
return tok
}
needs := false
for i := range len(tok) {
c := tok[i]
if c == 't' || c == 'z' || (c == ' ' && i == 10) {
needs = true
break
}
}
if !needs {
return tok
}
b := []byte(tok)
for i, c := range b {
switch {
case c == 't':
b[i] = 'T'
case c == 'z':
b[i] = 'Z'
case c == ' ' && i == 10:
b[i] = 'T'
}
}
return string(b)
}
// parseDateTime classifies and parses a bare token as a TOML date-time value.
// It returns the decoded value (time.Time, LocalDateTime, LocalDate, or
// LocalTime) and whether the token was a date-time at all.
func parseDateTime(tok string) (any, bool) {
// It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or
// LocalTime), whether the token was a date-time at all, and an error for a
// token whose shape is a date-time a component of which lies outside its
// range: an hour of 24, a day the month does not hold. Such a token is a
// broken date-time, not some other value, so the error names it instead of
// leaving it to the number decoder's complaint.
func parseDateTime(tok string) (any, bool, error) {
if tok == "" || tok[0] < '0' || tok[0] > '9' {
return nil, false
return nil, false, nil
}
if !strings.ContainsAny(tok, "-:") {
return nil, false
return nil, false, nil
}
if !dateTimeShape.MatchString(tok) {
return nil, false
kind, seconds := scanDateTimeShape(tok)
if kind == dateTimeNone {
return nil, false, nil
}
if m := offsetBounds.FindStringSubmatch(tok); m != nil {
hour, _ := strconv.Atoi(m[2])
minute, _ := strconv.Atoi(m[3])
if hour > 23 || minute > 59 {
return nil, false
norm := normaliseDateTimeToken(tok, kind)
switch kind {
case dateTimeOffset:
layout := offsetClockLayout
if seconds {
layout = offsetDateTimeLayout
}
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
// The ABNF accepts lowercase "t"/"z"; time.Parse only matches uppercase.
norm := strings.ToUpper(tok)
for _, layout := range offsetDateTimeLayouts {
if t, err := time.Parse(layout, norm); err == nil {
return t, true
// A zero offset carries its own anonymous location from time.Parse,
// while the written form is "Z" either way; normalising to UTC keeps
// the tree identical across the round trip.
if _, off := t.Zone(); off == 0 {
t = t.In(time.UTC)
}
return OffsetDateTime{t}, true, nil
case dateTimeLocal:
layout := localClockLayout
if seconds {
layout = localDateTimeLayout
}
for _, layout := range localDateTimeLayouts {
if t, err := time.Parse(layout, norm); err == nil {
return LocalDateTime{t}, true
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalDateTime{t}, true, nil
case dateTimeDate:
t, err := time.Parse(localDateOnlyLayout, norm)
if err != nil {
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
if t, err := time.Parse("2006-01-02", norm); err == nil {
return LocalDate{t}, true
return LocalDate{t}, true, nil
case dateTimeClock:
layout := localTimeClockLayout
if seconds {
layout = localTimeLayout
}
for _, layout := range localTimeLayouts {
if t, err := time.Parse(layout, norm); err == nil {
return LocalTime{t}, true
t, err := time.Parse(layout, norm)
if err != nil {
return nil, false, fmt.Errorf("invalid date-time %q", tok)
}
return LocalTime{t}, true, nil
}
return nil, false
return nil, false, nil
}
// wholeMinuteOffset reports an error when the zone offset carries seconds, a
// shape no TOML offset can hold: writing only the minutes would silently
// shift the instant on the way back, so the encoder refuses the value rather
// than corrupting it.
func wholeMinuteOffset(t time.Time) error {
if _, off := t.Zone(); off%60 != 0 {
return fmt.Errorf("interpres: date-time offset of %d seconds is not a whole number of minutes, which TOML cannot write", off)
}
return nil
}
// isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a
+405 -24
View File
@@ -4,23 +4,164 @@
package interpres
import (
"context"
"encoding"
"fmt"
"maps"
"reflect"
"slices"
"strings"
"sync"
"sync/atomic"
"time"
)
// decoder maps a parsed TOML tree onto Go values via reflection.
// decoder maps a parsed TOML tree onto Go values via reflection. ctx is the
// context a cancellable entry point handed in, and reaches an
// UnmarshalerContext destination; entry points without one leave it nil.
// nodes is the document's node index, present only when a destination can
// reach an OrderedMap and the parse built the tree its key order is read
// from. loc is the zone a local date-time is carried in when it decodes into
// a time.Time destination; nil keeps the wrapper-only default.
type decoder struct {
disallowUnknown bool
ctx context.Context
nodes nodeIndex
loc *time.Location
}
func newDecoder() *decoder { return &decoder{} }
// ctxOrBackground returns the context the decode carries, and Background when
// none was given, so a custom decoder never receives a nil context.
func (d *decoder) ctxOrBackground() context.Context {
if d.ctx == nil {
return context.Background()
}
return d.ctx
}
var timeType = reflect.TypeFor[time.Time]()
var (
unmarshalerType = reflect.TypeFor[Unmarshaler]()
ctxUnmarshalerType = reflect.TypeFor[UnmarshalerContext]()
textUnmarshalerType = reflect.TypeFor[encoding.TextUnmarshaler]()
numberType = reflect.TypeFor[Number]()
)
// The per-type flags record which interface lookups a decode into that type
// can succeed at, so the hot path consults the cache instead of boxing every
// value into an interface to ask. The bits name the receiver the method is
// found on: the value itself, or its address.
const (
flagUnmarshaler uint8 = 1 << iota
flagAddrUnmarshaler
flagCtxUnmarshaler
flagAddrCtxUnmarshaler
flagTextUnmarshaler
flagAddrTextUnmarshaler
)
// typeFlagCache holds one flag entry per destination type. A set is immutable
// once published, the same trade-off structSchemaCache makes; the cache grows
// with the number of distinct types decoded, never per document. The hint
// below re-points at these published entries, so a hot lookup allocates
// nothing.
var typeFlagCache sync.Map // reflect.Type -> *flagHintEntry
// flagHintEntry pairs a type with its cached flags for the monomorphic hint
// below. Both caches share the entry shape.
type flagHintEntry struct {
typ reflect.Type
flags uint8
}
// typeFlagHint remembers the entry resolved last, because a decode walks one
// type across consecutive fields and elements. A lost race loses only the
// hint: every value it can hold came from the cache.
var typeFlagHint atomic.Pointer[flagHintEntry]
func typeFlags(t reflect.Type) uint8 {
if e := typeFlagHint.Load(); e != nil && e.typ == t {
return e.flags
}
if v, ok := typeFlagCache.Load(t); ok {
entry := v.(*flagHintEntry)
typeFlagHint.Store(entry)
return entry.flags
}
var f uint8
if t.Implements(unmarshalerType) {
f |= flagUnmarshaler
}
if t.Implements(ctxUnmarshalerType) {
f |= flagCtxUnmarshaler
}
pt := reflect.PointerTo(t)
if pt.Implements(unmarshalerType) {
f |= flagAddrUnmarshaler
}
if pt.Implements(ctxUnmarshalerType) {
f |= flagAddrCtxUnmarshaler
}
// The date-time types are excluded from the text path: they carry
// time.Time's UnmarshalText through an embedded field while their only
// accepted form is a bare timestamp.
if !isDateTimeType(t) {
if t.Implements(textUnmarshalerType) {
f |= flagTextUnmarshaler
}
if pt.Implements(textUnmarshalerType) {
f |= flagAddrTextUnmarshaler
}
}
actual, _ := typeFlagCache.LoadOrStore(t, &flagHintEntry{t, f})
published := actual.(*flagHintEntry)
typeFlagHint.Store(published)
return published.flags
}
// unmarshalerOf resolves the Unmarshaler for dst through the flag cache, so
// an interface value is built only where the cache says the assertion can
// succeed. An interface destination is asked dynamically, because the value
// it will hold may implement the interface even when the interface type
// itself does not.
func unmarshalerOf(dst reflect.Value) (Unmarshaler, bool) {
if dst.Kind() == reflect.Interface {
u, ok := dst.Interface().(Unmarshaler)
return u, ok
}
f := typeFlags(dst.Type())
if f&flagUnmarshaler != 0 {
u, ok := dst.Interface().(Unmarshaler)
return u, ok
}
if f&flagAddrUnmarshaler != 0 && dst.CanAddr() {
u, ok := dst.Addr().Interface().(Unmarshaler)
return u, ok
}
return nil, false
}
// ctxUnmarshalerOf is the same resolution for UnmarshalerContext.
func ctxUnmarshalerOf(dst reflect.Value) (UnmarshalerContext, bool) {
if dst.Kind() == reflect.Interface {
u, ok := dst.Interface().(UnmarshalerContext)
return u, ok
}
f := typeFlags(dst.Type())
if f&flagCtxUnmarshaler != 0 {
u, ok := dst.Interface().(UnmarshalerContext)
return u, ok
}
if f&flagAddrCtxUnmarshaler != 0 && dst.CanAddr() {
u, ok := dst.Addr().Interface().(UnmarshalerContext)
return u, ok
}
return nil, false
}
func (d *decoder) decode(tree map[string]any, v any) error {
rv := reflect.ValueOf(v)
if rv.Kind() != reflect.Pointer || rv.IsNil() {
@@ -45,12 +186,18 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
return nil
}
// Types implementing Unmarshaler get the parsed data wholesale and are
// responsible for setting their own state. The decoder does not consult
// any return value; whatever the receiver stores is kept. The lookup
// covers both T and *T so a pointer-receiver UnmarshalTOML method is
// invoked on an addressable struct field.
// Types implementing UnmarshalerContext get the context beside the parsed
// data, and are responsible for setting their own state. They win over
// Unmarshaler, which wins over the text path. The lookups cover both T and
// *T so a pointer-receiver method is invoked on an addressable struct
// field.
if dst.CanInterface() {
if u, ok := ctxUnmarshalerOf(dst); ok {
if err := u.UnmarshalTOMLContext(d.ctxOrBackground(), data); err != nil {
return fmt.Errorf("unmarshal: %w", err)
}
return nil
}
u, ok := dst.Interface().(Unmarshaler)
if !ok && dst.CanAddr() {
u, ok = dst.Addr().Interface().(Unmarshaler)
@@ -63,6 +210,19 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
}
}
// A TOML string fills a destination that implements
// encoding.TextUnmarshaler, the rule encoding/json follows. Every other
// value kind keeps its own rule, so an integer still reaches a numeric
// destination.
if s, isString := data.(string); isString {
if tu, ok := textUnmarshalerOf(dst); ok {
if err := tu.UnmarshalText([]byte(s)); err != nil {
return fmt.Errorf("unmarshal text: %w", err)
}
return nil
}
}
switch v := data.(type) {
case map[string]any:
return d.assignTable(v, dst)
@@ -71,19 +231,40 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
case []any:
return d.assignSlice(v, dst)
case string:
if dst.Type() == durationType {
return setDuration(dst, v)
}
return setBasic(dst, reflect.ValueOf(v), "string")
case Number:
return setNumber(dst, v)
case bool:
return setBasic(dst, reflect.ValueOf(v), "bool")
case int64:
return setInt(dst, v)
case float64:
return setFloat(dst, v)
case OffsetDateTime:
return setOffsetDateTime(v, dst)
case time.Time:
if dst.Type() != timeType {
return fmt.Errorf("interpres: cannot assign datetime to %s", dst.Type())
}
return setDateTime(v, dst)
case LocalDateTime:
if dst.Type() == localDateTimeType {
dst.Set(reflect.ValueOf(v))
return nil
}
return d.setLocalTimeValue(v.Time, dst)
case LocalDate:
if dst.Type() == localDateType {
dst.Set(reflect.ValueOf(v))
return nil
}
return d.setLocalTimeValue(v.Time, dst)
case LocalTime:
if dst.Type() == localTimeType {
dst.Set(reflect.ValueOf(v))
return nil
}
return d.setLocalTimeValue(v.Time, dst)
default:
rv := reflect.ValueOf(data)
if rv.IsValid() && dst.Type() == rv.Type() {
@@ -94,7 +275,32 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
}
}
// textUnmarshalerOf is the same resolution for encoding.TextUnmarshaler,
// with the date-time types excluded for the reason typeFlags records.
func textUnmarshalerOf(dst reflect.Value) (encoding.TextUnmarshaler, bool) {
if !dst.CanInterface() || isDateTimeType(dst.Type()) {
return nil, false
}
if dst.Kind() == reflect.Interface {
tu, ok := dst.Interface().(encoding.TextUnmarshaler)
return tu, ok
}
f := typeFlags(dst.Type())
if f&flagTextUnmarshaler != 0 {
tu, ok := dst.Interface().(encoding.TextUnmarshaler)
return tu, ok
}
if f&flagAddrTextUnmarshaler != 0 && dst.CanAddr() {
tu, ok := dst.Addr().Interface().(encoding.TextUnmarshaler)
return tu, ok
}
return nil, false
}
func (d *decoder) assignTable(tbl map[string]any, dst reflect.Value) error {
if dst.Type() == orderedMapType {
return d.fillOrderedMap(tbl, dst)
}
switch dst.Kind() {
case reflect.Struct:
return d.assignStruct(tbl, dst)
@@ -112,6 +318,9 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
// deterministically: the smallest one.
unknown := ""
for key := range tbl {
if _, ok := schema.byName[key]; ok {
continue
}
if _, ok := schema.byName[strings.ToLower(key)]; ok {
continue
}
@@ -123,22 +332,42 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
return fmt.Errorf("interpres: unknown field %q for %s", unknown, dst.Type())
}
}
for key, val := range tbl {
field, ok := schema.byName[strings.ToLower(key)]
// The keys that resolved to a field are remembered while the table walks,
// but only a struct that demands one pays for the set.
var seen map[string]bool
if len(schema.required) > 0 {
seen = make(map[string]bool, len(tbl))
}
for _, key := range d.tableKeys(tbl) {
val := tbl[key]
// A key that is already lowercase, which document keys usually are,
// hits the map directly; only a miss pays for the case fold.
resolved := key
field, ok := schema.byName[key]
if !ok {
resolved = strings.ToLower(key)
field, ok = schema.byName[resolved]
}
if !ok {
if schema.embedMaps != nil {
// Leftover keys land in an untagged embedded map, the inverse
// of the encoder inlining that map's entries.
// of the encoder inlining that map's entries. The assign call
// rather than assignMap itself lets it allocate the embedded
// pointer the field may be, the way any other destination is
// reached.
mv, err := fieldByIndex(dst, schema.embedMaps[0])
if err != nil {
return newDecodeError(key, err)
}
if err := d.assignMap(map[string]any{key: val}, mv); err != nil {
if err := d.assign(map[string]any{key: val}, mv); err != nil {
return newDecodeError(key, err)
}
}
continue
}
if seen != nil {
seen[resolved] = true
}
fv, err := fieldByIndex(dst, field.index)
if err != nil {
return newDecodeError(key, err)
@@ -147,9 +376,26 @@ func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
return newDecodeError(key, err)
}
}
for _, key := range schema.required {
if !seen[key] {
return fmt.Errorf("interpres: missing required key %q", key)
}
}
return nil
}
// tableKeys returns the keys of tbl in the order the document wrote them
// when the node index knows it, and in sorted order otherwise, the order a
// hand-built tree or a node-free parse offers. The order settles which of
// two keys that differ only in case wins one field: the same key wins every
// run, instead of whichever a map iteration happened to hand out.
func (d *decoder) tableKeys(tbl map[string]any) []string {
if node := d.nodeOf(tbl); node != nil {
return node.Keys()
}
return slices.Sorted(maps.Keys(tbl))
}
func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
if dst.Type().Key().Kind() != reflect.String {
return fmt.Errorf("interpres: map key must be a string, got %s", dst.Type().Key())
@@ -169,9 +415,8 @@ func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
}
func (d *decoder) assignSlice(items []any, dst reflect.Value) error {
if dst.Kind() != reflect.Slice {
return fmt.Errorf("interpres: cannot assign array to %s", dst.Type())
}
switch dst.Kind() {
case reflect.Slice:
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil {
@@ -180,12 +425,27 @@ func (d *decoder) assignSlice(items []any, dst reflect.Value) error {
}
dst.Set(out)
return nil
case reflect.Array:
// A fixed-size array takes the elements in place; a length mismatch is
// the error, because a TOML array carries no way to name a default for
// the elements it is short of, and the surplus has nowhere to go.
if dst.Len() != len(items) {
return fmt.Errorf("interpres: cannot assign %d elements to %s", len(items), dst.Type())
}
for i, item := range items {
if err := d.assign(item, dst.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
return nil
default:
return fmt.Errorf("interpres: cannot assign array to %s", dst.Type())
}
}
func (d *decoder) assignTableSlice(items []map[string]any, dst reflect.Value) error {
if dst.Kind() != reflect.Slice {
return fmt.Errorf("interpres: cannot assign array of tables to %s", dst.Type())
}
switch dst.Kind() {
case reflect.Slice:
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil {
@@ -194,18 +454,119 @@ func (d *decoder) assignTableSlice(items []map[string]any, dst reflect.Value) er
}
dst.Set(out)
return nil
case reflect.Array:
if dst.Len() != len(items) {
return fmt.Errorf("interpres: cannot assign %d elements to %s", len(items), dst.Type())
}
for i, item := range items {
if err := d.assign(item, dst.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err)
}
}
return nil
default:
return fmt.Errorf("interpres: cannot assign array of tables to %s", dst.Type())
}
}
// --- low-level setters -----------------------------------------------------
// setOffsetDateTime stores an offset date-time: in a wrapper destination as it
// is, and in a plain time.Time, which takes the instant with the offset the
// document wrote, so a timestamp field does not have to name the wrapper.
func setOffsetDateTime(v OffsetDateTime, dst reflect.Value) error {
switch dst.Type() {
case offsetDateTimeType:
dst.Set(reflect.ValueOf(v))
case timeType:
dst.Set(reflect.ValueOf(v.Time))
default:
return fmt.Errorf("interpres: cannot assign datetime to %s", dst.Type())
}
return nil
}
// setDateTime stores a time.Time that reached the tree directly, which is the
// shape a tree built by hand carries. Dates the parser produced arrive as
// OffsetDateTime instead.
func setDateTime(v time.Time, dst reflect.Value) error {
switch dst.Type() {
case timeType:
dst.Set(reflect.ValueOf(v))
case offsetDateTimeType:
dst.Set(reflect.ValueOf(OffsetDateTime{v}))
default:
return fmt.Errorf("interpres: cannot assign datetime to %s", dst.Type())
}
return nil
}
// setLocalTimeValue stores a local date-time value into a plain time.Time
// destination, which the decoder permits only when LocalTimeLocation fixed
// the zone the wall-clock value is carried in; without it the wrapper types
// are the only destinations a local kind fills, as they always have been.
func (d *decoder) setLocalTimeValue(t time.Time, dst reflect.Value) error {
if dst.Type() == timeType {
if d.loc != nil {
// A local value is a wall clock, so the zone choice relabels it
// rather than shifting the instant: 07:32 in the document is
// 07:32 in the location, not an hour later.
dst.Set(reflect.ValueOf(time.Date(
t.Year(), t.Month(), t.Day(),
t.Hour(), t.Minute(), t.Second(), t.Nanosecond(), d.loc)))
return nil
}
return fmt.Errorf("interpres: cannot assign local date-time to time.Time; set LocalTimeLocation to choose the zone")
}
return fmt.Errorf("interpres: cannot assign local date-time to %s", dst.Type())
}
func setBasic(dst, val reflect.Value, kind string) error {
if dst.Kind() != val.Kind() {
return fmt.Errorf("interpres: cannot assign %s to %s", kind, dst.Type())
}
dst.Set(val)
// Convert rather than assign: a value of the predeclared type is not
// assignable to a defined type of the same kind, so a plain Set panics on
// a destination such as `type Name string`.
dst.Set(val.Convert(dst.Type()))
return nil
}
// setDuration reads a duration literal into a time.Duration destination. TOML
// has no duration type, so the encoder writes the canonical Go form and the
// decoder reads that back; a bare integer stays the nanosecond count it has
// always been, and reaches the destination through setInt.
func setDuration(dst reflect.Value, s string) error {
d, err := time.ParseDuration(s)
if err != nil {
return fmt.Errorf("interpres: invalid duration %q", s)
}
dst.SetInt(int64(d))
return nil
}
// setNumber stores a Number, the literal NumbersAsLiterals keeps. A Number destination
// takes the literal as it is; every other destination takes the evaluated
// value through the ordinary rules, so an integer field, a float field and a
// duration field all read a Number the way they read the evaluated kind.
func setNumber(dst reflect.Value, n Number) error {
if dst.Type() == numberType {
dst.SetString(string(n))
return nil
}
v, err := decodeNumber(string(n))
if err != nil {
return fmt.Errorf("interpres: %w", err)
}
switch v := v.(type) {
case int64:
return setInt(dst, v)
case float64:
return setFloat(dst, v)
}
return fmt.Errorf("interpres: cannot assign number to %s", dst.Type())
}
func setInt(dst reflect.Value, v int64) error {
switch dst.Kind() {
case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64:
@@ -253,10 +614,12 @@ func setFloat(dst reflect.Value, v float64) error {
// structFieldLoc locates one destination field by its index path from the
// struct root and by the depth the field sits at, which breaks name clashes
// in favour of the shallower field.
// in favour of the shallower field. required records the tag option of the
// field that won the name.
type structFieldLoc struct {
index []int
depth int
required bool
}
// structSchema flattens the exported fields of t for decode, mirroring the
@@ -264,10 +627,11 @@ type structFieldLoc struct {
// keys of the same table, and an untagged embedded map is recorded in
// embedMaps (first declaration first) as the destination for leftover keys.
// When two fields resolve to one name, the shallower wins, then the later
// declaration.
// declaration. required holds the keys a `toml:"...,required"` tag demands.
type structSchema struct {
byName map[string]structFieldLoc
embedMaps [][]int
required []string
}
// structSchemaCache holds one schema per struct type. A schema is immutable
@@ -303,11 +667,20 @@ func newStructSchema(t reflect.Type) structSchema {
}
path := append(append([]int{}, prefix...), i)
name := ""
required := false
if tag, ok := f.Tag.Lookup("toml"); ok {
name, _, _ = strings.Cut(tag, ",")
var opts string
name, opts, _ = strings.Cut(tag, ",")
if name == "-" {
continue
}
for opts != "" {
var opt string
opt, opts, _ = strings.Cut(opts, ",")
if opt == "required" {
required = true
}
}
}
if f.Anonymous && name == "" {
ft := f.Type
@@ -331,11 +704,19 @@ func newStructSchema(t reflect.Type) structSchema {
}
key := strings.ToLower(name)
if existing, ok := s.byName[key]; !ok || depth <= existing.depth {
s.byName[key] = structFieldLoc{index: path, depth: depth}
s.byName[key] = structFieldLoc{index: path, depth: depth, required: required}
}
}
}
walk(t, nil, 0)
// The missing-key error must not depend on map order, so the demanded keys
// come out sorted.
for key, loc := range s.byName {
if loc.required {
s.required = append(s.required, key)
}
}
slices.Sort(s.required)
return s
}
+1061 -14
View File
File diff suppressed because it is too large Load Diff
+615 -103
View File
@@ -1,32 +1,230 @@
# API
The library exports the surface below from the `sourcedock.dev/petrbalvin/interpres`
The library exports the surface below from the `sourcedock.dev/petrbalvin/interpres/v2`
package. The snippets assume:
```go
import "sourcedock.dev/petrbalvin/interpres"
import "sourcedock.dev/petrbalvin/interpres/v2"
```
The parser implements TOML 1.1: date-times and times without seconds, the
`\e` and `\xHH` escape sequences, and multi-line inline tables with comments
and trailing commas. The encoder emits TOML 1.1.
## Functions
### `func Parse(data []byte) (map[string]any, error)`
### `func Parse(data []byte) (*Document, error)`
Decodes a TOML document into an untyped tree, using the value mapping in the
[Decoding](#decoding) section below. Returns `*SyntaxError` on a malformed
document. Input that is not valid UTF-8 is rejected before the parser runs.
Equivalent to `ParseContext(context.Background(), data)`.
Decodes a TOML document into a [Document](#documents): the values, the order the
keys were written in, whether a table was written inline, and the comments.
The values follow the mapping in the [Decoding](#decoding) section below.
Returns `*SyntaxError` on a malformed document. Input that is not valid UTF-8
is rejected with a `SyntaxError` naming the line where the invalid byte
appears, because validity is checked during the scan. Equivalent to
`ParseContext(context.Background(), data)`.
```go
tree, err := interpres.Parse([]byte("title = \"x\"\nport = 8080\n"))
doc, err := interpres.Parse([]byte("title = \"x\"\nport = 8080\n"))
tree := doc.Map()
```
### `func ParseContext(ctx context.Context, data []byte) (map[string]any, error)`
### `func ParseContext(ctx context.Context, data []byte) (*Document, error)`
The cancellable variant of `Parse`. An already-cancelled context returns
`ctx.Err()` before any work. During parsing the context is checked every 64
top-level statements, so a long document aborts without running to completion.
### `func Unmarshal(data []byte, v any) error`
### `func ParseMap(data []byte) (map[string]any, error)`
Decodes a TOML document into an untyped tree, the shape this package parsed
into before [Document](#documents) existed: the order of the keys and the
comments are not part of a map, so they are dropped. Use it when only the
values matter, or when the extra bookkeeping of a document is not wanted.
Equivalent to `ParseMapContext(context.Background(), data)`.
```go
tree, err := interpres.ParseMap([]byte("title = \"x\"\nport = 8080\n"))
```
### `func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error)`
The cancellable variant of `ParseMap`.
### `func ParseFile(path string) (*Document, error)`
Reads the file at `path` and parses it into a [Document](#documents), the shape
`Parse` gives. Both a read failure and a parse failure come back with the file
name as their first words, wrapped so `errors.AsType` still reaches the
`SyntaxError` inside a parse failure.
```go
doc, err := interpres.ParseFile("config.toml")
```
### `func Valid(data []byte) error`
Reports whether `data` is a valid TOML document: `nil` when the parser accepts
it, the parse error when it does not. It is the library call the `-validate`
mode of interpres-decode is built on.
```go
if err := interpres.Valid(data); err != nil {
fmt.Println("invalid:", err)
}
```
### Options
The decode and encode calls take variadic options, the shape
encoding/json/v2 uses for its own. Each is a function value over the private
settings of one call, and they compose by listing:
```go
cfg, err := interpres.Unmarshal(data, &cfg2,
interpres.RejectUnknownFields(true),
interpres.NumbersAsLiterals(true))
```
Decode options:
| Option | Default | Effect |
|---|---|---|
| `RejectUnknownFields(v bool)` | off | a key with no matching struct field is an error |
| `NumbersAsLiterals(v bool)` | off | integers and floats decode into `Number`, which carries the literal; see [Numbers as literals](#numbers-as-literals) |
| `MaxNestingDepth(depth int)` | `10000` | bound how deeply arrays and inline tables may nest |
| `MaxInputSize(size int)` | no limit | bound the size of the document, in bytes |
| `LocalTimeLocation(loc)` | nil | the zone a local date-time is carried in when it decodes into a `time.Time` |
Encode options:
| Option | Default | Effect |
|---|---|---|
| `Layout(kind LayoutKind)` | `LayoutKindGrouped` | group entries as scalars, then sub-tables, then arrays of tables; `LayoutKindDeclaration` preserves declaration order |
| `OmitEmptyArrays(v bool)` | off | skip `key = []` for empty scalar arrays |
| `LiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
| `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header |
### `func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error)`
The generic shorthand for `Unmarshal` with a destination variable:
```go
cfg, err := interpres.ParseAs[Config](data)
```
The zero `T` comes back with the error.
### `func NewSchema[T any]()`
Precompiles the codec for `T`: the struct schema both directions walk and the
interface flags the decoder and encoder resolve through are built once and
cached, so the first document pays the cost instead of the hot path. A `T`
that is not a struct warms nothing.
### `func Statements(r io.Reader) iter.Seq2[Statement, error]`
Iterates the top-level statements of the document r carries, in written
order: key/value statements, a value array or an inline table among them as
one statement whatever it holds, a `[table]` header as one statement carrying
its `Table` node, and an `[[array of tables]]` as one statement per element
with the element's node and its `Index`. Iteration stops at the first error
and at a false yield, so a caller looking for one section reads no further.
The reader is consumed in full before the first yield, because the parser
scans the source in place.
```go
for stmt, err := range interpres.Statements(file) {
if err != nil {
return err
}
if stmt.Table != nil {
fmt.Println(stmt.Key, stmt.Table.Keys())
}
}
```
## Documents
`Parse` returns a `Document`: the value tree together with what a map cannot
carry, which is the order the keys were written in, whether a table was written
as an inline table or under a header, and the comments. `ParseMap` gives the
plain tree when none of that is wanted.
```go
doc, err := interpres.Parse(data)
if err != nil {
return err
}
root := doc.Root()
for _, key := range root.Keys() { // written order, not sorted
entry, _ := root.Get(key)
fmt.Println(key, entry.Value())
}
```
The values are shared with the tree `ParseMap` returns, so a value read from a
document and from `doc.Map()` is the same value.
| Type | Meaning |
|---|---|
| `Document` | the parsed document: `Root()` for the top-level table, `Map()` for the value tree, `Footer()` for a comment block at the end |
| `Table` | one TOML table: `Keys()` and `Entries()` in written order, `Get(key)`, `Values()` for its part of the value tree, `Inline()` |
| `Entry` | one key: `Value()`, `Inline()`, `Table()` when the value is a table, `Elements()` for the tables of an array value |
`Elements()` holds one node per element of an array value: the tables of an
array of tables, and the inline tables inside a value array, with `nil` for the
elements that are not tables.
### Editing a document
The document is writable, which makes the read-change-write loop a round trip
through one value. The typed getters read with one call:
| Getter | Returns |
|---|---|
| `GetString(key)` | `(string, bool)` |
| `GetInt(key)` | `(int64, bool)` |
| `GetFloat(key)` | `(float64, bool)` |
| `GetBool(key)` | `(bool, bool)` |
| `GetArray(key)` | `([]any, bool)` |
| `GetTable(key)` | `(*Table, bool)` |
`Set(key, value)` stores a value, keeping an existing key's position and
comments and appending a new key to the end; a `map[string]any` value becomes
a table of its own under a header, its keys in sorted order. `Delete(key)`
removes a key and everything it holds. Every method exists on `Document` for
the root table and on `Table` for the table itself.
`Marshal` writes the document back as it stands: keys in written order, the
comments above the lines and headers they belonged to, tables that were
written inline written inline again. `UnmarshalDocument(doc, v)` decodes the
edited document into a typed destination without parsing again.
```go
doc, err := interpres.Parse(data)
doc.Set("port", 9090)
out, err := interpres.Marshal(doc)
```
### Comments
A comment belongs to the line it precedes or follows, and to the node that line
introduced:
| Written | Carried by |
|---|---|
| lines above a key | that key's `Entry`, through `Comments()` |
| a comment beside a key | that key's `Entry`, through `Trailing()` |
| lines above a `[header]` or `[[header]]` | that `Table`, through `Comments()` |
| a comment beside a header | that `Table`, through `Trailing()` |
| a comment block after the last statement | the `Document`, through `Footer()` |
`SetComments` and `SetTrailing` replace them. A line carries no leading `#`
and no surrounding space, so `# note` is stored as `note` and a bare `#` as
`""`.
### `func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error`
Parses `data` and stores the result in the value pointed to by `v`, typically a
pointer to a struct or to `map[string]any`. Equivalent to
@@ -39,11 +237,11 @@ if err := interpres.Unmarshal(data, &cfg); err != nil {
}
```
### `func UnmarshalContext(ctx context.Context, data []byte, v any) error`
### `func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...UnmarshalOption) error`
The cancellable variant of `Unmarshal`.
### `func Marshal(v any) ([]byte, error)`
### `func Marshal(v any, opts ...MarshalOption) ([]byte, error)`
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
TOML document. The emission rules are in the [Encoding](#encoding) section
@@ -53,11 +251,16 @@ below. Equivalent to `MarshalContext(context.Background(), v)`.
out, err := interpres.Marshal(cfg)
```
### `func MarshalContext(ctx context.Context, v any) ([]byte, error)`
### `func MarshalContext(ctx context.Context, v any, opts ...MarshalOption) ([]byte, error)`
The cancellable variant of `Marshal`. The context is checked before any work
and every 64 fields during the reflection walk.
### `func MarshalAppend(buf []byte, v any, opts ...MarshalOption) ([]byte, error)`
Appends the TOML encoding of `v` to `buf` and returns the extended buffer, the
shape `json.MarshalAppend` has. A failed encoding leaves `buf` untouched.
## Decoding
### Value mapping
@@ -70,7 +273,7 @@ and every 64 fields during the reflection walk.
| integer | `int64` |
| float | `float64` |
| boolean | `bool` |
| offset date-time | `time.Time` |
| offset date-time | `OffsetDateTime` |
| local date-time | `LocalDateTime` |
| local date | `LocalDate` |
| local time | `LocalTime` |
@@ -80,15 +283,20 @@ and every 64 fields during the reflection walk.
When decoding into a struct, these values convert onto the destination's
concrete types: any integer or unsigned width, floats, slices, nested structs
and `map[string]T`.
and `map[string]T`. `NumbersAsLiterals` replaces the two numeric rows of the
table with `Number`, which keeps the literal; see
[Numbers as literals](#numbers-as-literals).
### Target constraints
`Unmarshal` and `(*Decoder).Decode` write into a non-nil pointer:
`Unmarshal`, `UnmarshalRead` and `UnmarshalContext` write into a non-nil pointer:
- `*struct`, matched per the field rules below
- `*map[string]any` or `*map[string]T`, keys become map keys and values decode
into `T` recursively
into `T` recursively; a map that already holds entries is merged into, the
document's values replacing same-named keys and the rest left standing
- `*OrderedMap`, the keys fill in the order the document wrote them; see
[Ordered tables](#ordered-tables)
- `*any`, receives the whole parsed tree unchanged
Anything else returns `interpres: decode target must be a non-nil pointer`.
@@ -104,7 +312,9 @@ For a struct destination, a TOML key matches a field as follows:
into the embedded struct and matches its own fields against the same keys,
mirroring how the encoder flattens it. A nil embedded pointer struct is
allocated on demand. An untagged embedded map receives the keys no field
claims.
claims; when a struct embeds several untagged maps, the first one
declared takes all of them and the rest stay untouched, so the rule stays
predictable.
4. The key itself is lower-cased before lookup, so the match is
case-insensitive on both sides: `DATABASEURL` matches a field named
`DatabaseUrl`.
@@ -118,6 +328,11 @@ one declared later wins.
Unknown keys are ignored by default, landing in an untagged embedded map when
the struct has one; [Strict decoding](#strict-decoding) rejects them instead.
The tag may carry the `required` option, `toml:"host,required"`: the decode
fails with `missing required key "host"` when no key of the document resolved
to the field. The check runs after the table is read, so the other fields
carry their values whether the required one is present or not.
### Numeric conversion
The parser produces `int64` for every integer and `float64` for every float.
@@ -129,18 +344,55 @@ The decoder converts to the destination type with explicit overflow checks:
| `uint`, `uint8`, `uint16`, `uint32`, `uint64` | the value must be non-negative and must not overflow the destination's own width, `uint` on a 32-bit platform included; `uint64` accepts any non-negative `int64` |
| `float32`, `float64` | copied verbatim, except that a finite value beyond the `float32` range is an overflow error rather than a silent infinity; an integer also coerces, so TOML `5` decodes into `5.0` |
| `bool`, `string` | exact kind match only, no coercion across kinds |
| `time.Time` | offset date-times only; no implicit conversion to or from the local variants |
| `time.Time`, `OffsetDateTime` | offset date-times only; no implicit conversion to or from the local variants |
A conversion that the rules do not allow produces an error wrapped with the
offending key or index, for example `p: interpres: integer 300 overflows uint8`.
### Numbers as literals
`NumbersAsLiterals(true)` decodes every integer and float into `Number`, a
string type that carries the literal the document wrote: `0x1f`, `1_000`,
`+1.0`, `inf`. The shape is validated as strictly as ever, so `01` and `1__0`
remain parse errors; only the evaluated value is replaced by the literal. A
round trip through the value tree and `Marshal` keeps the spelling, where the
default tree normalises `0x1f` to `31` and `+1.0` to `1.0`.
```go
var tree map[string]any
err := interpres.Unmarshal(data, &tree, interpres.NumbersAsLiterals(true))
lit := tree["rate"].(interpres.Number) // "1_000"
```
A destination of a concrete kind is unaffected: an `int64` field, a `float64`
field and a `time.Duration` field take the evaluated value they always took,
and a `Number` field takes the literal. `Number.Float64` and `Number.Int64`
evaluate the literal on demand, with an error for a float asked as an integer
and for a literal that is not a valid TOML number. `Marshal` writes a `Number`
as its bare literal and rejects one that is not a valid TOML number, whether it
stands alone or inside a value array.
### Date-time values
Offset date-times decode into `time.Time` and keep their offset. The local
variants decode into `LocalDateTime`, `LocalDate` and `LocalTime`, whose
embedded `time.Time` is normalised to UTC (midnight UTC for a local date, the
zero date for a local time). There is no implicit conversion between the offset
and local kinds; assigning one to the other is an error.
Offset date-times decode into `OffsetDateTime`, whose embedded `time.Time` is the
instant with the offset the document wrote; a destination of the plain
`time.Time` takes the same value, so a timestamp field does not have to name the
wrapper. The local variants decode into `LocalDateTime`, `LocalDate` and
`LocalTime`, whose embedded `time.Time` is normalised to UTC (midnight UTC for a
local date, the zero date for a local time). Every kind may omit the seconds as
of TOML 1.1 (`07:32`, `1979-05-27T07:32`); such a value carries a zero second,
and the encoder writes the seconds only when the value carries them, so a
document written without seconds comes back without them. There is no implicit
conversion between the offset and local kinds; assigning one to the other is an
error. The date-time types take a bare timestamp and never a quoted string, so a
document that writes a date-time with quotes does not decode into them, and
neither `encoding.TextUnmarshaler` nor the embedded `time.Time` changes that.
`LocalTimeLocation(loc)` lets a local date-time fill a plain
`time.Time` destination as well: the wall-clock value is carried in the
location given, relabelled rather than shifted, so `07:32` in the document is
`07:32` in the zone. Without the option the wrapper types are the only
destinations a local kind fills.
### Arrays of tables
@@ -149,6 +401,11 @@ destination is a slice, each element decodes into the slice's element type
(`[]struct` or `[]map[string]V`); a mismatch on one element surfaces as an
error wrapped with `[i]:` and the element index.
A value array also decodes into a fixed-size array, `[N]T`, the mirror of the
encoder's ability to encode one. The element count has to match: an array
whose length differs from `N` is an error, `interpres: cannot assign 2
elements to [3]int`, wrapped with the key path.
### Custom decoding: `Unmarshaler`
A type that wants full control of its decode implements:
@@ -160,7 +417,7 @@ type Unmarshaler interface {
```
`data` is whatever the parser produced for that key: `string`, `bool`, `int64`,
`float64`, `time.Time`, `LocalDateTime`, `LocalDate`, `LocalTime`, `[]any`, or
`float64`, `OffsetDateTime`, `LocalDateTime`, `LocalDate`, `LocalTime`, `[]any`, or
`map[string]any`. The method inspects the value and mutates its own receiver;
the decoder keeps whatever state the receiver stored.
@@ -171,15 +428,80 @@ automatically, and a nil pointer destination is allocated first. An error
returned from `UnmarshalTOML` halts the decode and propagates wrapped with the
key path, for example `addr: unmarshal: not a string`.
### Strict decoding
### Custom decoding: `UnmarshalerContext`
By default unknown keys are dropped silently. A `Decoder` built with
`DisallowUnknownFields` rejects them instead:
`UnmarshalerContext` is `Unmarshaler` with the decode's context handed in:
```go
err := interpres.NewDecoder().
DisallowUnknownFields().
Decode(data, &cfg)
type UnmarshalerContext interface {
UnmarshalTOMLContext(ctx context.Context, data any) error
}
```
A type that implements both gets `UnmarshalTOMLContext`, so a long custom
decode can abort on cancellation instead of running to completion. The
context a non-cancellable entry point carries is `context.Background`, never
nil.
### Custom decoding: `encoding.TextUnmarshaler`
A destination type that implements `encoding.TextUnmarshaler` receives a TOML
string as its text content, the rule `encoding/json` follows:
```go
func (ip *IP) UnmarshalText(text []byte) error
```
The decoder looks for the method on the destination and on its address, so a
pointer-receiver `UnmarshalText` is invoked on an addressable struct field, and
the elements of a slice destination are reached the same way. The text path
applies to TOML strings only: every other value kind keeps its own rule, so
`r = 1` does not reach a receiver that expects text. An error from
`UnmarshalText` halts the decode and propagates with the key path and the
prefix `unmarshal text:`, for example `addr: unmarshal text: not an address`.
[`UnmarshalTOML`](#custom-decoding-unmarshaler) wins over `UnmarshalText` when
a type implements both, and the four [date-time
types](#date-time-values) are excluded: a quoted string stays a string and
never becomes an `OffsetDateTime` or one of the local wrappers.
### Durations
TOML has no duration type, so `time.Duration` has a rule of its own. The
encoder writes the canonical Go form in a TOML string, `1h30m0s`, and the
decoder reads that string back with `time.ParseDuration`. A bare integer is
still the nanosecond count it has always been, so `from_int = 5400000000000`
and `from_text = "1h30m"` decode to the same duration. Text that
`time.ParseDuration` rejects, `d = "90"` among it, fails with
`interpres: invalid duration "90"`.
### Ordered tables
`OrderedMap` is a string-keyed table that remembers the order its keys were
set in, the shape a `map[string]any` cannot carry. Decoding into one fills it
in the order the document wrote the keys, and `Marshal` writes one back in
that order, where a map destination carries no order and a map source sorts
its keys. The type is a decode target on its own, in a struct field, and as
the element of an array of tables.
```go
var cfg OrderedMap
err := interpres.Unmarshal(data, &cfg)
out, err := interpres.Marshal(&cfg) // the keys come back in written order
```
The values are untyped, the shape the parser produces, so a nested table
inside an `OrderedMap` is a plain `map[string]any`; the order is kept at the
level the `OrderedMap` sits at. Inside a value array an `OrderedMap` renders
as an ordinary inline table, whose keys are sorted.
### Strict decoding
By default unknown keys are dropped silently. The `RejectUnknownFields`
option rejects them instead:
```go
err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
```
A typo such as `database_urls` then fails with
@@ -189,12 +511,31 @@ depth, including struct elements inside slices; map destinations accept every
key by nature. When several keys are unknown, the message names the smallest
one, so it does not depend on map iteration order.
### Direct decoding
For a struct destination whose type graph carries no untagged embedded map and
no custom decode hook, the decode parses straight into
the destination: the table skeleton is resolved against the struct schema while
the document scans, and no intermediate value tree is kept. Values still flow
through the ordinary assignment rules, so every conversion, hook and error the
[Decoding](#decoding) section states holds verbatim; the parity with the tree
path is pinned by a differential fuzz target that decodes every generated
document both ways and compares the results.
A document or destination the direct skeleton cannot model (an unknown table
under strictness it must sink, a hook that needs the whole parsed value, an
embedded map filler) falls back to the tree path and reruns, so the
observable behaviour is always the tree path's, exactly. Nothing changes for
`Parse`, `ParseMap` or the document API: the tree remains theirs.
### Cancellation
`ParseContext`, `UnmarshalContext` and `(*Decoder).DecodeContext` accept a
`ParseContext`, `UnmarshalContext` and `MarshalContext` accept a
`context.Context`. An already-cancelled context short-circuits with
`context.Canceled` before any work begins; afterwards the context is checked
every 64 top-level statements.
every 64 top-level statements, and inside a value too: an array, an inline
table and a multi-line string check every 64 elements or lines, so one huge
value cannot hold the parse past its cancellation.
### Flow
@@ -203,12 +544,12 @@ sequenceDiagram
participant Caller
participant Unmarshal as Unmarshal
participant Parser as parser
participant Decoder as decoder
participant Decode as decode
Caller->>Unmarshal: data, v
Unmarshal->>Parser: ParseContext(ctx, data)
Parser-->>Unmarshal: tree or *SyntaxError
Unmarshal->>Decoder: decode(tree, reflect value)
Decoder-->>Unmarshal: nil or wrapped field error
Unmarshal->>Parser: targeted parse straight into v
Parser-->>Unmarshal: nil, *SyntaxError, or fallback
Unmarshal->>Decode: tree rerun on fallback
Decode-->>Unmarshal: nil or wrapped field error
Unmarshal-->>Caller: error
```
@@ -216,9 +557,11 @@ sequenceDiagram
### Input constraints
`Marshal` and `(*Encoder).Marshal` accept a `struct`, a `map[string]V`, or a
non-nil pointer to one, where `V` is any value `Marshal` itself understands. A
different top-level value fails:
`Marshal` and `MarshalWrite` accept a `struct`, a `map[string]V`, or a
non-nil pointer to one, where `V` is any value `Marshal` itself understands.
An `OrderedMap` and a `Document` are accepted as themselves: the first in its
written key order, the second written back as it stands. A different
top-level value fails:
| Input | Error |
|---|---|
@@ -226,6 +569,11 @@ different top-level value fails:
| a nil `any` | `interpres: cannot marshal nil value` |
| a nil pointer | `interpres: cannot marshal nil pointer` |
The encoding walk carries a nesting limit of 10000 levels, the parser's own
figure: a value that nests deeper, which cyclic data always does, is rejected
with an error that names the limit and suggests the cycle, instead of running
the stack out.
### Field matching
Struct fields become TOML keys as follows:
@@ -243,22 +591,35 @@ nil map emits nothing.
### Tag options
The part of a `toml` tag after the first comma carries options. Both options
shape emission only; the decoder ignores them.
The part of a `toml` tag after the first comma carries options. They shape
emission only; the decoder ignores them, so a value that round-trips keeps
its key whether the table it came from was written inline or under a header.
- `omitzero` skips the field when its value is the zero value of its type. A
type with an `IsZero() bool` method (time.Time among them) decides through
that method, so a zero `time.Time` or an all-zero struct disappears from
the output.
- `omitempty` skips the field when it holds an empty collection: a nil or
empty slice or array, or a nil or empty map. Strings and other scalars are
not covered by `omitempty`; use `omitzero` for those.
- `omitempty` skips the field when it holds an empty value in the
encoding/json sense: an empty string, a zero number, `false`, a nil pointer
or interface, and a nil or empty slice, array or map. This is a change of
semantics against 1.x, where only collections were covered.
- `inline` forces a struct or map field to emit as `name = {…}`, the inline
table form, instead of a header section, whatever its size; a named
embedded struct tagged this way does the same. A field holding an array of
tables is an error under `inline`, because the inline form would re-parse
as a value array and change the value's Go type.
- `comment=text` carries a comment for the field, which
`EmitFieldComments(true)` prints above the field's line or
header, each line of a multi-line text with its own `# ` marker. Go doc
comments are not visible to reflection, so the tag is the channel that
carries the text; without the encoder option the tag is ignored.
```go
type Config struct {
Host string `toml:"host,omitzero"`
Started time.Time `toml:"started,omitzero"`
Tags []string `toml:"tags,omitempty"`
Retry Retry `toml:"retry,inline"`
}
```
@@ -274,7 +635,7 @@ them.
By default every table is emitted with its entries grouped by kind:
1. scalars (`string`, `int64`, `float64`, `bool`, `time.Time`,
`LocalDateTime`, `LocalDate`, `LocalTime`)
`OffsetDateTime`, `LocalDateTime`, `LocalDate`, `LocalTime`)
2. sub-tables (structs and `map[string]V` values)
3. arrays of tables (`[]struct` and `[]map[string]V`)
@@ -285,11 +646,11 @@ parsed as keys of the sub-table.
### Preserving declaration order
`GroupByKind(false)` on an `Encoder` walks the entries in declaration order
`Layout(LayoutKindDeclaration)` walks the entries in declaration order
instead, emitting each header immediately before its content:
```go
out, err := interpres.NewEncoder().GroupByKind(false).Marshal(cfg)
out, err := interpres.Marshal(cfg, interpres.Layout(interpres.LayoutKindDeclaration))
```
The output remains parseable, but a scalar declared after a sub-table lands
@@ -309,11 +670,20 @@ type Marshaler interface {
The returned value is encoded as if it had been passed in place of the
receiver, so it may be a scalar, a slice, an array of tables, or another
struct or map, including the `Marshaler` result of another type; the encoder
recurses. An error returned from `MarshalTOML` fails the marshal wrapped with
the key path, for example `interpres: server.port: bad timestamp`. A result
of `nil` with a nil error fails the same way with
`MarshalTOML returned a nil value`: nil has no TOML representation, so
dropping the field silently is not an option.
recurses, and the result is normalised like any other value, so a method may
return a plain `int` or a `time.Duration`.
An error returned from `MarshalTOML` fails the marshal wrapped with the key
path, for example `interpres: server.port: bad timestamp`. A result of `nil` with
a nil error fails the same way with `MarshalTOML returned a nil value`: nil has
no TOML representation, so dropping the field silently is not an option.
The method is reached for every value the walk meets, array elements included:
an element that renders itself as a table keeps the `[[header]]` form, one that
renders itself as a scalar turns the array into a value array, and the method
runs once per element. It is looked up on the value and on its address, so a
pointer-receiver method is called for a field or an element, exactly as
`MarshalText` is.
```go
type Port int
@@ -323,6 +693,27 @@ func (p Port) MarshalTOML() (any, error) {
}
```
### Custom encoding: `encoding.TextMarshaler`
A type that implements `encoding.TextMarshaler` is encoded as a TOML string
holding the text the method returns, which is the rule `encoding/json` follows:
```go
func (ip IP) MarshalText() ([]byte, error)
```
The encoder looks for the method on the value and on its address, so a
pointer-receiver `MarshalText` is found on a struct field of an addressable
value (pass a pointer to `Marshal`) and always on a slice element. `net.IP`,
`netip.Addr` and user types follow this rule, and a struct that implements the
interface becomes a string rather than a table. `MarshalTOML` wins when a type
implements both, the four [date-time types](#date-time-values) keep their bare
timestamp form, and text that is not valid UTF-8 is an error rather than a
replacement character.
A duration carries no text method of its own; see [Durations](#durations) for
its rule.
### Arrays
An array whose every element is a table (`[]struct`, `[]map[string]V`, after
@@ -344,30 +735,71 @@ across a round-trip.
A nil slice is always omitted. An empty (length 0) array of tables is always
omitted, because TOML forbids an empty `[[a]]`. Other empty arrays emit as
`key = []` by default; `OmitEmptyArrays()` skips them as well, so
`key = []` by default; `OmitEmptyArrays(true)` skips them as well, so
`[]string{}` is treated like a nil slice.
### Long strings
By default every string is emitted as a basic `"..."` string with the escapes
TOML requires, and a string containing a newline is emitted as an escaped
multi-line basic string. `UseLiteralMultiline(threshold)` switches strings that
contain a newline and are at least `threshold` bytes long to the literal
`'''...'''` form, which carries the newlines verbatim:
TOML requires, a newline among them as `\n`. `LiteralMultiline(threshold)`
switches strings that contain a newline and are at least `threshold` bytes long
to the literal `'''...'''` form, which carries the newlines verbatim:
```go
out, err := interpres.NewEncoder().UseLiteralMultiline(80).Marshal(cfg)
out, err := interpres.Marshal(cfg, interpres.LiteralMultiline(80))
```
Single-line strings keep the basic form regardless of the threshold, and a
threshold of `0` or less disables the option. A string the literal form cannot
carry verbatim (an embedded run of three single quotes, a control character
other than tab, or a carriage return outside a CRLF pair) also keeps the basic
form, so the output always re-parses to the same value.
other than tab or newline, or a carriage return outside a CRLF pair) also keeps
the basic form, so the output always re-parses to the same value.
### Inline tables
A table element of a value array, and a sub-table inlined by
[`InlineTables`](#compact-documents), is written as one `{a = 1, b = 2}` line
while it fits. An inline table that would pass the hundredth column carries
newlines and a trailing comma instead, which TOML 1.1 allows:
```toml
arr = [1, {
n = 1,
name = "a value long enough to push this line well past the one hundred column limit",
}]
```
The closing brace and the entries are indented one tab per nesting level, a
nested table is measured on its own line, and the output re-parses to the same
value either way.
### Compact documents
`InlineTables(threshold)` writes a sub-table as an inline table when its
single-line rendering is at most `threshold` bytes, and as a table header
section when it is longer. A document of small tables therefore grows shorter:
```go
out, err := interpres.Marshal(cfg, interpres.InlineTables(60))
```
With `60` and a table of three short entries, the same value is written
```toml
server = {host = "127.0.0.1", port = 9090, tls = {on = false}}
```
instead of three lines under a `[server]` header and a `[server.tls]` section.
A nested sub-table takes part in the same way, and the whole option is off at
`0` or less. Two limits are deliberate. An array of tables keeps the `[[a]]`
header form, because its inline form re-parses as a value array and would change
the value's Go type. And because an inlined table is a value line, every one of
them precedes the first header of its document, so a table inlined next to a
header is not read back as part of that header's section.
### Cancellation
`MarshalContext` and `(*Encoder).MarshalContext` accept a `context.Context`. The
`MarshalContext` accepts a `context.Context`. The
context is checked before any work and every 64 fields during the reflection
walk.
@@ -379,6 +811,8 @@ The output is not byte-identical to any document that produced the value:
- map keys are emitted in sorted order
- the choice between `[table]` headers and inline tables is not preserved
- strings use the basic quoted form unless the literal option above applies
- a date-time drops its zero seconds and the trailing zeros of its fraction, so
`07:32:00` is written `07:32`; both are the same value
- floats always carry a `.` or an exponent, so a float `1` is emitted as `1.0`
and stays distinguishable from the integer `1` across a round-trip; negative
zero is normalised to `0.0`
@@ -402,71 +836,132 @@ sequenceDiagram
Marshal-->>Caller: bytes, error
```
## Coming from encoding/json and encoding/json/v2
The API follows the shapes encoding/json made familiar and the option style
encoding/json/v2 made current, with the differences TOML asks for:
| encoding/json or encoding/json/v2 | interpres | Notes |
|---|---|---|
| `json.Unmarshal(data, v)` | `Unmarshal(data, v)` | the same shape; the value mapping is TOML's |
| `json.Marshal(v)` | `Marshal(v)` | the same shape; the output is TOML 1.1 |
| `json.MarshalAppend(buf, v)` | `MarshalAppend(buf, v)` | the same shape, options included |
| `json.MarshalWrite(w, v)` | `MarshalWrite(w, v)` | the same shape, options included |
| `json.UnmarshalRead(r, v)` | `UnmarshalRead(r, v)` | the same shape, options included |
| `json/v2 RejectUnknownMembers` | `RejectUnknownFields(true)` | the same effect under TOML vocabulary |
| `(*json.Decoder).DisallowUnknownFields` | `RejectUnknownFields(true)` | the variadic option replaces the stateful decoder |
| `json.Number`, `StringifyNumbers` | `Number`, `NumbersAsLiterals(true)` | the TOML literal carries its radix and separators, so `0x1f` stays `0x1f` |
| `json/v2 MarshalOptions` fields | `MarshalOption` values | `Layout`, `OmitEmptyArrays`, `LiteralMultiline`, `InlineTables`, `EmitFieldComments` |
| `json/v2 JoinOptions` | listing | options compose by listing them in the call |
| `json.MarshalIndent` | none | TOML is the presentation format; the `-json` mode of interpres-decode prints plain JSON |
| tag `json:"name,omitempty"` | tag `toml:"name,omitempty"` | the empty-value rules match encoding/json as of 2.0 |
| tag `json:"name,omitzero"` | tag `toml:"name,omitzero"` | the same, `IsZero()` honoured |
| tag `json:"name,inline"` (v2) | tag `toml:"name,inline"` | forces the inline table form on encode |
| `json/v2 Marshalers` | `Marshaler` (`MarshalTOML`) | the TOML method returns a value the encoder renders, not bytes |
| `json/v2 Unmarshalers` | `Unmarshaler` (`UnmarshalTOML`) | the data arrives decoded, not as bytes |
| `encoding.TextMarshaler`, `TextUnmarshaler` | honoured, the same | a type that renders itself as text becomes a TOML string, both ways |
| `*json.UnmarshalTypeError` | `*DecodeError` | the path is segments with a `String()` renderer, not a dotted string |
| `*json.SyntaxError` | `*SyntaxError` | the TOML error adds the byte `Offset` and the `Column` to the line |
| context support | `*Context` variants of the parse, decode and marshal entries | encoding/json has none |
## Types
### `type SyntaxError struct{ Line int; Msg string }`
### `type SyntaxError struct{ Line, Offset, Column int; Msg string }`
Describes a malformed TOML document; `Line` is 1-based and `Error()` renders as
`interpres: line N: msg`. Read the structured fields with a type assertion or
Describes a document the parser rejected: the 1-based `Line` at which it gave
up, the `Offset` in bytes the scan stopped at, the 1-based `Column` on that
line, and `Error()` rendering as `interpres: line N: msg`. A malformed document
is the usual cause; the nesting limit and an input that is not valid UTF-8
report through the same type, with the UTF-8 message naming the offset of the
first invalid byte. Read the structured fields with a type assertion or
`errors.AsType`:
```go
if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
fmt.Println(se.Line, se.Msg)
fmt.Println(se.Line, se.Offset, se.Column, se.Msg)
fmt.Println(se.SourceLine(data)) // the line, with a caret under Offset
}
```
### `type DecodeError struct{ Path []string; Err error }`
`SourceLine(src)` renders the source line the error points at from `src`,
followed by a caret line marking the column, for messages the reader sees
under the input.
### `type DecodeError struct{ Path Path; Err error }`
Wraps a decoding failure with the key path at which it happened. `Path` lists
one segment per level from the document root, the outermost key first: a key
contributes its name, an array element its bracketed index, so the path of the
`weight` field in the first item reads `["items", "[0]", "weight"]`. The
rendered message is unchanged by the type; read the fields instead of parsing
the message:
`weight` field in the first item reads `["items", "[0]", "weight"]` and its
`String()` renders `items[0].weight`. Read the fields instead of parsing the
message:
```go
if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
fmt.Println(de.Path, de.Err)
fmt.Println(de.Path.String(), de.Err)
}
```
### `type EncodeError struct{ Path string; Err error }`
### `type EncodeError struct{ Path Path; Err error }`
Wraps an encoding failure with the key path of the value that failed, in the
document's own notation: `server.ports[2]`. Read it with `errors.AsType` the
same way.
Wraps an encoding failure with the key path of the value that failed, the
same `Path` type the decode error carries, so `server.ports[2]` reads the
same on both sides. Read it with `errors.AsType` the same way.
### `type Decoder`
### `type Path []string`
Configurable strictness for decoding, constructed with `NewDecoder`. Set up
with `DisallowUnknownFields`, then call `Decode` or `DecodeContext` any number
of times. A configured `Decoder` holds no per-call state and is safe for
concurrent use.
The path both error wrappers carry, one segment per level from the document
root. `String()` renders the TOML notation: keys join with dots, an index
attaches to the previous segment in brackets, `items[0].weight`.
### `type Encoder`
### Options
Configurable emission policy, constructed with `NewEncoder`. The option state
is private; set it with the chainable methods, each of which returns the
encoder:
The decode and encode entries take variadic options, the shape
encoding/json/v2 uses for its own. Each option is a stateless function value
over the private settings of one call; they compose by listing in the call,
and there is no stateful Decoder or Encoder to share or guard.
| Method | Default | Effect |
Decode options:
| Option | Default | Effect |
|---|---|---|
| `GroupByKind(v bool)` | `true` | group entries as scalars, then sub-tables, then arrays of tables; `false` preserves declaration order |
| `OmitEmptyArrays()` | off | skip `key = []` for empty scalar arrays |
| `UseLiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
| `RejectUnknownFields(v bool)` | off | a key with no matching struct field is an error |
| `NumbersAsLiterals(v bool)` | off | integers and floats decode into `Number`, which carries the literal; see [Numbers as literals](#numbers-as-literals) |
| `MaxNestingDepth(depth int)` | `10000` | bound how deeply arrays and inline tables may nest |
| `MaxInputSize(size int)` | no limit | bound the size of the document, in bytes |
| `LocalTimeLocation(loc)` | nil | the zone a local date-time is carried in when it decodes into a `time.Time` |
Encode options:
| Option | Default | Effect |
|---|---|---|
| `Layout(kind LayoutKind)` | `LayoutKindGrouped` | group entries as scalars, then sub-tables, then arrays of tables; `LayoutKindDeclaration` preserves declaration order |
| `OmitEmptyArrays(v bool)` | off | skip `key = []` for empty scalar arrays |
| `LiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
| `EmitFieldComments(v bool)` | off | print the `comment=` tag option of a field above its line or header |
```go
out, err := interpres.NewEncoder().
GroupByKind(false).
OmitEmptyArrays().
UseLiteralMultiline(80).
MarshalContext(ctx, cfg)
out, err := interpres.MarshalContext(ctx, cfg,
interpres.Layout(interpres.LayoutKindDeclaration),
interpres.OmitEmptyArrays(true),
interpres.LiteralMultiline(80),
interpres.InlineTables(60))
```
A configured `Encoder` holds no per-call state; each `Marshal` or
`MarshalContext` call copies the options and is safe for concurrent use, as
long as no setter races with a call.
The nesting limit protects the stack, because the parser is a recursive
descent: a deeper document is rejected with a `SyntaxError` naming the limit
rather than running the stack out. `Parse` and `ParseContext` carry that same
default but take no options. The size limit is off by default, because the
caller already holds the bytes and the size is therefore a policy, not a
protection the library can impose on its own.
### `type Document`, `type Table`, `type Entry`
See [Documents](#documents). A `Document` is what `Parse` returns, and
`Marshal` writes it back: the keys in written order, the comments in place,
the inline tables inline. `UnmarshalDocument(doc, v)` decodes it without
parsing again.
### `type Marshaler interface{ MarshalTOML() (any, error) }`
@@ -474,29 +969,46 @@ See [Custom encoding](#custom-encoding-marshaler).
### `type Unmarshaler interface{ UnmarshalTOML(data any) error }`
See [Custom decoding](#custom-decoding-unmarshaler).
See [Custom decoding](#custom-decoding-unmarshaler). `UnmarshalerContext`
carries the decode's context through `UnmarshalTOMLContext(ctx, data)` and
wins when a type implements both.
### `type Number string`
The literal a number was written with, what `NumbersAsLiterals` decodes into and what
`Marshal` writes back as it is. See
[Numbers as literals](#numbers-as-literals).
### `type OrderedMap`
The string-keyed table that keeps its key order on both the encode and the
decode side. See [Ordered tables](#ordered-tables).
### Date-time wrappers
```go
type OffsetDateTime struct{ time.Time } // 1979-05-27T07:32:00Z
type LocalDateTime struct{ time.Time } // 1979-05-27T07:32:00
type LocalDate struct{ time.Time } // 1979-05-27
type LocalTime struct{ time.Time } // 07:32:00.999999
```
Each carries a `String()` method returning the TOML-canonical rendering, with
the fractional second zero-padded to nanosecond precision when present. The
types are produced by `Parse` and accepted by `Marshal`.
Each carries a `String()` method returning the TOML-canonical rendering: the
seconds appear only when the value carries them, and a fractional second drops
its trailing zeros, so `07:32:00` renders as `07:32` and a half second as
`00.5`. The types are produced by `Parse` and accepted by `Marshal`, which
writes them through `String()`.
## Errors
The entry points return:
- `*SyntaxError` for a malformed document, with the 1-based line
- `*SyntaxError` for a malformed document, with the 1-based line; the nesting
limit reports through it as well
- `*DecodeError` for a decoding failure, with the key path in `Path`
- `*EncodeError` for an encoding failure, with the key path in `Path`
- a plain error for the rest: a non-pointer decode target, a cancelled
context, a key that is not valid UTF-8
context, a key that is not valid UTF-8, an input over the size limit
Decode and encode failures carry the key path or element index in the typed
wrappers above, so `errors.Is` and `errors.AsType` see through them and the
+53 -30
View File
@@ -5,16 +5,17 @@ source tree; nothing is aspirational.
## Overview
interpres is one public library package, one command, and one example. The
library implements the whole of TOML 1.0 and 1.1, decoding and encoding, in the
standard library alone; the command wraps the parser for the toml-test
compliance harness, against which it stands at 214 valid and 467 invalid cases
with zero failures; the example demonstrates the API.
interpres is one public library package, one command, and two examples. The
library implements the whole of TOML 1.1, decoding and encoding, in the
standard library alone; the command wraps the parser and the encoder for the
toml-test compliance harness, against which it stands at 214 valid, 467 invalid
and 214 encoder cases with zero failures; the examples demonstrate the API: one the document round trip, one the statement iterator.
```mermaid
flowchart TD
CLI[cmd/interpres-decode<br/>toml-test adapter] --> API
EX[examples/basic<br/>usage demo] --> API
EX2[examples/statements<br/>statement iterator demo] --> API
subgraph Lib [package interpres]
API[interpres.go<br/>public API and types]
API --> P[parser.go<br/>recursive-descent parser]
@@ -35,28 +36,40 @@ strict validation.
| Path | Responsibility |
|---|---|
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, `Decoder`, `Encoder`, `Marshaler`, `Unmarshaler`, `SyntaxError`, the local date-time types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter. Reads TOML on stdin, writes tagged JSON on stdout. Owns no parsing logic. |
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, the option constructors, `Marshaler`, `Unmarshaler`, `SyntaxError`, the error and option types); everything below it is unexported. |
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `--encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
| `examples/basic` | A runnable tour of the API. Documentation in executable form, not part of the library. |
| `examples/statements` | The `Statements` iterator over a document, the shape a configuration tool reads. Documentation in executable form. |
Inside the library package, one file owns one concern:
| File | Responsibility |
|---|---|
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 and 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree, records the nodes a [Document](API.md#documents) is built from, and enforces the structural rules of TOML 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
| `document.go` | The parsed-document types: `Document`, `Table` and `Entry`, which carry the key order, whether a table was written inline, and the comments. The values they expose are the parser's own tree, not a copy. |
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `datetime.go` | The four date-time types (`OffsetDateTime` and the three local wrappers) and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `orderedmap.go` | `OrderedMap`, the table that keeps its key order, and the node index the decoder reads the written order from. |
| `target.go` | The targeted parse: the struct skeleton resolved against the document while it scans, no intermediate tree. Falls back to the tree path for every shape it does not model. |
| `docwrite.go` | The write side of the document pipeline: `UnmarshalDocument` and the writer that renders a `Document` back with its order and comments. |
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
| `encode.go` | The reverse walk: builds an intermediate `tomlDoc` per table (which is what preserves declaration order and enables the group-by-kind partition) and then emits it as TOML. |
The boundary that matters: `parser.go` produces only untyped trees
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); `decode.go` and
`encode.go` are the only files that touch `reflect`; the command never touches
either, it consumes `Parse` alone.
(`map[string]any`, `[]any`, `[]map[string]any`, scalars); the reflection work
lives in `decode.go`, `encode.go`, `target.go` and `orderedmap.go`; the command
consumes `ParseMap`, `Parse` and `Marshal`, and owns no parsing or emission
logic of its own.
## Data flow
Decoding is parse, then one reflection walk. `SyntaxError` values are produced
Decoding has two paths. The direct one parses straight into a struct
destination: `target.go` resolves the table skeleton against the struct
schema while the document scans, and values assign through the ordinary
decoder rules, so no intermediate tree exists; that is the hot path every
`Unmarshal` into a struct takes. A document or destination the direct
skeleton cannot model falls back to the tree path: parse the whole document,
then one reflection walk over the tree. `SyntaxError` values are produced
inside `parser.go` and returned as-is; conversion errors are produced inside
`decode.go` and wrapped with the key path as they unwind.
@@ -65,10 +78,13 @@ sequenceDiagram
participant Caller
participant API as interpres.go
participant P as parser.go
participant T as target.go
participant D as decode.go
Caller->>API: Unmarshal(data, v)
API->>P: ParseContext(ctx, data)
P->>P: number and datetime atoms
API->>T: targeted parse into the struct
T->>P: scanner, grammar, atoms
T-->>API: result, error or fallback
API->>P: on fallback, ParseContext(ctx, data)
P-->>API: map tree or *SyntaxError
API->>D: decode(tree, reflect value)
D-->>API: nil or wrapped field error
@@ -76,7 +92,7 @@ sequenceDiagram
```
Encoding walks the other way. `encode.go` first builds a `tomlDoc` from the
value, then emits it; the two phases are why `GroupByKind` can reorder entries
value, then emits it; the two phases are why `Layout` can reorder entries
without a second reflection pass, and why cancellation is checked during both.
```mermaid
@@ -95,21 +111,28 @@ sequenceDiagram
## State and lifetime
- The exported `Decoder` and `Encoder` hold configuration only. Every
`Decode`, `DecodeContext`, `Marshal` and `MarshalContext` call allocates its
own unexported worker, so a configured type is safe for concurrent use; the
setter methods are not, and must finish before the value is shared.
- The parser is allocated per `ParseContext` call; nothing is cached between
documents.
- The one piece of shared state is the struct-schema cache in `decode.go`: a
`sync.Map` keyed by `reflect.Type`, holding the flattened field layout the
decoder and the encoder both consult. A schema is immutable once published,
so concurrent callers only race to build an identical value, the same
trade-off `encoding/json`'s field cache makes. The cache grows with the
number of distinct struct types, never with document size.
- The option values are stateless: every `Unmarshal`, `Marshal` and their
variants apply their own options into a per-call unexported worker, so the
entries are safe for concurrent use.
- The parser is allocated per `ParseContext` call; the parser itself caches
nothing between documents.
- The shared state is a set of caches and pools whose entries are immutable
once published, each growing with the number of distinct types rather than
with document size: the struct-schema cache in `decode.go` (a `sync.Map`
keyed on `reflect.Type`, holding the flattened field layout the decoder and
the encoder both consult), the per-type interface flag caches in `decode.go`
and `encode.go` (recording where `Marshaler`, `Unmarshaler` and the text
interfaces can be found, so a walk builds an interface value only where the
assertion can succeed), each fronted by a monomorphic hint holding the type
resolved last, and the encoder's output-buffer pool in `encode.go`
(`sync.Pool`, buffers returned to it only within a 1 MiB retention cap). A
published schema or flag set never mutates, so concurrent callers only race
to build an identical value, the same trade-off `encoding/json`'s field
cache makes.
- The date-time wrappers are values, not pointers, and are immutable in use.
- Nothing in the library starts goroutines; apart from the schema cache above,
which never mutates a published entry, there is no shared mutable state.
- Nothing in the library starts goroutines; apart from the caches and the pool
above, which never mutate a published entry, there is no shared mutable
state.
## Dependencies
+48
View File
@@ -0,0 +1,48 @@
# Benchmarking
How the performance numbers attached to this project are measured, so that a
number in a changelog entry or a release note can be reproduced and trusted.
## The suite
The benchmarks live in `bench_test.go`, next to the code they measure:
| Benchmark | What it measures |
|---|---|
| `BenchmarkParse` | `ParseMap` over a representative configuration document |
| `BenchmarkMarshal` | `Marshal` of the tree `ParseMap` produced from the same document |
| `BenchmarkStrictDecode` | `Unmarshal` into a struct under `RejectUnknownFields` (the targeted parse) |
| `BenchmarkParseLong` | `ParseMap` over a generated document with about 2000 array-of-tables entries |
| `BenchmarkStrictDecodeLong` | `Unmarshal` into a typed document under `RejectUnknownFields`, over the same long document |
| `BenchmarkMarshalLong` | `Marshal` of the tree `ParseMap` produced from the long document |
| `BenchmarkStrictDecodeTree` | the tree-path reference decode of the representative document: parse, then the reflection walk |
| `BenchmarkStrictDecodeTreeLong` | the tree-path reference decode of the long document, the A/B baseline of the targeted parse |
## Running
```sh
just bench
```
The recipe runs the suite with `-benchmem -count=5`. Every benchmark uses
`b.Loop`, so setup runs outside the timed region, and `ReportAllocs` records
allocations per operation. The parse and marshal benchmarks set `SetBytes`, so
their results read as input bytes per second.
## Method
- An idle machine only: a loaded box times whatever else is running, and the
fastest sample can land on the wrong function.
- An A/B comparison runs both variants inside one process, in one binary;
separate processes of identical binaries differ by more than the effect
being measured.
- The five counts are compared through their medians, allocations and bytes
per operation alongside the times. Differences within 1 to 2 percent are
noise; only a difference beyond that is a result.
- When timing is hopeless, the allocation and byte counts are the result.
## Reports
The repository stores no benchmark reports. A performance claim in
`CHANGELOG.md` is measured with the method above on the change that makes it,
and the number travels with the claim.
+132 -22
View File
@@ -1,45 +1,87 @@
# Command line
The reference below is taken from the program itself. `interpres-decode` is
the toml-test harness adapter, and it also validates documents. Install it
with Go itself, no release assets involved:
the toml-test harness adapter in both directions, decoding TOML into tagged
JSON and encoding tagged JSON back into TOML, and it also validates documents.
The same reference ships as the manual page `man/interpres-decode.1`.
Install it with Go itself, no release assets involved:
```sh
go install sourcedock.dev/petrbalvin/interpres/cmd/interpres-decode@latest
go install sourcedock.dev/petrbalvin/interpres/v2/cmd/interpres-decode@latest
```
## Synopsis
```sh
interpres-decode [flags]
interpres-decode -validate [file ...]
interpres-decode --encode
interpres-decode --validate [file ...]
interpres-decode --validate [directory ...]
interpres-decode --json
interpres-decode --struct
interpres-decode --schema TYPE file.go
interpres-decode --version
```
Without `-validate` the program is the toml-test adapter: it takes no
arguments, reads one TOML document from stdin, and writes the toml-test
tagged-JSON form to stdout. Build it locally with `just build`, which
compiles it into `bin/interpres-decode`, or run it straight from the module
directory with `just run`.
Without `--validate`, `--encode`, `--json`, `--struct` or `--schema` the
program is the decoding half of the toml-test adapter: it takes no arguments,
reads one TOML document from stdin, and writes the toml-test tagged-JSON form
to stdout. Build it locally with `just build`, which compiles it into
`bin/interpres-decode`, or run it straight from the module directory with
`just run`.
With `-validate` the program parses each named file instead, or stdin when no
With `--encode` the direction is reversed: the program reads a tagged-JSON
description from stdin and writes the TOML document it describes to stdout,
which is the shape toml-test expects of an encoder command. It takes no
arguments either, and the mode flags cannot be combined.
With `--validate` the program parses each named file instead, or stdin when no
file is named, and prints one line per invalid document to stderr. It is
quiet on valid documents, which is the shape a CI step wants. The `-` name
means stdin.
means stdin. A named directory is walked for `.toml` files, every one of them
validated, and the walk closes with a summary on stderr naming how many
documents were checked and how many were invalid.
With `--json` the decoding half prints plain indented JSON instead of the
tagged form, the shape for people and diffs: the values keep their types as
JSON sees them, and the date-time wrappers print in their TOML form. The flag
shapes the decoding output only, so it is rejected together with the mode
flags.
With `--struct` the program reads a TOML document from stdin and prints a Go
struct definition shaped like it: one field per key in written order, nested
tables as nested struct types, and an array of tables as a slice. The
printed type compiles and decodes the document it came from.
With `--schema` the program reads a Go source file and writes a TOML template
for the named struct type: one key per exported field, the `comment=` tag
option printed as a comment above it, and the `default=` option as the value
where one is set. It is the inverse of `--struct`, for config-driven
applications that generate their example configuration from the type.
`--version` prints the binary's version and exits. The release pipeline builds
at the tag, so a released binary prints its own tag; a build from a working
tree prints `(devel)`.
## Flags
| Flag | Effect |
|---|---|
| `-validate` | validate the documents instead of emitting tagged JSON |
| `-h` | print the usage |
| `--validate` | validate the documents instead of emitting tagged JSON; directories are walked for `.toml` files |
| `--encode` | read tagged JSON from stdin and write TOML instead |
| `--json` | with the default mode, print plain indented JSON instead of tagged JSON |
| `--struct` | infer a Go struct definition from the document on stdin and print it |
| `--schema TYPE` | write a TOML template for the struct type TYPE from the Go source file named as the first argument |
| `--version` | print the version and exit |
| `--help` | print the usage |
## Exit codes
| Code | Meaning |
|---|---|
| `0` | adapter: the document parsed and the tagged JSON was written; validate: every document parsed |
| `1` | adapter: parse error; validate: at least one document is invalid |
| `2` | a usage error, a read failure, or a value with no tagged representation |
| `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed; schema, struct, version: the output was written |
| `1` | adapter: parse error; validate: at least one document is invalid; struct: the document on stdin failed to parse |
| `2` | a usage error, a read or write failure, malformed tagged JSON, or a value with no TOML representation |
## Wire format
@@ -66,6 +108,14 @@ wrapped in an object with a `type` and a `value`:
| local date | `date-local` | `1979-05-27` |
| local time | `time-local` | `07:32:00.999999` |
The `--encode` mode reads exactly this form back. Two properties of it are
worth knowing. A float whose value has no fraction and no exponent is written
as a bare integer string, `{"type": "float", "value": "1"}`, so there the tag
decides the type and not the literal. And the form cannot tell an array of
tables from a value array of inline tables, so the adapter writes the header
form, `[[a]]`, for an array whose every element is a JSON object; a mixed
array keeps the value form.
## Examples
Echo a small document through the adapter:
@@ -78,22 +128,80 @@ port = 9090
' | ./bin/interpres-decode
```
The output is the equivalent value tree as one JSON object. Validate the
TOML files of another repository in CI:
The output is the equivalent value tree as one JSON object. Turn a description
back into TOML with `--encode`:
```sh
interpres-decode -validate config.toml deploy/example.toml
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode --encode
```
```toml
title = "hello"
```
Validate the TOML files of another repository in CI:
```sh
interpres-decode --validate config.toml deploy/example.toml
```
An invalid document reports the file and the library's line number:
```sh
$ interpres-decode -validate bad.toml
bad.toml: interpres: line 1: expected a value
$ interpres-decode --validate bad.toml
interpres-decode: bad.toml: interpres: line 1: expected a value
$ echo $?
1
```
Sweep a whole directory tree of configuration, with the summary the walk
closes on:
```sh
$ interpres-decode --validate configs/
interpres-decode: configs/old.toml: interpres: line 3: duplicate key "port"
checked 14 documents, 1 invalid
$ echo $?
1
```
See the document a `--struct` template would decode:
```sh
echo 'host = "db"
port = 5432
' | ./bin/interpres-decode --struct
```
```go
// Generated by interpres-decode --struct; decode with
// sourcedock.dev/petrbalvin/interpres/v2.
type inferred struct {
Host string `toml:"host"`
Port int64 `toml:"port"`
}
```
Write the template back from the type, comments and defaults included, where
the Go source declares fields tagged
`toml:"host,comment=The host to dial,default=example.org"`:
```sh
./bin/interpres-decode --schema Config config.go
```
```toml
# The host to dial
host = "example.org"
```
Print the binary's version:
```sh
$ ./bin/interpres-decode --version
interpres-decode v2.0.0
```
Run the official compliance suite against the adapter:
```sh
@@ -101,5 +209,7 @@ just toml-test
```
That recipe needs the `toml-test` binary on `PATH`, installed with
`go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0`. The full
`go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0`. It runs
the suite in both directions: the decoder against the valid and invalid
corpora, and the encoder against the tagged JSON of the valid one. The full
reference for the library itself is [API.md](API.md).
+14 -7
View File
@@ -41,11 +41,14 @@ prints the same list.
| `just run` | `go run ./cmd/interpres-decode`, reads TOML from stdin |
| `just dev` | the same run, for iterating |
| `just example` | `go run ./examples/basic`, the usage tour |
| `just toml-test` | builds the adapter and runs the official toml-test compliance suite against it |
| `just toml-test` | builds the adapter and runs the official toml-test compliance suite against it, decoder and encoder |
| `just coverage-html` | `just test`, then `go tool cover -html` into `coverage.html` |
| `just install` | builds, then copies the binary into `~/.local/bin` (`BINDIR` overrides) |
| `just uninstall` | removes the installed binary |
| `just clean` | removes `bin/` and `coverage.out` |
| `just cross` | cross-compile smoke of the library and the command for arm64, loong64, riscv64 and the browser and edge runtimes; a hand-run convenience, not a gate |
| `just release-check X.Y.Z` | the release pre-flight: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release |
| `just docs-drift` | compares the toml-test counts the documentation quotes with a live suite run |
## Running a single test
@@ -77,7 +80,9 @@ just bench
```
Benchmark on an idle machine, and compare only runs made in one process against
each other. The recipe sweeps `./...` five times with `-benchmem`.
each other. The recipe sweeps `./...` five times with `-benchmem`. The binding
measurement method, and what counts as a result, is in
[docs/BENCHMARKING.md](BENCHMARKING.md).
## Debugging the build
@@ -97,12 +102,14 @@ pipeline.
|---|---|---|
| `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite |
| `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally |
| `release.yml` | a `v*` tag | the same gates plus the race detector, then the Gitea release from the CHANGELOG section |
| `fuzz.yml` | `workflow_dispatch`, by hand | a 30 second fuzz smoke per target over the seeds and the gathered corpus |
| `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut |
## Releases
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The
tag drives the release workflow: it validates the tag, runs the full gate set
including the race detector, extracts the matching `## [X.Y.Z]` section from
`CHANGELOG.md`, and publishes the release with that section as its body. A
library ships no binaries, so the release carries the notes and nothing else.
tag drives the release workflow: it validates the tag, runs the static gates
and the test suite with the coverage floor, extracts the matching `## [X.Y.Z]`
section from `CHANGELOG.md`, and publishes the release with that section as its
body. A library ships no binaries, so the release carries the notes and nothing
else.
+448
View File
@@ -0,0 +1,448 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"maps"
"slices"
)
// A Document is a parsed TOML document: the values, plus what the map shape
// cannot carry, which is the order the keys were written in, whether a table
// was written inline or under a header, and the comments.
//
// The values are the tree ParseMap returns, shared rather than copied, so a
// value read from a Document and from that map is the same value. A comment
// belongs to the statement it precedes: the lines above a key belong to the
// key, the lines above a header belong to the header's table, and a comment
// block after the last statement belongs to the Document.
//
// Comments inside array and inline-table values are not carried yet; the
// parser skips them as it always has.
type Document struct {
root *Table
footer []string
}
// Root returns the document's root table. A nil document has no root.
func (d *Document) Root() *Table {
if d == nil {
return nil
}
return d.root
}
// Map returns the value tree, the shape ParseMap gives. It is the tree the
// document was parsed into, not a copy. A nil document or one with no root
// holds no values.
func (d *Document) Map() map[string]any {
if d == nil || d.root == nil {
return nil
}
return d.root.values
}
// Footer returns the comment lines that follow the last statement, and every
// line of a document that holds no statement at all.
func (d *Document) Footer() []string {
if d == nil {
return nil
}
return d.footer
}
// SetFooter replaces those lines.
func (d *Document) SetFooter(lines []string) {
if d == nil {
return
}
d.footer = lines
}
// The document-level convenience forms of the Table edit API; they act on
// the root table.
// Get returns the root table's entry for key, and whether the document has
// one. See Table.Get.
func (d *Document) Get(key string) (*Entry, bool) { return d.Root().Get(key) }
// GetString returns the string the key holds, and whether it holds one.
func (d *Document) GetString(key string) (string, bool) { return d.Root().GetString(key) }
// GetInt returns the integer the key holds, and whether it holds one.
func (d *Document) GetInt(key string) (int64, bool) { return d.Root().GetInt(key) }
// GetFloat returns the float the key holds, and whether it holds one.
func (d *Document) GetFloat(key string) (float64, bool) { return d.Root().GetFloat(key) }
// GetBool returns the boolean the key holds, and whether it holds one.
func (d *Document) GetBool(key string) (bool, bool) { return d.Root().GetBool(key) }
// GetArray returns the value array the key holds, and whether it holds one.
func (d *Document) GetArray(key string) ([]any, bool) { return d.Root().GetArray(key) }
// GetTable returns the node of the table the key holds, and whether it holds
// one.
func (d *Document) GetTable(key string) (*Table, bool) { return d.Root().GetTable(key) }
// Set stores value under the key in the root table. See Table.Set.
func (d *Document) Set(key string, value any) { d.Root().Set(key, value) }
// Delete removes the key from the root table. See Table.Delete.
func (d *Document) Delete(key string) { d.Root().Delete(key) }
// A Table is one TOML table: its values, its keys in written order, and the
// comments around the header or the key that introduced it.
type Table struct {
values map[string]any
entries []*Entry
index map[string]*Entry
// inline records that the table was written as an inline table, `{…}`,
// rather than under a header or as a dotted key.
inline bool
// dotted records that a dotted key introduced the table, `a.b = 1`
// building the a around the leaf: the write side gives such a table back
// as dotted key lines, the form that holds the position of a line.
dotted bool
// comments are the lines above the table's header, trailing is the comment
// on the header's own line. Both are empty for a table a dotted key
// introduced, which has no line of its own.
comments []string
trailing string
}
func newTable(values map[string]any) *Table {
return &Table{values: values, index: map[string]*Entry{}}
}
// Keys returns the table's keys in the order they were written. A nil table
// holds none, the answer a document without a root gives through Root.
func (t *Table) Keys() []string {
if t == nil {
return nil
}
keys := make([]string, len(t.entries))
for i, e := range t.entries {
keys[i] = e.key
}
return keys
}
// Values returns the table's values, which is the map the value tree holds for
// it.
func (t *Table) Values() map[string]any {
if t == nil {
return nil
}
return t.values
}
// Entries returns the table's entries in written order.
func (t *Table) Entries() []*Entry {
if t == nil {
return nil
}
return t.entries
}
// Get returns the entry for key, and whether the table has one.
func (t *Table) Get(key string) (*Entry, bool) {
if t == nil {
return nil, false
}
e, ok := t.index[key]
return e, ok
}
// Inline reports whether the table was written as an inline table, `{…}`,
// rather than under a header or introduced by a dotted key.
func (t *Table) Inline() bool { return t != nil && t.inline }
// Comments returns the comment lines above the table's header, or above the
// key that introduced it. Lines carry no leading '#' and no surrounding space.
func (t *Table) Comments() []string {
if t == nil {
return nil
}
return t.comments
}
// SetComments replaces those lines. Each line is written back with a "# " in
// front of it, so a line should not carry one.
func (t *Table) SetComments(lines []string) {
if t == nil {
return
}
t.comments = lines
}
// Trailing returns the comment on the header's own line, without the '#'.
func (t *Table) Trailing() string {
if t == nil {
return ""
}
return t.trailing
}
// SetTrailing replaces that comment.
func (t *Table) SetTrailing(line string) {
if t == nil {
return
}
t.trailing = line
}
// addValue records a key of the table, in written order. The caller gives
// the entry a table node or element nodes when the value has that shape; a
// map value left without a node writes as an inline table.
func (t *Table) addValue(key string, val any, inline bool) *Entry {
e := &Entry{table: t, key: key, inline: inline}
t.entries = append(t.entries, e)
t.index[key] = e
return e
}
// addTable records a key whose value is a table introduced by a header or a
// dotted key, and returns the table's node.
func (t *Table) addTable(key string, values map[string]any) *Table {
if e, ok := t.index[key]; ok {
// The key was seen before, as the leaf of an earlier dotted key.
if e.child == nil {
e.child = newTable(values)
}
return e.child
}
e := t.addValue(key, values, false)
e.child = newTable(values)
return e.child
}
// addElement records one element of an array of tables, and returns its node.
func (t *Table) addElement(key string, values map[string]any) *Table {
e, ok := t.index[key]
if !ok {
e = t.addValue(key, nil, false)
e.elements = []*Table{}
}
el := newTable(values)
e.elements = append(e.elements, el)
return el
}
// child returns the node of a table-valued key, or nil.
func (t *Table) child(key string) *Table {
if t == nil {
return nil
}
if e, ok := t.index[key]; ok {
return e.child
}
return nil
}
// lastElement returns the node of the newest element of an array of tables.
func (t *Table) lastElement(key string) *Table {
if e, ok := t.index[key]; ok && len(e.elements) > 0 {
return e.elements[len(e.elements)-1]
}
return nil
}
// An Entry is one key of a table: the value and the comments around the key.
type Entry struct {
table *Table
key string
inline bool
comments []string
trailing string
// child is the table the value is, and elements are the tables of an array
// of tables; one of them is set only when the value has that shape.
child *Table
elements []*Table
}
// Key returns the key as it was written.
func (e *Entry) Key() string { return e.key }
// Value returns the value the key holds. It is read from the table's map, so
// it stays current if that map is changed.
func (e *Entry) Value() any { return e.table.values[e.key] }
// Inline reports whether the value was written as an inline table, `{…}`.
func (e *Entry) Inline() bool { return e.inline }
// Table returns the table the value is, or nil when it is not a table.
func (e *Entry) Table() *Table { return e.child }
// Elements returns the tables of an array of tables, or nil when the value is
// not one.
func (e *Entry) Elements() []*Table { return e.elements }
// Comments returns the comment lines above the key. Lines carry no leading '#'
// and no surrounding space.
func (e *Entry) Comments() []string { return e.comments }
// SetComments replaces those lines. Each line is written back with a "# " in
// front of it, so a line should not carry one.
func (e *Entry) SetComments(lines []string) { e.comments = lines }
// Trailing returns the comment on the key's own line, without the '#'.
func (e *Entry) Trailing() string { return e.trailing }
// SetTrailing replaces that comment.
func (e *Entry) SetTrailing(line string) { e.trailing = line }
// GetString returns the string the key holds, and whether it holds one.
func (t *Table) GetString(key string) (string, bool) {
if t == nil {
return "", false
}
v, ok := t.values[key]
s, ok := v.(string)
return s, ok
}
// GetInt returns the integer the key holds, and whether it holds one.
func (t *Table) GetInt(key string) (int64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key]
i, ok := v.(int64)
return i, ok
}
// GetFloat returns the float the key holds, and whether it holds one.
func (t *Table) GetFloat(key string) (float64, bool) {
if t == nil {
return 0, false
}
v, ok := t.values[key]
f, ok := v.(float64)
return f, ok
}
// GetBool returns the boolean the key holds, and whether it holds one.
func (t *Table) GetBool(key string) (bool, bool) {
if t == nil {
return false, false
}
v, ok := t.values[key]
b, ok := v.(bool)
return b, ok
}
// GetArray returns the value array the key holds, and whether it holds one.
func (t *Table) GetArray(key string) ([]any, bool) {
if t == nil {
return nil, false
}
v, ok := t.values[key]
a, ok := v.([]any)
return a, ok
}
// GetTable returns the node of the table the key holds, and whether it holds
// one, whichever way the document wrote the table.
func (t *Table) GetTable(key string) (*Table, bool) {
c := t.child(key)
return c, c != nil
}
// Set stores value under key. A key the table already has keeps its position
// and its comments; a new one joins the end. A value of map[string]any
// becomes a table node of its own, written under a header like any other
// table, and replaces the node the key held, which belonged to the value the
// key held; a Go map carries no order, so its keys take sorted order. A value
// of []map[string]any becomes an array-of-tables node.
func (t *Table) Set(key string, value any) {
if t == nil {
return
}
e, ok := t.index[key]
if !ok {
t.values[key] = value
e = t.addValue(key, value, false)
if m, isMap := value.(map[string]any); isMap {
e.child = newOrderedTable(m)
}
if items, isArray := value.([]map[string]any); isArray {
e.elements = make([]*Table, len(items))
for i, item := range items {
e.elements[i] = newOrderedTable(item)
}
}
return
}
t.values[key] = value
switch v := value.(type) {
case map[string]any:
// The node is rebuilt rather than patched: the entries and the index
// belong to the table the key held, and writing the new value
// through them would leave the old table's keys in the output.
e.child = newOrderedTable(v)
e.elements = nil
case []map[string]any:
e.child = nil
e.elements = make([]*Table, len(v))
for i, item := range v {
e.elements[i] = newOrderedTable(item)
}
default:
e.child = nil
e.elements = nil
}
}
// newOrderedTable builds a table node for a value the caller set, its keys
// entered in sorted order, the order Marshal writes maps in.
func newOrderedTable(m map[string]any) *Table {
return orderedTable(m, 0)
}
// orderedTable is newOrderedTable's recursion. The depth bound is the value
// encoder's: a cyclic map stopped here is written by the value writer, which
// reports it instead of running the stack out.
func orderedTable(m map[string]any, depth int) *Table {
t := newTable(m)
for _, k := range slices.Sorted(maps.Keys(m)) {
v := m[k]
e := t.addValue(k, v, false)
if depth >= maxEncodeDepth {
continue
}
switch val := v.(type) {
case map[string]any:
e.child = orderedTable(val, depth+1)
case []map[string]any:
e.elements = make([]*Table, len(val))
for i, item := range val {
e.elements[i] = orderedTable(item, depth+1)
}
}
}
return t
}
// Delete removes key and everything it holds.
func (t *Table) Delete(key string) {
if t == nil {
return
}
if _, ok := t.values[key]; !ok {
return
}
delete(t.values, key)
delete(t.index, key)
for i, e := range t.entries {
if e.key == key {
t.entries = append(t.entries[:i], t.entries[i+1:]...)
break
}
}
}
+559
View File
@@ -0,0 +1,559 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"reflect"
"slices"
"strings"
"testing"
)
// mustEntry returns the entry a table must have, and fails the test when it
// does not.
func mustEntry(t *testing.T, tbl *Table, key string) *Entry {
t.Helper()
e, ok := tbl.Get(key)
if !ok {
t.Fatalf("%q is missing from the table", key)
}
return e
}
func TestDocumentKeepsKeyOrder(t *testing.T) {
doc, err := Parse([]byte(`
b = 1
a = 2
inline = {y = 1, x = 2}
[table]
z = 3
m = 4
`))
if err != nil {
t.Fatalf("parse: %v", err)
}
// The root's keys come back in written order, not sorted.
if got := doc.Root().Keys(); !slices.Equal(got, []string{"b", "a", "inline", "table"}) {
t.Errorf("root keys = %v, want [b a inline table]", got)
}
// So do the keys of an inline table, which the map shape loses.
inline, ok := doc.Root().Get("inline")
if !ok {
t.Fatal("inline is missing from the root")
}
if !inline.Inline() {
t.Error("inline is not marked inline")
}
if got := inline.Table().Keys(); !slices.Equal(got, []string{"y", "x"}) {
t.Errorf("inline keys = %v, want [y x]", got)
}
// And the keys of a table written under a header, which is not inline.
tbl, ok := doc.Root().Get("table")
if !ok {
t.Fatal("table is missing from the root")
}
if tbl.Inline() {
t.Error("table is marked inline")
}
if got := tbl.Table().Keys(); !slices.Equal(got, []string{"z", "m"}) {
t.Errorf("table keys = %v, want [z m]", got)
}
}
func TestDocumentValuesAreTheTree(t *testing.T) {
doc, err := Parse([]byte("n = 7\ns = \"x\"\n\n[t]\nk = true\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
if got := mustEntry(t, doc.Root(), "n").Value(); got != int64(7) {
t.Errorf("n = %#v, want int64(7)", got)
}
tbl := mustEntry(t, doc.Root(), "t").Table()
if got := mustEntry(t, tbl, "k").Value(); got != true {
t.Errorf("t.k = %#v, want true", got)
}
// Map gives the tree ParseMap would have returned, the same values.
tree := doc.Map()
if tree["n"] != int64(7) || tree["s"] != "x" {
t.Errorf("Map = %#v", tree)
}
if tree["t"].(map[string]any)["k"] != true {
t.Errorf("Map[t] = %#v", tree["t"])
}
if tbl.Values()["k"] != true {
t.Errorf("t.Values() = %#v", tbl.Values())
}
}
func TestDocumentComments(t *testing.T) {
doc, err := Parse([]byte(`# above b
b = 1 # trailing b
# above the table
[table] # trailing table
# above m
m = 2
# footer
`))
if err != nil {
t.Fatalf("parse: %v", err)
}
b, ok := doc.Root().Get("b")
if !ok {
t.Fatal("b is missing")
}
if got := b.Comments(); !slices.Equal(got, []string{"above b"}) {
t.Errorf("b comments = %q, want [above b]", got)
}
if got := b.Trailing(); got != "trailing b" {
t.Errorf("b trailing = %q, want \"trailing b\"", got)
}
tbl, ok := doc.Root().Get("table")
if !ok {
t.Fatal("table is missing")
}
// A [header] line introduces the table, so the comments around it belong
// to the table node; the entry that names it stays bare.
if got := tbl.Table().Comments(); !slices.Equal(got, []string{"above the table"}) {
t.Errorf("table comments = %q, want [above the table]", got)
}
if got := tbl.Table().Trailing(); got != "trailing table" {
t.Errorf("table trailing = %q, want \"trailing table\"", got)
}
if got := tbl.Comments(); got != nil {
t.Errorf("entry comments = %q, want none", got)
}
m, ok := tbl.Table().Get("m")
if !ok {
t.Fatal("table.m is missing")
}
if got := m.Comments(); !slices.Equal(got, []string{"above m"}) {
t.Errorf("m comments = %q, want [above m]", got)
}
if got := doc.Footer(); !slices.Equal(got, []string{"footer"}) {
t.Errorf("footer = %q, want [footer]", got)
}
}
func TestDocumentCommentsAreWritable(t *testing.T) {
doc, err := Parse([]byte("# above\nk = 1\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
entry, ok := doc.Root().Get("k")
if !ok {
t.Fatal("k is missing")
}
entry.SetComments([]string{"first", "second"})
entry.SetTrailing("beside")
if got := entry.Comments(); !slices.Equal(got, []string{"first", "second"}) {
t.Errorf("comments = %q", got)
}
if got := entry.Trailing(); got != "beside" {
t.Errorf("trailing = %q", got)
}
tbl := doc.Root()
tbl.SetComments([]string{"above the root"})
if got := tbl.Comments(); !slices.Equal(got, []string{"above the root"}) {
t.Errorf("root comments = %q", got)
}
doc.SetFooter([]string{"end"})
if got := doc.Footer(); !slices.Equal(got, []string{"end"}) {
t.Errorf("footer = %q", got)
}
}
func TestDocumentArrayOfTables(t *testing.T) {
doc, err := Parse([]byte(`# first element
[[item]]
a = 1
[[item]]
b = 2 # beside b
`))
if err != nil {
t.Fatalf("parse: %v", err)
}
entry, ok := doc.Root().Get("item")
if !ok {
t.Fatal("item is missing")
}
elems := entry.Elements()
if len(elems) != 2 {
t.Fatalf("elements = %d, want 2", len(elems))
}
if got := elems[0].Keys(); !slices.Equal(got, []string{"a"}) {
t.Errorf("first element keys = %v, want [a]", got)
}
if got := elems[0].Comments(); !slices.Equal(got, []string{"first element"}) {
t.Errorf("first element comments = %q", got)
}
if got := elems[1].Keys(); !slices.Equal(got, []string{"b"}) {
t.Errorf("second element keys = %v, want [b]", got)
}
if got := mustEntry(t, elems[1], "b").Trailing(); got != "beside b" {
t.Errorf("b trailing = %q, want \"beside b\"", got)
}
// The value keeps the map shape the decoder reads.
if _, ok := entry.Value().([]map[string]any); !ok {
t.Errorf("item value = %#v, want []map[string]any", entry.Value())
}
}
func TestDocumentDottedKeysAndValueArrays(t *testing.T) {
doc, err := Parse([]byte("a.b.c = 1\narr = [1, {x = 1}]\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
// A dotted key builds tables, and they are not inline ones.
a, ok := doc.Root().Get("a")
if !ok {
t.Fatal("a is missing")
}
if a.Inline() {
t.Error("a is marked inline")
}
b, ok := a.Table().Get("b")
if !ok {
t.Fatal("a.b is missing")
}
if b.Inline() {
t.Error("a.b is marked inline")
}
if got := b.Table().Keys(); !slices.Equal(got, []string{"c"}) {
t.Errorf("a.b keys = %v, want [c]", got)
}
// An inline table inside a value array keeps its node in the elements
// slice; the scalar before it has none.
arr, ok := doc.Root().Get("arr")
if !ok {
t.Fatal("arr is missing")
}
elems := arr.Elements()
if len(elems) != 2 {
t.Fatalf("elements = %d, want 2", len(elems))
}
if elems[0] != nil {
t.Errorf("elements[0] = %#v, want nil for a scalar", elems[0])
}
if got := elems[1].Keys(); !slices.Equal(got, []string{"x"}) {
t.Errorf("elements[1] keys = %v, want [x]", got)
}
}
func TestDocumentWithoutStatements(t *testing.T) {
doc, err := Parse([]byte("# only a comment\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
if got := doc.Root().Keys(); len(got) != 0 {
t.Errorf("keys = %v, want none", got)
}
if got := doc.Footer(); !slices.Equal(got, []string{"only a comment"}) {
t.Errorf("footer = %q, want [only a comment]", got)
}
empty, err := Parse(nil)
if err != nil {
t.Fatalf("parse of nothing: %v", err)
}
if len(empty.Root().Keys()) != 0 || len(empty.Footer()) != 0 {
t.Errorf("empty document = %v / %q", empty.Root().Keys(), empty.Footer())
}
}
func TestParseMapIsTheValueTree(t *testing.T) {
// ParseMap is the path that does not build a document, and it gives the
// tree the decoder reads.
tree, err := ParseMap([]byte("a = 1\n\n[t]\nb = \"x\"\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
if tree["a"] != int64(1) {
t.Errorf("a = %#v", tree["a"])
}
if tree["t"].(map[string]any)["b"] != "x" {
t.Errorf("t = %#v", tree["t"])
}
}
func TestMarshalDocument(t *testing.T) {
// A Document writes back: the keys in written order, the comments above
// the lines and headers they belonged to, and inline tables inline again.
doc, err := Parse([]byte("# leading\na = 1 # trailing\n\n[t]\nb = \"x\"\n\ninline = { n = 1 }\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("marshal of a Document: %v", err)
}
want := "# leading\na = 1 # trailing\n\n[t]\nb = \"x\"\ninline = {n = 1}\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
// The written document parses back to the same values.
re, err := Parse(out)
if err != nil {
t.Fatalf("re-parse: %v", err)
}
if got := re.Map()["a"]; got != int64(1) {
t.Errorf("a = %#v", got)
}
if _, err := Marshal(*doc); err != nil {
t.Errorf("marshal of a Document value: %v", err)
}
}
func TestDocumentEditPipeline(t *testing.T) {
doc, err := Parse([]byte("host = \"db\"\nport = 5432\n\n# The cache section\ntimeout = 1.5\n"))
if err != nil {
t.Fatal(err)
}
t.Run("typed getters", func(t *testing.T) {
if s, ok := doc.GetString("host"); !ok || s != "db" {
t.Errorf("host = %q, %v", s, ok)
}
if i, ok := doc.GetInt("port"); !ok || i != 5432 {
t.Errorf("port = %d, %v", i, ok)
}
if f, ok := doc.GetFloat("timeout"); !ok || f != 1.5 {
t.Errorf("timeout = %g, %v", f, ok)
}
if _, ok := doc.GetBool("host"); ok {
t.Error("host claimed as bool")
}
})
t.Run("set keeps the position and the comments", func(t *testing.T) {
doc.Set("port", int64(9090))
if got := doc.Root().Keys(); !slices.Equal(got, []string{"host", "port", "timeout"}) {
t.Fatalf("keys = %v", got)
}
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(out), "port = 9090") {
t.Errorf("output missing the new value:\n%s", out)
}
})
t.Run("a new key joins the end", func(t *testing.T) {
doc.Set("lang", "cs")
if got := doc.Root().Keys(); !slices.Equal(got, []string{"host", "port", "timeout", "lang"}) {
t.Fatalf("keys = %v", got)
}
})
t.Run("a set table keeps an order of its own", func(t *testing.T) {
sub := map[string]any{"z": int64(1), "a": int64(2)}
doc.Set("cache", sub)
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(out), "[cache]\na = 2\nz = 1\n") {
t.Errorf("output missing the new table in order:\n%s", out)
}
})
t.Run("delete removes the key", func(t *testing.T) {
doc.Delete("lang")
if _, ok := doc.Get("lang"); ok {
t.Fatal("lang survived Delete")
}
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
if strings.Contains(string(out), "lang") {
t.Errorf("output still names lang:\n%s", out)
}
})
t.Run("UnmarshalDocument decodes without reparsing", func(t *testing.T) {
type Cfg struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
var cfg Cfg
if err := UnmarshalDocument(doc, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Host != "db" || cfg.Port != 9090 {
t.Errorf("decoded %+v", cfg)
}
})
t.Run("comments survive the round trip", func(t *testing.T) {
src := "# header comment\n[a]\n# key comment\nb = 2\n"
doc, err := Parse([]byte(src))
if err != nil {
t.Fatal(err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatal(err)
}
for _, want := range []string{"# header comment", "[a]", "# key comment", "b = 2"} {
if !strings.Contains(string(out), want) {
t.Errorf("output missing %q:\n%s", want, out)
}
}
})
t.Run("a nil document refuses to decode", func(t *testing.T) {
var cfg struct {
A int `toml:"a"`
}
if err := UnmarshalDocument(nil, &cfg); err == nil {
t.Error("UnmarshalDocument(nil) succeeded, want an error")
}
})
}
// TestMarshalDocumentRoundTrips pins that a parsed document written back
// re-parses to the same tree: arrays of tables keep exactly one header per
// element, dotted keys hold their line position without swallowing the keys
// after them, inline tables inside value arrays keep their written order,
// and comments travel with their statements.
func TestMarshalDocumentRoundTrips(t *testing.T) {
tests := []struct {
name string
src string
}{
{"array of tables", "[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n"},
{"array of tables with comments", "# about items\n[[items]] # first\nname = \"a\"\n"},
{"dotted key before a later key", "a.b = 1\nc = 2\n"},
{"dotted keys grouped", "a.b = 1\na.c = 2\nd = 3\n"},
{"dotted key with a nested leaf", "a.b.c = 1\nz = 2\n"},
{"header section after a dotted key", "a.b = 1\n\n[a.x]\ny = 2\n"},
{"inline tables in a value array keep order", "arr = [{y = 1, x = 2}, {second = true, first = false}]\n"},
{"nested array of tables", "[[items]]\nn = 1\n\n[items.sub]\nk = \"v\"\n"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
doc, err := Parse([]byte(tt.src))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
reparsed, err := Parse(out)
if err != nil {
t.Fatalf("re-parse of %q: %v", out, err)
}
if !reflect.DeepEqual(doc.Map(), reparsed.Map()) {
t.Errorf("round trip changed the tree:\nin: %#v\nout: %#v", doc.Map(), reparsed.Map())
}
if got, want := reparsed.Root().Keys(), doc.Root().Keys(); !slices.Equal(got, want) {
t.Errorf("root keys = %v, want %v", got, want)
}
})
}
}
// TestMarshalDocumentArrayComments pins where the comments of an array of
// tables land: above and beside the [[header]] itself.
func TestMarshalDocumentArrayComments(t *testing.T) {
doc, err := Parse([]byte("# element one\n[[items]] # trailing\nname = \"a\"\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "# element one\n[[items]] # trailing\nname = \"a\"\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetReplacesTableNode pins that Set over a key holding a table
// rebuilds the node, so the new map's keys are the ones written.
func TestTableSetReplacesTableNode(t *testing.T) {
doc, err := Parse([]byte("[cache]\nz = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("cache", map[string]any{"a": int64(2)})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
want := "[cache]\na = 2\n"
if string(out) != want {
t.Errorf("output = %q, want %q", out, want)
}
}
// TestTableSetNestedArraysOfTables pins that a value set through the edit API
// carries its arrays of tables into the header form.
func TestTableSetNestedArraysOfTables(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
doc.Set("t", map[string]any{"items": []map[string]any{{"n": int64(1)}, {"n": int64(2)}}})
out, err := Marshal(doc)
if err != nil {
t.Fatalf("Marshal: %v", err)
}
if !strings.Contains(string(out), "[[t.items]]") {
t.Errorf("output = %q, want the array of tables under a header", out)
}
}
// TestTableSetCyclicMapErrors pins that a cyclic map set through the edit API
// reaches the depth limit instead of the stack.
func TestTableSetCyclicMapErrors(t *testing.T) {
doc, err := Parse([]byte("x = 1\n"))
if err != nil {
t.Fatalf("Parse: %v", err)
}
m := map[string]any{}
m["self"] = m
doc.Set("cyclic", m)
if _, err := Marshal(doc); err == nil || !strings.Contains(err.Error(), "nests deeper") {
t.Errorf("err = %v, want the depth-limit complaint", err)
}
}
// TestDocumentNilSafety pins that the nil document answers its readers
// instead of panicking, the contract Root already carries.
func TestDocumentNilSafety(t *testing.T) {
var doc *Document
if doc.Map() != nil {
t.Errorf("Map = %v", doc.Map())
}
if doc.Footer() != nil {
t.Errorf("Footer = %v", doc.Footer())
}
doc.SetFooter([]string{"x"})
if e, ok := doc.Get("k"); e != nil || ok {
t.Errorf("Get = %v, %v", e, ok)
}
if _, ok := doc.GetString("k"); ok {
t.Error("GetString on a nil document reports a value")
}
if _, ok := doc.GetTable("k"); ok {
t.Error("GetTable on a nil document reports a value")
}
doc.Set("k", 1)
doc.Delete("k")
if keys := doc.Root().Keys(); keys != nil {
t.Errorf("Keys = %v", keys)
}
if doc.Root().Entries() != nil {
t.Errorf("Entries = %v", doc.Root().Entries())
}
}
+314
View File
@@ -0,0 +1,314 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"fmt"
)
// UnmarshalDocument decodes a parsed Document into v without parsing again,
// the shape an edit pipeline wants: read the document, change the values it
// holds, decode the result into a typed destination. The key order and the
// comments the document carries are untouched; the decode reads the value
// tree the document shares with its nodes.
//
// UnmarshalDocument accepts the same destinations Unmarshal does.
func UnmarshalDocument(doc *Document, v any) error {
if doc == nil {
return fmt.Errorf("interpres: cannot decode a nil Document")
}
dec := newDecoder()
dec.nodes = indexNodes(doc.Root())
return dec.decode(doc.Map(), v)
}
// writeDocument renders a Document back to TOML: the keys in written order,
// the comments above the lines and headers they belonged to, tables that
// were written inline written inline again, and an array of tables in its
// header form. It is the write side of the edit pipeline: read with Parse,
// change with the Table and Document mutators, write with Marshal.
func (e *encoder) writeDocument(doc *Document) error {
if err := e.checkCtx(); err != nil {
return err
}
if doc == nil || doc.root == nil {
return fmt.Errorf("interpres: cannot marshal a nil Document")
}
if err := e.writeTableEntries(doc.root, nil); err != nil {
return err
}
e.writeDocumentFooter(doc.footer)
return nil
}
// writeDocumentFooter writes the comment lines that follow the last
// statement. The parser collects them wherever they sit after it, so the
// writer needs no blank line of its own to have them read back.
func (e *encoder) writeDocumentFooter(footer []string) {
for _, line := range footer {
e.buf.WriteString("# ")
e.buf.WriteString(line)
e.buf.WriteByte('\n')
}
}
// writeTableEntries writes one table at the given header path, nil for the
// document root, whose keys need no header: the blank line, the comments,
// the header line with its trailing comment, then the body.
func (e *encoder) writeTableEntries(t *Table, path []string) error {
if t == nil {
return nil
}
if path != nil {
e.writeBlankLine()
e.writeComments(t.Comments())
e.buf.WriteString("[")
if err := e.writeKeyPath(path); err != nil {
return err
}
e.buf.WriteString("]")
if tr := t.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
}
return e.writeTableBody(t, path)
}
// writeTableBody writes one table's entries: the value lines first, in
// written order, then the header sections. In a valid document every line at
// one level precedes the headers below it, so the split reorders nothing;
// what it prevents is a table a dotted key introduced, which the parse nests
// as a sub-table at the position of a line, from swallowing the lines that
// follow it into its header.
func (e *encoder) writeTableBody(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if !e.isLineEntry(entry) {
continue
}
if err := e.writeLineEntry(entry, path); err != nil {
return err
}
}
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
// A dotted table writes as lines above; its own header-form
// sub-tables are sections the document placed after those lines,
// so the section pass reaches through the dotted entry.
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeDottedSections writes the header-form sub-tables of a dotted table:
// the sections the document placed after the dotted lines, reached through
// the dotted entry itself.
func (e *encoder) writeDottedSections(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && child.dotted && !entry.Inline() {
if err := e.writeDottedSections(child, append(append([]string{}, path...), entry.Key())); err != nil {
return err
}
continue
}
if e.isLineEntry(entry) {
continue
}
if err := e.writeSectionEntry(entry, path); err != nil {
return err
}
}
return nil
}
// writeSectionEntry writes one entry the line pass left behind: a table or
// an array of tables under its header, at the path this level carries.
func (e *encoder) writeSectionEntry(entry *Entry, path []string) error {
if _, isTables := entry.Value().([]map[string]any); isTables {
// An array of tables keeps its header form, one element per header
// with the element's own comments above it; the body that follows is
// the element's, with no header of its own to repeat.
elemPath := append(append([]string{}, path...), entry.Key())
for i, el := range entry.Elements() {
e.writeBlankLine()
if i == 0 {
e.writeComments(entry.Comments())
}
e.writeComments(el.Comments())
e.buf.WriteString("[[")
if err := e.writeKeyPath(elemPath); err != nil {
return err
}
e.buf.WriteString("]]")
if tr := el.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
e.buf.WriteByte('\n')
if err := e.writeTableBody(el, elemPath); err != nil {
return err
}
}
return nil
}
headerPath := append(append([]string{}, path...), entry.Key())
return e.writeTableEntries(entry.Table(), headerPath)
}
// isLineEntry reports whether an entry writes as one or more "key = value"
// lines at its own level: a value, an inline table, or a table a dotted key
// introduced, which goes back as dotted keys. An emptied array of tables
// counts as one only so the line pass can drop it, the omission the value
// encoder applies to an empty array of tables too.
func (e *encoder) isLineEntry(entry *Entry) bool {
if child := entry.Table(); child != nil {
return entry.Inline() || child.dotted
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return len(entry.Elements()) == 0
}
return true
}
// writeLineEntry writes one entry as lines at this level, and drops an
// emptied array of tables, which has no TOML form.
func (e *encoder) writeLineEntry(entry *Entry, path []string) error {
if child := entry.Table(); child != nil && !entry.Inline() {
return e.writeDottedTable(child, append(append([]string{}, path...), entry.Key()))
}
if _, isTables := entry.Value().([]map[string]any); isTables {
return nil
}
return e.writeDocumentEntry(entry)
}
// writeDottedTable writes a table a dotted key introduced as one dotted line
// per leaf, in written order: `a.b = 1`. A sub-table the document added
// under a header stays a section and is left to the section pass.
func (e *encoder) writeDottedTable(t *Table, path []string) error {
for _, entry := range t.Entries() {
if err := e.checkCtx(); err != nil {
return err
}
if child := entry.Table(); child != nil && !entry.Inline() && !child.dotted {
continue
}
leafPath := append(append([]string{}, path...), entry.Key())
if child := entry.Table(); child != nil && !entry.Inline() {
if err := e.writeDottedTable(child, leafPath); err != nil {
return err
}
continue
}
e.writeComments(entry.Comments())
if err := e.writeKeyPath(leafPath); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
}
return nil
}
// writeDocumentEntry writes one "key = value" line of a document, with the
// comments the key carried. A value that is itself an inline table renders
// inline from its node, in the written order.
func (e *encoder) writeDocumentEntry(entry *Entry) error {
e.writeComments(entry.Comments())
if err := e.writeKey(entry.Key()); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeEntryValueNodes(entry); err != nil {
return err
}
e.buf.WriteByte('\n')
return nil
}
// writeEntryValueNodes writes the value of a document entry. An inline table
// node keeps the written key order even inside a value array, where the
// ordinary value writer would sort the keys.
func (e *encoder) writeEntryValueNodes(entry *Entry) error {
if child := entry.Table(); child != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err
}
} else if arr, ok := entry.Value().([]any); ok {
elems := entry.Elements()
e.buf.WriteByte('[')
for i, item := range arr {
if i > 0 {
e.buf.WriteString(", ")
}
if i < len(elems) && elems[i] != nil {
if err := e.writeInlineTableNode(elems[i]); err != nil {
return err
}
continue
}
if err := e.writeValue(item); err != nil {
return err
}
}
e.buf.WriteByte(']')
} else if err := e.writeValue(entry.Value()); err != nil {
return err
}
if tr := entry.Trailing(); tr != "" {
e.buf.WriteString(" # ")
e.buf.WriteString(tr)
}
return nil
}
// writeInlineTableNode renders a table node as an inline table, its keys in
// written order, values that are tables inline in turn.
func (e *encoder) writeInlineTableNode(t *Table) error {
e.buf.WriteByte('{')
for i, key := range t.Keys() {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeKey(key); err != nil {
return err
}
e.buf.WriteString(" = ")
entry, _ := t.Get(key)
if child := entry.Table(); child != nil {
if err := e.writeInlineTableNode(child); err != nil {
return err
}
continue
}
if err := e.writeValue(t.Values()[key]); err != nil {
return err
}
}
e.buf.WriteByte('}')
return nil
}
+1184 -158
View File
File diff suppressed because it is too large Load Diff
+1086 -62
View File
File diff suppressed because it is too large Load Diff
+101
View File
@@ -0,0 +1,101 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres_test
import (
"fmt"
"log"
"sourcedock.dev/petrbalvin/interpres/v2"
)
func ExampleParse() {
const doc = `
title = "interpres"
[server]
host = "127.0.0.1"
port = 9090
`
d, err := interpres.Parse([]byte(doc))
if err != nil {
log.Fatal(err)
}
for _, key := range d.Root().Keys() { // written order, not sorted
entry, _ := d.Root().Get(key)
fmt.Println(key, "=", entry.Value())
}
// Output:
// title = interpres
// server = map[host:127.0.0.1 port:9090]
}
func ExampleUnmarshal() {
type Config struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
var cfg Config
err := interpres.Unmarshal([]byte("host = \"db\"\nport = 5432\n"), &cfg)
if err != nil {
log.Fatal(err)
}
fmt.Println(cfg.Host, cfg.Port)
// Output: db 5432
}
func ExampleMarshal() {
type Server struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
type Config struct {
Title string `toml:"title"`
Server Server `toml:"server"`
}
out, err := interpres.Marshal(Config{
Title: "demo",
Server: Server{Host: "127.0.0.1", Port: 9090},
})
if err != nil {
log.Fatal(err)
}
fmt.Printf("%s", out)
// Output:
// title = "demo"
//
// [server]
// host = "127.0.0.1"
// port = 9090
}
func ExampleNumbersAsLiterals() {
var tree map[string]any
err := interpres.Unmarshal([]byte("rate = 1_000\n"), &tree,
interpres.RejectUnknownFields(true),
interpres.NumbersAsLiterals(true))
if err != nil {
log.Fatal(err)
}
fmt.Println(tree["rate"], string(tree["rate"].(interpres.Number)))
// Output: 1_000 1_000
}
func ExampleInlineTables() {
type Config struct {
Title string `toml:"title"`
Extras map[string]string `toml:"extras,inline"`
}
out, err := interpres.Marshal(Config{Title: "demo", Extras: map[string]string{"b": "two", "a": "one"}},
interpres.Layout(interpres.LayoutKindDeclaration),
interpres.LiteralMultiline(80),
interpres.InlineTables(40))
if err != nil {
log.Fatal(err)
}
fmt.Printf("%s", out)
// Output:
// title = "demo"
// extras = {a = "one", b = "two"}
}
+44 -11
View File
@@ -3,22 +3,22 @@
// Command basic demonstrates decoding and encoding a TOML document with
// interpres. It exercises struct mapping, arrays of tables, Marshaler
// customisation, the Decoder's strict mode, and the Encoder's policy
// options, covering every feature a regular user would reach for.
// customisation, and the Encoder's policy options, covering every feature a
// regular user would reach for.
package main
import (
"errors"
"fmt"
"io"
"os"
"time"
"sourcedock.dev/petrbalvin/interpres"
"sourcedock.dev/petrbalvin/interpres/v2"
)
// document is a small but realistic configuration: it has scalars, a
// sub-table, an array of tables, and a date-time. We pick a 32-bit port so
// the demonstration also covers overflow-safe integer conversion.
// sub-table, an array of tables, and a date-time.
const document = `
title = "interpres demo"
launched = 2024-11-04T09:00:00Z
@@ -39,13 +39,16 @@ admin = false
// Config mirrors the document above. The Server field is a named struct so
// the reader sees explicit subtable boundaries; Users is a slice of named
// structs so the array-of-tables path is exercised.
// structs so the array-of-tables path is exercised. Retries carries the
// `omitzero` tag option: a zero value of the field's type drops from the
// output, and a `time.Duration` zero is zero nanoseconds.
type Config struct {
Title string `toml:"title"`
Launched time.Time `toml:"launched"`
Debug bool `toml:"debug"`
Server Server `toml:"server"`
Users []User `toml:"users"`
Retries time.Duration `toml:"retries,omitzero"`
}
type Server struct {
@@ -58,9 +61,11 @@ type User struct {
Admin bool `toml:"admin"`
}
// Port is a typed alias that controls how its value appears in TOML. The
// MarshalTOML hook returns a string, so a Port field is rendered as
// "host:port" instead of the raw integer.
// Port is a typed string alias that carries a Marshaler. The MarshalTOML
// hook returns the string unchanged, so a Port field is rendered as the
// string it holds, a string the encoding would print the same way without
// the hook; the demonstration that a Marshaler reshapes a value is
// Endpoint's below.
type Port string
func (p Port) MarshalTOML() (any, error) {
@@ -130,12 +135,12 @@ func Run(stdout, stderr io.Writer) int {
}
fmt.Fprintf(stdout, "\n--- marshal (group by kind, default) ---\n%s", out)
out2, err := interpres.NewEncoder().GroupByKind(false).Marshal(cfg)
out2, err := interpres.Marshal(cfg, interpres.Layout(interpres.LayoutKindDeclaration))
if err != nil {
fmt.Fprintln(stderr, "marshal:", err)
return 1
}
fmt.Fprintf(stdout, "\n--- marshal (GroupByKind=false) ---\n%s", out2)
fmt.Fprintf(stdout, "\n--- marshal (LayoutKindDeclaration) ---\n%s", out2)
// Demonstrate Unmarshaler-style mutation: re-decode the second output to
// prove it round-trips back into the same Go value.
@@ -147,5 +152,33 @@ func Run(stdout, stderr io.Writer) int {
fmt.Fprintf(stdout, "\n--- round-trip --- ok (title=%q, users=%d)\n",
roundTripped.Title, len(roundTripped.Users))
// Typed errors: a decode failure names the key path it failed at, and
// errors.AsType reaches the DecodeError to read the path and the cause
// separately, without parsing the message text.
bad := []byte("[[users]]\nname = \"x\"\nadmin = \"not-a-bool\"\n")
var badCfg Config
err = interpres.Unmarshal(bad, &badCfg)
if err == nil {
fmt.Fprintln(stderr, "expected a decode error")
return 1
}
if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
fmt.Fprintf(stdout, "\n--- typed error --- path %s: %v\n", de.Path.String(), de.Err)
} else {
fmt.Fprintln(stderr, "expected a DecodeError")
return 1
}
// omitzero: the retries field carries the tag option and a zero duration,
// so the re-encoded config above simply has no retries line. Give it a
// value and the line appears.
cfg.Retries = 30 * time.Second
out3, err := interpres.Marshal(cfg)
if err != nil {
fmt.Fprintln(stderr, "marshal:", err)
return 1
}
fmt.Fprintf(stdout, "\n--- omitzero ---\n%s", out3)
return 0
}
+1 -1
View File
@@ -25,7 +25,7 @@ func TestRunPrintsConfigAndMarshal(t *testing.T) {
"admin=false",
"--- marshal (group by kind, default) ---",
`title = "interpres demo"`,
"--- marshal (GroupByKind=false) ---",
"--- marshal (LayoutKindDeclaration) ---",
"[server]",
"port = 9090",
"[[users]]",
+39
View File
@@ -0,0 +1,39 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
// Command statements walks the top-level statements of a TOML document with
// interpres.Statements, the shape a configuration tool uses to read the
// sections it cares about and skip the rest.
package main
import (
"fmt"
"io"
"os"
"sourcedock.dev/petrbalvin/interpres/v2"
)
func main() {
if err := run(os.Stdin, os.Stdout); err != nil {
fmt.Fprintln(os.Stderr, err)
os.Exit(1)
}
}
func run(stdin io.Reader, stdout io.Writer) error {
for stmt, err := range interpres.Statements(stdin) {
if err != nil {
return err
}
switch {
case stmt.Index >= 0:
fmt.Fprintf(stdout, "[[%s]] #%d\n", stmt.Key, stmt.Index)
case stmt.Table != nil:
fmt.Fprintf(stdout, "[%s] keys: %v\n", stmt.Key, stmt.Table.Keys())
default:
fmt.Fprintf(stdout, "%s = %v\n", stmt.Key, stmt.Value)
}
}
return nil
}
+39
View File
@@ -0,0 +1,39 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package main
import (
"strings"
"testing"
)
func TestStatementsExample(t *testing.T) {
in := strings.NewReader(`title = "demo"
port = 8080
[server]
host = "127.0.0.1"
[[items]]
name = "a"
[[items]]
name = "b"
`)
var out strings.Builder
if err := run(in, &out); err != nil {
t.Fatal(err)
}
for _, want := range []string{
"title = demo",
"port = 8080",
"[server] keys: [host]",
"[[items]] #0",
"[[items]] #1",
} {
if !strings.Contains(out.String(), want) {
t.Errorf("output missing %q:\n%s", want, out.String())
}
}
}
+142
View File
@@ -0,0 +1,142 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"context"
"errors"
"reflect"
"strings"
"testing"
"time"
)
type fuzzNested struct {
X int `toml:"x"`
Y string `toml:"y"`
}
type fuzzDoc struct {
Num int `toml:"num"`
Flt float64 `toml:"flt"`
Str string `toml:"str"`
Flag bool `toml:"flag"`
Small uint8 `toml:"small"`
When time.Time `toml:"when"`
Tags []string `toml:"tags"`
Lims map[string]any `toml:"lims"`
Tab fuzzNested `toml:"tab"`
Arr []fuzzNested `toml:"arr"`
Other string `toml:"other"`
}
// fuzzStmts is the statement pool the generated documents draw from: every
// destination kind the targeted parse handles, beside the shapes that make
// it fall back (overflow, unknown tables, duplicate keys).
var fuzzStmts = []string{
`num = 1`, `num = 300`, `small = 300`, `small = 7`,
`flt = 2.5`, `str = "x"`, `flag = true`,
`when = 1979-05-27T07:32:00Z`,
`tags = ["a", "b"]`, `tags = []`, `lims = { k = 1 }`,
`[tab]`, `tab.x = 1`, `tab.y = "s"`, `x = 2`, `y = "t"`,
`[[arr]]`, `x = 3`, `y = "u"`,
`[tab.nested]`, `x = 4`,
`other = "o"`, `zz = 1`, `[zz]`, `k = 1`,
`num = 2`,
}
func fuzzDocument(data []byte) []byte {
var b strings.Builder
for i, by := range data {
if i > 0 {
b.WriteByte('\n')
}
b.WriteString(fuzzStmts[int(by)%len(fuzzStmts)])
}
return []byte(b.String())
}
// treeDecodeInto is the reference decode: the ordinary tree path, non-strict
// like the fuzz decode; the strict contracts have their own deterministic
// tests.
func treeDecodeInto(data []byte, v any) error {
dec := newDecoder()
tree, _, err := parseWithOptions(context.Background(), data, parseOptions{}, false)
if err != nil {
return err
}
return dec.decode(tree, v)
}
// decodeFinding normalises an error for the comparison. Decode-stage
// findings several tables may produce (an unknown field, a missing required
// key) compare as their class alone: the tree decode picks the reporting
// table by map order and so does not promise one. Everything else compares
// as its exact text.
func decodeFinding(err error) string {
if err == nil {
return ""
}
if de, ok := errors.AsType[*DecodeError](err); ok {
if strings.Contains(de.Err.Error(), "unknown field") {
return "unknown"
}
if strings.Contains(de.Err.Error(), "missing required key") {
return "required"
}
return de.Path.String() + ": " + de.Err.Error()
}
return err.Error()
}
// FuzzTargetedDecode holds the targeted parse to the tree decode as its
// reference: for every generated document the two paths must agree on the
// error class and on the decoded value.
func FuzzTargetedDecode(f *testing.F) {
seeds := []string{
"num = 1\nstr = \"x\"\n[tab]\nx = 2\n[[arr]]\nx = 3\n",
"small = 300\n",
"[tab]\ntab.x = 1\n",
"lims = { k = 1 }\ntags = [\"a\"]\n",
"[[arr]]\ny = \"u\"\n[zz]\nk = 1\n",
"when = 07:32:00\n[tab.nested]\n",
"small = 300\n[[arr]]\nflt = 2.5\n",
}
for _, s := range seeds {
f.Add([]byte(s))
}
f.Fuzz(func(t *testing.T, data []byte) {
doc := fuzzDocument(data)
var tgt fuzzDoc
tgtErr := Unmarshal(doc, &tgt)
if tgtErr != nil {
// A document with several decode-stage findings reports a different
// one per run (the tree decode walks its maps in random order), so the
// reference gets a few chances to produce the finding the targeted
// side carries. The targeted error is either the tree's own or the
// fallback already reran the tree.
for i := range 8 {
var ref fuzzDoc
refErr := treeDecodeInto(doc, &ref)
if refErr == nil {
t.Fatalf("reference succeeded on retry %d, targeted failed: %v\ndoc:\n%s", i, tgtErr, doc)
}
if decodeFinding(refErr) == decodeFinding(tgtErr) {
return
}
if i == 7 {
t.Fatalf("errors disagree after retries:\ntargeted: %v\nlast tree: %v\ndoc:\n%s", tgtErr, refErr, doc)
}
}
}
var ref fuzzDoc
refErr := treeDecodeInto(doc, &ref)
if refErr != nil {
t.Fatalf("reference failed, targeted succeeded: %v\ndoc:\n%s", refErr, doc)
}
if !reflect.DeepEqual(ref, tgt) {
t.Fatalf("values disagree:\ntree: %#v\ntargeted: %#v\ndoc:\n%s", ref, tgt, doc)
}
})
}
+70 -2
View File
@@ -41,7 +41,7 @@ func FuzzParse(f *testing.F) {
f.Add([]byte(s))
}
f.Fuzz(func(t *testing.T, data []byte) {
tree, err := Parse(data)
tree, err := ParseMap(data)
if err != nil {
return
}
@@ -49,7 +49,7 @@ func FuzzParse(f *testing.F) {
if err != nil {
t.Fatalf("marshal of a parsed tree failed: %v\ntree: %#v", err, tree)
}
re, err := Parse(out)
re, err := ParseMap(out)
if err != nil {
t.Fatalf("re-parse of the emitted document failed: %v\ndoc:\n%s", err, out)
}
@@ -121,3 +121,71 @@ func tomlEqual(a, b any) bool {
return reflect.DeepEqual(a, b)
}
}
// FuzzMarshal drives the encoder with generated Go values and holds it to
// the same round-trip invariant FuzzParse holds the parser to: a value built
// only of encodable kinds must marshal, the document must re-parse, and the
// tree must equal the value it came from.
func FuzzMarshal(f *testing.F) {
seeds := [][]byte{
{},
{0, 0, 1, 2},
{1, 1, 2, 3, 2, 2, 3, 4},
{0, 3, 1, 9, 3, 3, 2, 8, 1, 0, 1, 7},
}
for _, s := range seeds {
f.Add(s)
}
f.Fuzz(func(t *testing.T, data []byte) {
v := fuzzValue(data)
out, err := Marshal(v)
if err != nil {
t.Fatalf("marshal of an encodable value failed: %v\nvalue: %#v", err, v)
}
tree, err := ParseMap(out)
if err != nil {
t.Fatalf("re-parse of the emitted document failed: %v\ndoc:\n%s", err, out)
}
if !tomlEqual(v, tree) {
t.Fatalf("round-trip changed the value\nvalue: %#v\ndoc:\n%s\ntree: %#v", v, out, tree)
}
})
}
// fuzzKeys is the fixed key pool the generated values draw from, so keys are
// always valid bare keys and repeat often.
var fuzzKeys = []string{"alpha", "beta", "gamma", "delta"}
// fuzzValue builds a map[string]any of encodable kinds from data: integers,
// positive floats, short strings, nested tables and scalar arrays. The bytes
// decide the shape deterministically.
func fuzzValue(data []byte) map[string]any {
root := map[string]any{}
cur := root
depth := 0
for i := 0; i+3 < len(data); i += 4 {
key := fuzzKeys[int(data[i])%len(fuzzKeys)]
switch data[i+1] % 5 {
case 0:
cur[key] = int64(data[i+2])<<8 | int64(data[i+3])
case 1:
cur[key] = float64(int(data[i+2])%1000)/8.0 + 0.125
case 2:
cur[key] = string(rune('a' + int(data[i+2])%26))
case 3:
cur[key] = []any{
int64(data[i+2]),
float64(int(data[i+3])%100)/4.0 + 0.25,
string(rune('a' + int(data[i+3])%26)),
}
case 4:
if depth < 6 {
next := map[string]any{}
cur[key] = next
cur = next
depth++
}
}
}
return root
}
+1 -1
View File
@@ -1,3 +1,3 @@
module sourcedock.dev/petrbalvin/interpres
module sourcedock.dev/petrbalvin/interpres/v2
go 1.27.1
+583 -141
View File
@@ -11,25 +11,39 @@
//
// out, err := interpres.Marshal(cfg)
//
// or, for an untyped tree:
// or, for the document with its key order and comments:
//
// tree, err := interpres.Parse(data)
// doc, err := interpres.Parse(data)
// tree := doc.Map()
//
// A Decoder allows strict decoding that rejects keys without a matching
// struct field, mirroring (*json.Decoder).DisallowUnknownFields.
// Strict decoding that rejects keys without a matching struct field is an
// option, mirroring the RejectUnknownMembers option of encoding/json/v2:
//
// err := interpres.Unmarshal(data, &cfg, interpres.RejectUnknownFields(true))
package interpres
import (
"bytes"
"context"
"errors"
"fmt"
"unicode/utf8"
"io"
"iter"
"os"
"reflect"
"slices"
"strings"
"time"
)
// A SyntaxError describes a malformed TOML document, including the 1-based
// line on which the problem was detected.
// A SyntaxError describes a malformed TOML document. Line is the 1-based line
// the problem was detected on. Offset is the byte offset in the input the scan
// stopped at, and Column is the 1-based column on that line; both are new in
// 2.0 and a struct literal that names Line and Msg alone still builds.
type SyntaxError struct {
Line int
Offset int
Column int
Msg string
}
@@ -37,37 +51,87 @@ func (e *SyntaxError) Error() string {
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
}
// SourceLine returns the source line the error points at, rendered from src,
// followed by a caret line marking the column. It is meant for a message the
// reader sees under the input:
//
// port = = 8080
// ^
//
// The caret sits at Offset when it falls inside src, and at the start of the
// line when the error carries no position.
func (e *SyntaxError) SourceLine(src []byte) string {
off := min(e.Offset, len(src))
start := 0
if i := bytes.LastIndexByte(src[:off], '\n'); i >= 0 {
start = i + 1
}
end := len(src)
if i := bytes.IndexByte(src[start:], '\n'); i >= 0 {
end = start + i
}
return string(src[start:end]) + "\n" + strings.Repeat(" ", off-start) + "^"
}
// A Path names a value in a document, one segment per level from the root:
// a key contributes its name and an array element its bracketed index, so the
// path of the weight field of the first item is the segments
// ["items", "[0]", "weight"]. String renders the TOML notation,
// "items[0].weight".
type Path []string
// String renders the path the way a TOML document writes it: keys join with
// dots and an index attaches to the previous segment in brackets.
func (p Path) String() string {
var b strings.Builder
for _, s := range p {
if strings.HasPrefix(s, "[") {
b.WriteString(s)
continue
}
if b.Len() > 0 {
b.WriteByte('.')
}
b.WriteString(s)
}
return b.String()
}
// A DecodeError wraps a decoding failure with the key path at which it
// happened. Path lists one segment per level from the document root, the
// outermost key first: a key contributes its name and an array element its
// bracketed index, so the path of the weight field in the first item reads
// ["items", "[0]", "weight"]. The rendered message is unchanged by the type;
// read it programmatically with errors.AsType:
// happened. Read the path programmatically with errors.AsType:
//
// if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
// fmt.Println(de.Path, de.Err)
// fmt.Println(de.Path.String(), de.Err)
// }
type DecodeError struct {
// Path is the key path from the document root, outermost key first.
Path []string
Path Path
// Err is the failure at that path.
Err error
}
func (e *DecodeError) Error() string { return e.Path[0] + ": " + e.Err.Error() }
func (e *DecodeError) Error() string {
msg := strings.TrimPrefix(e.Err.Error(), "interpres: ")
if p := e.Path.String(); p != "" {
return "interpres: " + p + ": " + msg
}
return "interpres: " + msg
}
// Unwrap returns the failure the path points at.
func (e *DecodeError) Unwrap() error { return e.Err }
// newDecodeError wraps err with one path segment. The rest of the path comes
// from the DecodeError err already carries, if any: the decoder wraps each
// key and index on its way down, so the innermost wrap holds the deepest
// segments and each outer wrap prepends one.
// key and index on its way down, so the wrap flattens that inner error's
// segments onto the front and keeps the failure it pointed at, leaving one
// path and one failure to render.
func newDecodeError(key string, err error) *DecodeError {
path := make([]string, 0, 4)
path := make(Path, 0, 4)
path = append(path, key)
if de, ok := errors.AsType[*DecodeError](err); ok {
path = append(path, de.Path...)
err = de.Err
}
return &DecodeError{Path: path, Err: err}
}
@@ -79,104 +143,317 @@ func newDecodeError(key string, err error) *DecodeError {
// unchanged by the type; read it programmatically with errors.AsType.
type EncodeError struct {
// Path is the key path of the failing value.
Path string
Path Path
// Err is the failure at that path.
Err error
}
func (e *EncodeError) Error() string { return "interpres: " + e.Path + ": " + e.Err.Error() }
func (e *EncodeError) Error() string {
msg := strings.TrimPrefix(e.Err.Error(), "interpres: ")
if p := e.Path.String(); p != "" {
return "interpres: " + p + ": " + msg
}
return "interpres: " + msg
}
// Unwrap returns the failure the path points at.
func (e *EncodeError) Unwrap() error { return e.Err }
// Parse decodes a TOML document into a nested map[string]any.
// Parse decodes a TOML document into a Document: the values, the order the
// keys were written in, whether a table was written inline, and the comments.
// ParseMap gives the plain value tree instead.
//
// Values are mapped to Go types as follows: strings to string, integers to
// int64, floats to float64, booleans to bool, date-times to time.Time, arrays
// to []any, and tables (including inline tables) to map[string]any.
// int64, floats to float64, booleans to bool, offset date-times to
// OffsetDateTime, the local date-time kinds to their wrappers, arrays to
// []any, and tables (including inline tables) to map[string]any.
//
// Parse is equivalent to ParseContext with context.Background.
func Parse(data []byte) (map[string]any, error) {
func Parse(data []byte) (*Document, error) {
return ParseContext(context.Background(), data)
}
// ParseContext decodes a TOML document into a nested map[string]any, obeying
// ctx. The context is checked between top-level statements so cancellation is
// honoured before the parser has done substantial work.
func ParseContext(ctx context.Context, data []byte) (map[string]any, error) {
if err := ctx.Err(); err != nil {
return nil, err
// ParseContext decodes a TOML document into a Document, obeying ctx. The
// context is checked between top-level statements so cancellation is honoured
// before the parser has done substantial work.
func ParseContext(ctx context.Context, data []byte) (*Document, error) {
_, doc, err := parseWithOptions(ctx, data, parseOptions{}, true)
return doc, err
}
// ParseMap decodes a TOML document into a nested map[string]any, the value
// tree without the order and the comments a Document carries. It is the shape
// this package parsed into before [Document] existed.
//
// ParseMap is equivalent to ParseMapContext with context.Background.
func ParseMap(data []byte) (map[string]any, error) {
return ParseMapContext(context.Background(), data)
}
// ParseMapContext is the cancellable variant of ParseMap.
func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error) {
tree, _, err := parseWithOptions(ctx, data, parseOptions{}, false)
return tree, err
}
// ParseFile reads the TOML document at path and parses it into a Document,
// the shape Parse gives. Every error names the file it came from: a read
// failure and a parse failure alike carry the path as their first words,
// wrapped so errors.AsType still reaches the SyntaxError inside.
func ParseFile(path string) (*Document, error) {
data, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
if !utf8.Valid(data) {
return nil, &SyntaxError{Line: 1, Msg: "input is not valid UTF-8"}
doc, err := Parse(data)
if err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
return doc, nil
}
// Valid reports whether data is a valid TOML document: nil when the parser
// accepts it, and the parse error when it does not. It is the library call
// the --validate mode of interpres-decode is built on, and it reads nothing
// but the bytes it is given.
func Valid(data []byte) error {
_, err := ParseMapContext(context.Background(), data)
return err
}
// parseOptions bound the work one parse may do and the shape it produces. A
// zero field takes the default.
type parseOptions struct {
maxDepth int
maxInputSize int
useNumber bool
}
// parseWithOptions parses data, building the node tree of a Document when
// wantDoc asks for it, and returns both the value tree and that document.
func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantDoc bool) (map[string]any, *Document, error) {
if err := ctx.Err(); err != nil {
return nil, nil, err
}
if opts.maxInputSize > 0 && len(data) > opts.maxInputSize {
return nil, nil, fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), opts.maxInputSize)
}
// UTF-8 validity is not checked in a pass of its own: the scanner
// validates the multi-byte sequences where it meets them, so an invalid
// byte is reported on its own line instead of always on line 1.
maxDepth := opts.maxDepth
if maxDepth <= 0 {
maxDepth = maxNestingDepth
}
// The parser scans data in place; it only reads the buffer, and every
// string it stores in the tree is copied out of it.
p := &parser{src: data, line: 1, ctx: ctx}
return p.parse()
p := &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, wantDoc: wantDoc, useNumber: opts.useNumber}
tree, err := p.parse()
if err != nil {
return nil, nil, err
}
if !wantDoc {
return tree, nil, nil
}
return tree, &Document{root: p.doc, footer: p.footer}, nil
}
// Unmarshal parses a TOML document and stores the result in the value pointed
// to by v. v is typically a pointer to a struct or to a map[string]any.
// Options tune the call; with none, unknown keys are ignored, numbers are
// evaluated, and the nesting default applies.
//
// Struct fields are matched to TOML keys by the `toml:"name"` tag, or by a
// case-insensitive match on the field name when no tag is present. A tag of
// "-" skips the field.
// "-" skips the field. Two document keys that differ only in case and both
// match one field resolve deterministically: the lexicographically greater
// one wins, the same key winning every run.
//
// A destination implementing Unmarshaler receives the parsed value as it is,
// a TOML string fills a destination implementing encoding.TextUnmarshaler, and
// a time.Duration destination takes a duration literal such as `1h30m` or a
// bare integer as its nanosecond count.
//
// Unmarshal is equivalent to UnmarshalContext with context.Background.
func Unmarshal(data []byte, v any) error {
return UnmarshalContext(context.Background(), data, v)
func Unmarshal(data []byte, v any, opts ...UnmarshalOption) error {
return UnmarshalContext(context.Background(), data, v, opts...)
}
// ParseAs decodes a TOML document into T in one call, the generic shorthand
// for Unmarshal with a destination variable:
//
// cfg, err := interpres.ParseAs[Config](data, interpres.RejectUnknownFields(true))
//
// The options are Unmarshal's. The zero T comes back with the error.
func ParseAs[T any](data []byte, opts ...UnmarshalOption) (T, error) {
var v T
err := Unmarshal(data, &v, opts...)
return v, err
}
// NewSchema precompiles the codec for T: the struct schema both directions
// walk and the interface flags the decoder and the encoder resolve through
// are built once and cached, so the first document pays the cost instead of
// the hot path. A T that is not a struct warms nothing; there is nothing to
// precompute for a map or a slice.
func NewSchema[T any]() {
t := reflect.TypeFor[T]()
if t.Kind() != reflect.Struct {
return
}
cachedStructSchema(t)
_ = typeFlags(t)
_ = encTypeFlags(t)
pt := reflect.PointerTo(t)
_ = typeFlags(pt)
_ = encTypeFlags(pt)
}
// UnmarshalContext is the cancellable variant of Unmarshal.
func UnmarshalContext(ctx context.Context, data []byte, v any) error {
tree, err := ParseContext(ctx, data)
if err != nil {
return err
func UnmarshalContext(ctx context.Context, data []byte, v any, opts ...UnmarshalOption) error {
return settingsFor(opts).decode(ctx, data, v)
}
// UnmarshalRead reads the document from r and decodes it into v, the
// streaming-shaped entry the json/v2 vocabulary uses. The reader is
// consumed in full, because the parser scans its source in place; with
// MaxInputSize set, reading stops one byte past the limit so the size the
// option bounds is the memory held, not what a reader is drained into first.
// The options and the behaviour are Unmarshal's.
func UnmarshalRead(r io.Reader, v any, opts ...UnmarshalOption) error {
s := settingsFor(opts)
var data []byte
var err error
if s.maxInputSize > 0 {
data, err = io.ReadAll(io.LimitReader(r, int64(s.maxInputSize)+1))
} else {
data, err = io.ReadAll(r)
}
return newDecoder().decode(tree, v)
if err != nil {
return fmt.Errorf("interpres: read: %w", err)
}
return Unmarshal(data, v, opts...)
}
// A Decoder decodes a TOML document into a Go value with configurable
// strictness.
type Decoder struct {
disallowUnknown bool
}
// NewDecoder returns a Decoder.
func NewDecoder() *Decoder { return &Decoder{} }
// DisallowUnknownFields causes Decode to return an error when the document
// contains a key with no matching destination struct field.
func (d *Decoder) DisallowUnknownFields() *Decoder {
d.disallowUnknown = true
return d
}
// Decode parses data and stores the result in the value pointed to by v,
// honouring the decoder's strictness settings.
// An UnmarshalOption configures one Unmarshal, UnmarshalContext,
// UnmarshalRead or ParseAs call. Options are function values over the
// private decode settings, the shape encoding/json/v2 uses for its own
// options, and compose by simple listing:
//
// Decode is equivalent to DecodeContext with context.Background.
func (d *Decoder) Decode(data []byte, v any) error {
return d.DecodeContext(context.Background(), data, v)
// err := interpres.Unmarshal(data, &cfg,
// interpres.RejectUnknownFields(true),
// interpres.NumbersAsLiterals(true))
//
// A destination that the direct skeleton cannot model falls back to the
// tree path, so every option means the same thing on every document.
type UnmarshalOption func(*decodeSettings)
// decodeSettings is the option carrier of one decode call. The context is
// not one: it arrives as its own argument, because every entry point names it
// explicitly.
type decodeSettings struct {
disallowUnknown bool
useNumber bool
maxDepth int
maxInputSize int
localLoc *time.Location
}
// DecodeContext is the cancellable variant of Decode.
func (d *Decoder) DecodeContext(ctx context.Context, data []byte, v any) error {
tree, err := ParseContext(ctx, data)
func settingsFor(opts []UnmarshalOption) *decodeSettings {
s := &decodeSettings{}
for _, opt := range opts {
opt(s)
}
return s
}
// decode runs the decode the settings describe: the targeted parse when the
// destination takes it, the tree path otherwise or on fallback.
func (s *decodeSettings) decode(ctx context.Context, data []byte, v any) error {
dec := newDecoder()
dec.disallowUnknown = s.disallowUnknown
dec.ctx = ctx
dec.loc = s.localLoc
if canTargetDecode(v) {
// The targeted parse fills struct destinations without the
// intermediate tree; a document or destination it cannot model falls
// back to the tree path, whose contracts it keeps. The size limit is
// checked here, the targeted parse being the parse itself.
if s.maxInputSize > 0 && len(data) > s.maxInputSize {
return fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), s.maxInputSize)
}
if err := parseIntoTargeted(ctx, data, dec, s.useNumber, s.maxDepth, v); err != errTargetFallback {
return err
}
}
opts := parseOptions{
maxDepth: s.maxDepth,
maxInputSize: s.maxInputSize,
useNumber: s.useNumber,
}
tree, doc, err := parseWithOptions(ctx, data, opts, typeWantsOrder(reflect.TypeOf(v)))
if err != nil {
return err
}
dec := newDecoder()
dec.disallowUnknown = d.disallowUnknown
dec.nodes = indexNodes(doc.Root())
return dec.decode(tree, v)
}
// RejectUnknownFields makes the decode fail when the document contains a
// key with no matching destination struct field. Off by default: unknown
// keys are ignored.
func RejectUnknownFields(v bool) UnmarshalOption {
return func(s *decodeSettings) { s.disallowUnknown = v }
}
// NumbersAsLiterals keeps the numbers of the document as a Number carrying
// the literal the document wrote, so 0x1f, 1_000, +1.0 and inf survive a
// round trip with their spelling intact. A destination of a concrete numeric
// kind still takes the evaluated value; the literal is kept only where a
// Number, or an any, receives it. Off by default: numbers evaluate to
// int64 and float64.
func NumbersAsLiterals(v bool) UnmarshalOption {
return func(s *decodeSettings) { s.useNumber = v }
}
// LocalTimeLocation sets the zone a local date-time is placed in when it
// decodes into a time.Time destination. Without the option a local date-time
// fills only its own wrapper type (LocalDateTime, LocalDate, LocalTime),
// whose embedded time.Time is UTC; with the option, a time.Time destination
// takes the value too, carried in the location given. A nil location restores
// the default.
func LocalTimeLocation(loc *time.Location) UnmarshalOption {
return func(s *decodeSettings) { s.localLoc = loc }
}
// MaxNestingDepth bounds how deeply arrays and inline tables may nest in a
// document the decode accepts. The parser is a recursive descent, so a
// document that nests without bound would exhaust the stack; one that nests
// deeper than the limit is rejected with a SyntaxError naming it instead.
// Use 0 or any negative value for the default of 10000, which no
// hand-written document approaches.
func MaxNestingDepth(depth int) UnmarshalOption {
return func(s *decodeSettings) { s.maxDepth = depth }
}
// MaxInputSize bounds the size of a document the decode accepts, in bytes; a
// larger one is rejected before parsing starts. Use 0 or any negative value
// for no limit, which is the default: the caller already holds the bytes, so
// the size is a policy the caller sets rather than a protection the library
// imposes on its own.
func MaxInputSize(size int) UnmarshalOption {
return func(s *decodeSettings) { s.maxInputSize = size }
}
// Marshaler is the interface implemented by types that can produce a custom
// TOML representation of themselves. MarshalTOML returns a value that Marshal
// then encodes as if the returned value had been passed in its place, which
// is useful for emitting a Go type as a different TOML shape (for example, a
// struct as an inline table or a primitive alias as a richer value).
//
// MarshalTOML wins over encoding.TextMarshaler when a type implements both.
// A type that implements only encoding.TextMarshaler is encoded as a TOML
// string holding its text, and needs no method here.
type Marshaler interface {
MarshalTOML() (any, error)
}
@@ -184,32 +461,53 @@ type Marshaler interface {
// Unmarshaler is the inverse of Marshaler: a type that wants control over
// how it is decoded from a TOML value may implement UnmarshalTOML. The data
// argument is whatever the parser produced for that key: one of string,
// bool, int64, float64, time.Time, LocalDateTime, LocalDate, LocalTime,
// []any, or map[string]any. UnmarshalTOML may parse, inspect, or transform
// the value however it likes, then store the result by mutating its
// receiver through the standard pointer-indirection rules of the reflect
// package (i.e. via reflect.Value.Set or by reassigning fields through a
// pointer the receiver holds).
// bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime,
// []any, or map[string]any. A tree built by hand may carry a plain time.Time
// where the parser would put an OffsetDateTime, and NumbersAsLiterals a
// Number.
//
// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the
// UnmarshalTOML may parse, inspect, or transform the value however it likes,
// then store the result by mutating its receiver through the standard
// pointer-indirection rules of the reflect package (i.e. via
// reflect.Value.Set or by reassigning fields through a pointer the receiver
// holds).
//
// UnmarshalTOML is invoked from Unmarshal and its siblings when the
// destination type implements the interface. The decoder does not need to
// consult the concrete return value; whatever the receiver stores is kept.
//
// UnmarshalTOML wins over encoding.TextUnmarshaler when a type implements
// both. A type that implements only encoding.TextUnmarshaler is filled from a
// TOML string holding its text, and needs no method here.
type Unmarshaler interface {
UnmarshalTOML(data any) error
}
// Marshal returns the TOML encoding of v. The output stays within TOML 1.0,
// so it is valid under both TOML 1.0 and 1.1.
// UnmarshalerContext is Unmarshaler with the decode's context handed in. A
// type that implements both interfaces gets UnmarshalTOMLContext, so a long
// custom decode can abort on cancellation instead of running to completion.
// The context a non-cancellable entry point carries is context.Background,
// never nil.
type UnmarshalerContext interface {
UnmarshalTOMLContext(ctx context.Context, data any) error
}
// Marshal returns the TOML encoding of v. The output is valid TOML 1.1.
// Options tune the emission; with none, the layout groups entries by kind,
// empty arrays emit and sub-tables take the header form.
//
// Marshal traverses v using reflection and applies the following rules:
//
// - The top-level value must be a struct or a map[string]V. Pointers are
// followed; a nil top-level pointer is an error.
// - The top-level value must be a struct, a map[string]V or an OrderedMap
// (or a non-nil pointer to one). A Document writes itself back, and a nil
// one is an error.
// - Struct fields are matched by `toml:"name"` tag (case-insensitive
// fallback to field name; `-` skips). The tag options `omitzero` (skip
// the zero value of the field's type) and `omitempty` (skip an empty
// slice, array, or map) drop a field from the output on encode; the
// decoder ignores them. Anonymous (embedded) fields without a tag are
// fallback to field name; `-` skips). The tag option `omitzero` skips a
// field holding the zero value of its type (a type with an IsZero method
// decides through it), and `omitempty` skips a value that is empty in
// the encoding/json sense: an empty string, a zero number, false, a nil
// pointer or interface, and an empty slice, array or map. The decoder
// ignores both options. Anonymous (embedded) fields without a tag are
// inlined.
// - Maps use sorted keys for deterministic output.
// - Slices and arrays of structs or maps become TOML arrays of tables; a
@@ -219,95 +517,239 @@ type Unmarshaler interface {
// value array (for example an inline table in a mixed array) emits as an
// inline table.
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
// (offset date-time), and LocalDateTime/LocalDate/LocalTime (local
// variants).
// and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/
// LocalTime (local variants). A date-time writes its seconds only when
// the value carries them, and drops the trailing zeros of a fractional
// second. A zone offset that is not a whole number of minutes is refused,
// because TOML has no form that carries its seconds.
// - A table element of a value array, and a sub-table the InlineTables
// option inlines, is written as an inline table, across lines when it
// does not fit one.
// - Values implementing Marshaler are encoded by calling MarshalTOML and
// using its result.
// - Values implementing encoding.TextMarshaler, and not one of the
// date-time types, encode as a TOML string holding the text the method
// returns. time.Duration is written in its canonical Go form, `1h30m0s`.
// - nil pointer fields are omitted.
//
// Marshal cannot encode cyclic data structures; passing one will loop until
// the stack overflows. The output is not guaranteed to be byte-identical to
// the input that produced v: comments, whitespace, key order (for maps),
// string quoting style, and the choice between `[table]` headers and inline
// tables are not preserved.
// Marshal rejects a value that nests deeper than 10000 levels with an error
// naming the limit, so cyclic data is reported instead of running the stack
// out. The output is not guaranteed to be byte-identical to the input that
// produced v: comments, whitespace, key order (for maps), string quoting
// style, and the choice between `[table]` headers and inline tables are not
// preserved.
//
// Marshal is equivalent to MarshalContext with context.Background.
func Marshal(v any) ([]byte, error) {
return MarshalContext(context.Background(), v)
func Marshal(v any, opts ...MarshalOption) ([]byte, error) {
return MarshalContext(context.Background(), v, opts...)
}
// A Statement is one top-level statement of a document, what Statements
// yields: a key with its value, a table with its node, or one element of an
// array of tables with its node.
type Statement struct {
// Key is the key as the document wrote it.
Key string
// Value is the value of a key/value statement, and the value map of a
// table statement.
Value any
// Table is the node of a table or array-of-tables statement, carrying the
// written key order and the comments; nil for a plain key/value.
Table *Table
// Index is the element's position when the statement is one element of an
// array of tables, and -1 otherwise.
Index int
}
// Statements reads a TOML document from r and returns an iterator over its
// top-level statements in written order: key/value statements, including a
// value that is an array or an inline table, a [table] header as one
// statement carrying its Table node, and an [[array of tables]] as one
// statement per element, each with the element's node and its Index.
// Iteration stops at the first error, which arrives as the second value, and
// at a false yield: a caller that breaks after the statement it wanted reads
// no further ones.
//
// The reader is consumed in full before the first statement is yielded,
// because the parser scans the source in place; processing the yielded
// statements one at a time is what bounds what the caller holds, and a
// later direct-to-target parse removes the whole-source hold.
func Statements(r io.Reader) iter.Seq2[Statement, error] {
return func(yield func(Statement, error) bool) {
data, err := io.ReadAll(r)
if err != nil {
yield(Statement{Index: -1}, err)
return
}
doc, err := Parse(data)
if err != nil {
yield(Statement{Index: -1}, err)
return
}
for _, e := range doc.Root().Entries() {
// Only an array of tables yields per element, the branch the
// write side takes too: a value array is one statement whatever
// its elements, and an emptied array of tables holds no element
// to yield.
if _, isTables := e.Value().([]map[string]any); isTables && len(e.Elements()) > 0 {
for i, el := range e.Elements() {
if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: el, Index: i}, nil) {
return
}
}
continue
}
if child := e.Table(); child != nil {
if !yield(Statement{Key: e.Key(), Value: e.Value(), Table: child, Index: -1}, nil) {
return
}
continue
}
if !yield(Statement{Key: e.Key(), Value: e.Value(), Index: -1}, nil) {
return
}
}
}
}
// MarshalAppend appends the TOML encoding of v to buf and returns the extended
// buffer, the shape json/v2's MarshalAppendTo and json's MarshalAppend have.
// A failed encoding leaves buf untouched and comes back with a nil slice.
func MarshalAppend(buf []byte, v any, opts ...MarshalOption) ([]byte, error) {
out, err := Marshal(v, opts...)
if err != nil {
return nil, err
}
return append(buf, out...), nil
}
// MarshalContext is the cancellable variant of Marshal.
func MarshalContext(ctx context.Context, v any) ([]byte, error) {
func MarshalContext(ctx context.Context, v any, opts ...MarshalOption) ([]byte, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
return NewEncoder().MarshalContext(ctx, v)
return settingsForEncode(opts).marshal(ctx, v)
}
// An Encoder encodes Go values into TOML.
//
// All options default to behaviour that preserves byte-for-byte compatibility
// with previous releases and passes the toml-test compliance suite:
//
// GroupByKind: true (scalars first, then tables, then arrays of tables)
// OmitEmptyArrays: false (a nil/empty []string slice emits [] as a value;
// a nil/empty []Item struct slice is still skipped)
// LiteralMultilineAt: 0 (always emit basic multi-line strings with
// escape sequences, never literal ones)
//
// Use the chainable option methods to opt out. The option state is private;
// callers that need the underlying knobs reach for the methods rather than
// reading or mutating fields.
type Encoder struct {
groupByKind bool // default true; set via (*Encoder).GroupByKind
omitEmptyArrays bool // default false; set via (*Encoder).OmitEmptyArrays
literalMultilineAt int // default 0; set via (*Encoder).UseLiteralMultiline
// MarshalWrite encodes v and writes the document to w, the streaming-shaped
// entry the json/v2 vocabulary uses. The options and the behaviour are
// Marshal's.
func MarshalWrite(w io.Writer, v any, opts ...MarshalOption) error {
out, err := Marshal(v, opts...)
if err != nil {
return err
}
if _, err := w.Write(out); err != nil {
return fmt.Errorf("interpres: write: %w", err)
}
return nil
}
// NewEncoder returns an Encoder with default options.
func NewEncoder() *Encoder { return &Encoder{groupByKind: true} }
// A LayoutKind names the layout the encoder writes a document's entries in.
type LayoutKind int
// GroupByKind toggles whether fields at the same TOML level are reordered
// into the group-by-kind layout (scalars first, then tables, then arrays of
// tables). When set to false, the emitter preserves the source declaration
// order (struct field order, or sorted key order for maps).
func (e *Encoder) GroupByKind(v bool) *Encoder {
e.groupByKind = v
return e
const (
// LayoutKindGrouped reorders entries at one level: scalars first, then
// sub-tables, then arrays of tables. The default.
LayoutKindGrouped LayoutKind = iota
// LayoutKindDeclaration preserves the declaration order: struct field
// order, or sorted key order for maps.
LayoutKindDeclaration
)
// A MarshalOption configures one Marshal, MarshalContext, MarshalAppend or
// MarshalWrite call. Options are function values over the private encode
// settings, the shape encoding/json/v2 uses for its own, and compose by
// simple listing:
//
// out, err := interpres.Marshal(cfg,
// interpres.Layout(interpres.LayoutKindDeclaration),
// interpres.InlineTables(60))
type MarshalOption func(*encodeSettings)
// encodeSettings is the option carrier of one encode call. As on the decode
// side, the context arrives as its own argument.
type encodeSettings struct {
cfg encodeConfig
}
func settingsForEncode(opts []MarshalOption) *encodeSettings {
s := &encodeSettings{cfg: encodeConfig{layout: LayoutKindGrouped}}
for _, opt := range opts {
opt(s)
}
return s
}
// marshal runs the encode the settings describe.
func (s *encodeSettings) marshal(ctx context.Context, v any) ([]byte, error) {
enc := newEncoder()
enc.ctx = ctx
enc.opts = s.cfg
if err := enc.encode(v); err != nil {
enc.release()
return nil, err
}
// The output leaves the pooled buffer as a copy, so the next Marshal
// reuses the buffer without touching what the caller holds.
out := slices.Clone(enc.buf.Bytes())
enc.release()
return out, nil
}
// Layout sets the layout the encoder writes a document's entries in:
// LayoutKindGrouped, the default, reorders them scalars first, then tables,
// then arrays of tables; LayoutKindDeclaration preserves declaration order.
// A value the two constants do not name behaves as LayoutKindGrouped.
func Layout(kind LayoutKind) MarshalOption {
return func(s *encodeSettings) { s.cfg.layout = kind }
}
// OmitEmptyArrays opts in to skipping empty (non-nil, length 0) TOML arrays
// of scalars. The default emits them as "key = []". Nil slices and empty
// arrays of tables are already always omitted.
func (e *Encoder) OmitEmptyArrays() *Encoder {
e.omitEmptyArrays = true
return e
func OmitEmptyArrays(v bool) MarshalOption {
return func(s *encodeSettings) { s.cfg.omitEmptyArrays = v }
}
// UseLiteralMultiline sets the length threshold at which a multi-line string
// LiteralMultiline sets the length threshold at which a multi-line string
// is emitted as a literal triple-quoted string instead of the escaped form.
// Use 0 or any negative value to disable (always escaped). The literal form
// is selected only when the value contains an internal newline; otherwise the
// single-line basic form is used regardless of this setting.
func (e *Encoder) UseLiteralMultiline(threshold int) *Encoder {
e.literalMultilineAt = threshold
return e
func LiteralMultiline(threshold int) MarshalOption {
return func(s *encodeSettings) { s.cfg.literalMultilineAt = threshold }
}
// Marshal encodes v to TOML bytes. It is equivalent to calling Marshal with v.
// InlineTables sets the size limit, in bytes of the single-line rendering, at
// which a sub-table is written as an inline table instead of a table header,
// which makes a document of small tables shorter. Use 0 or any negative value
// to disable (always emit a header).
//
// Marshal is equivalent to MarshalContext with context.Background.
func (e *Encoder) Marshal(v any) ([]byte, error) {
return e.MarshalContext(context.Background(), v)
// A sub-table is inlined only when doing so keeps every value's type: an array
// of tables keeps its header form, because its inline form would re-parse as a
// value array. An inlined table that does not fit the line is written across
// lines, which TOML 1.1 allows.
//
// With LayoutKindDeclaration the layout is already for presentation only, and an
// inlined table follows the same rule as any other value line: it lands in the
// section of the header that precedes it.
func InlineTables(threshold int) MarshalOption {
return func(s *encodeSettings) { s.cfg.inlineTablesAt = threshold }
}
// MarshalContext is the cancellable variant of Marshal.
func (e *Encoder) MarshalContext(ctx context.Context, v any) ([]byte, error) {
enc := newEncoder()
enc.ctx = ctx
enc.opts = *e
if err := enc.encode(v); err != nil {
return nil, err
}
return enc.bytes(), nil
// EmitFieldComments turns on printing the comment a field's `toml` tag
// carries in a `comment=` option, above the field's line or header, the
// comments a round trip through the Go type would otherwise drop:
//
// Port int `toml:"port,comment=The port to listen on"`
//
// Go doc comments are not visible to reflection, so the tag is the channel
// that carries the text. Off by default, and a field without a `comment=`
// option prints none. Multi-line comments carry newlines in the tag, each
// line printed with its own "# " marker. The tag's options separate with
// commas, so the comment text itself cannot carry one; the first comma ends
// it.
func EmitFieldComments(v bool) MarshalOption {
return func(s *encodeSettings) { s.cfg.emitFieldComments = v }
}
+352 -50
View File
@@ -4,14 +4,20 @@
package interpres
import (
"errors"
"fmt"
"math"
"os"
"path/filepath"
"reflect"
"slices"
"strings"
"testing"
"time"
)
func TestParseScalars(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
title = "interpres"
count = 42
ratio = 3.14
@@ -49,7 +55,7 @@ expv = 1e3
}
func TestParseInfNan(t *testing.T) {
tree, err := Parse([]byte("pos = inf\nneg = -inf\nbad = nan\n"))
tree, err := ParseMap([]byte("pos = inf\nneg = -inf\nbad = nan\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -65,7 +71,7 @@ func TestParseInfNan(t *testing.T) {
}
func TestParseStrings(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
basic = "a\tb\nc"
literal = 'C:\path\no\escape'
quote = "say \"hi\""
@@ -89,7 +95,7 @@ unicode = "\u00e9"
}
func TestParseMultilineString(t *testing.T) {
tree, err := Parse([]byte("text = \"\"\"\nfirst\nsecond\"\"\"\n"))
tree, err := ParseMap([]byte("text = \"\"\"\nfirst\nsecond\"\"\"\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -99,7 +105,7 @@ func TestParseMultilineString(t *testing.T) {
}
func TestParseMultilineLineEndingBackslash(t *testing.T) {
tree, err := Parse([]byte("text = \"\"\"\\\n one \\\n two\"\"\"\n"))
tree, err := ParseMap([]byte("text = \"\"\"\\\n one \\\n two\"\"\"\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -109,7 +115,7 @@ func TestParseMultilineLineEndingBackslash(t *testing.T) {
}
func TestParseTablesAndDottedKeys(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
owner.name = "Petr"
[server]
@@ -137,7 +143,7 @@ enabled = true
}
func TestParseArrayOfTables(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
[[forms]]
name = "contact"
@@ -157,7 +163,7 @@ name = "feedback"
}
func TestParseArraysAndInlineTables(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
ports = [80, 443]
mixed = [
"a",
@@ -183,7 +189,7 @@ point = { x = 1, y = 2 }
}
func TestParseDateTime(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
offset = 1979-05-27T07:32:00Z
local = 1979-05-27T07:32:00
day = 1979-05-27
@@ -192,7 +198,7 @@ clock = 07:32:00
if err != nil {
t.Fatalf("parse: %v", err)
}
if off, ok := tree["offset"].(time.Time); !ok || off.Year() != 1979 || off.Hour() != 7 {
if off, ok := tree["offset"].(OffsetDateTime); !ok || off.Year() != 1979 || off.Hour() != 7 {
t.Errorf("offset = %#v (%T)", tree["offset"], tree["offset"])
}
if ldt, ok := tree["local"].(LocalDateTime); !ok || ldt.Year() != 1979 || ldt.Hour() != 7 {
@@ -207,15 +213,15 @@ clock = 07:32:00
}
func TestDateTimeFormats(t *testing.T) {
tree, err := Parse([]byte("a = 1987-07-05 17:45:00Z\nb = 1987-07-05t17:45:00z\nc = 1977-12-21T10:32:00.555\n"))
tree, err := ParseMap([]byte("a = 1987-07-05 17:45:00Z\nb = 1987-07-05t17:45:00z\nc = 1977-12-21T10:32:00.555\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
if _, ok := tree["a"].(time.Time); !ok {
t.Errorf("a is %T, want time.Time", tree["a"])
if _, ok := tree["a"].(OffsetDateTime); !ok {
t.Errorf("a is %T, want OffsetDateTime", tree["a"])
}
if _, ok := tree["b"].(time.Time); !ok {
t.Errorf("b is %T, want time.Time", tree["b"])
if _, ok := tree["b"].(OffsetDateTime); !ok {
t.Errorf("b is %T, want OffsetDateTime", tree["b"])
}
if _, ok := tree["c"].(LocalDateTime); !ok {
t.Errorf("c is %T, want LocalDateTime", tree["c"])
@@ -311,7 +317,7 @@ func TestDisallowUnknownFields(t *testing.T) {
}
var strict C
err := NewDecoder().DisallowUnknownFields().Decode(data, &strict)
err := Unmarshal(data, &strict, RejectUnknownFields(true))
if err == nil {
t.Fatal("expected error for unknown field, got nil")
}
@@ -327,7 +333,7 @@ func TestDisallowUnknownFieldsReportsSmallestKey(t *testing.T) {
data := []byte("known = \"x\"\nzeta = 1\nalpha = 2\nmu = 3\n")
for range 20 {
var c C
err := NewDecoder().DisallowUnknownFields().Decode(data, &c)
err := Unmarshal(data, &c, RejectUnknownFields(true))
if err == nil {
t.Fatal("expected error for unknown fields")
}
@@ -352,7 +358,7 @@ func TestSkippedFieldTag(t *testing.T) {
}
func TestSyntaxErrorReportsLine(t *testing.T) {
_, err := Parse([]byte("a = 1\nb = \nc = 3\n"))
_, err := ParseMap([]byte("a = 1\nb = \nc = 3\n"))
if err == nil {
t.Fatal("expected a syntax error")
}
@@ -366,7 +372,7 @@ func TestSyntaxErrorReportsLine(t *testing.T) {
}
func TestComments(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
# a leading comment
key = "value" # trailing comment
# another
@@ -380,7 +386,7 @@ key = "value" # trailing comment
}
func TestDuplicateKeyRejected(t *testing.T) {
_, err := Parse([]byte("a = 1\na = 2\n"))
_, err := ParseMap([]byte("a = 1\na = 2\n"))
if err == nil {
t.Fatal("expected duplicate key error")
}
@@ -395,7 +401,7 @@ func TestRejectsInvalidNumbers(t *testing.T) {
"0x", "0o", "0b", "0b2", "0o8", "0xG",
"+0x1",
} {
if _, err := Parse([]byte("v = " + tok + "\n")); err == nil {
if _, err := ParseMap([]byte("v = " + tok + "\n")); err == nil {
t.Errorf("%q: expected an error, got none", tok)
}
}
@@ -408,22 +414,22 @@ func TestParseRejectsOffsetOutOfRange(t *testing.T) {
"1979-05-27T07:32:00+24:00",
"1979-05-27T07:32:00+99:99",
} {
if _, err := Parse([]byte("v = " + tok + "\n")); err == nil {
if _, err := ParseMap([]byte("v = " + tok + "\n")); err == nil {
t.Errorf("%q: expected an error, got none", tok)
}
}
}
func TestParseAcceptsOffsetBounds(t *testing.T) {
tree, err := Parse([]byte("a = 1979-05-27T07:32:00+23:59\nb = 1979-05-27T07:32:00-23:59\n"))
tree, err := ParseMap([]byte("a = 1979-05-27T07:32:00+23:59\nb = 1979-05-27T07:32:00-23:59\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
a := tree["a"].(time.Time)
a := tree["a"].(OffsetDateTime)
if _, offset := a.Zone(); offset != 23*3600+59*60 {
t.Fatalf("a offset = %d, want %d", offset, 23*3600+59*60)
}
b := tree["b"].(time.Time)
b := tree["b"].(OffsetDateTime)
if _, offset := b.Zone(); offset != -(23*3600 + 59*60) {
t.Fatalf("b offset = %d", offset)
}
@@ -449,7 +455,7 @@ func TestAcceptsNumberEdgeCases(t *testing.T) {
"-2.5E-3": -2.5e-3,
}
for tok, want := range cases {
tree, err := Parse([]byte("v = " + tok + "\n"))
tree, err := ParseMap([]byte("v = " + tok + "\n"))
if err != nil {
t.Errorf("%q: %v", tok, err)
continue
@@ -461,14 +467,14 @@ func TestAcceptsNumberEdgeCases(t *testing.T) {
}
func TestRejectsTableRedefinition(t *testing.T) {
_, err := Parse([]byte("[a]\nx = 1\n\n[a]\ny = 2\n"))
_, err := ParseMap([]byte("[a]\nx = 1\n\n[a]\ny = 2\n"))
if err == nil {
t.Fatal("expected a table-redefinition error")
}
}
func TestAllowsImplicitThenExplicitTable(t *testing.T) {
tree, err := Parse([]byte("[a.b]\nx = 1\n\n[a]\ny = 2\n"))
tree, err := ParseMap([]byte("[a.b]\nx = 1\n\n[a]\ny = 2\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -482,13 +488,13 @@ func TestAllowsImplicitThenExplicitTable(t *testing.T) {
}
func TestRejectsControlCharInString(t *testing.T) {
if _, err := Parse([]byte("v = \"a\x01b\"\n")); err == nil {
if _, err := ParseMap([]byte("v = \"a\x01b\"\n")); err == nil {
t.Fatal("expected a control-character error")
}
}
func TestAllowsEscapedControlChar(t *testing.T) {
tree, err := Parse([]byte(`v = "\u0000"`))
tree, err := ParseMap([]byte(`v = "\u0000"`))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -498,7 +504,7 @@ func TestAllowsEscapedControlChar(t *testing.T) {
}
func TestMultilineQuotesAtDelimiter(t *testing.T) {
tree, err := Parse([]byte("a = '''''two quotes'''''\n"))
tree, err := ParseMap([]byte("a = '''''two quotes'''''\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -517,7 +523,7 @@ func TestRejectsInlineTableExtension(t *testing.T) {
"by nested array header": "a = { b = {} }\n[[a.b.c]]\nx = 2\n",
}
for name, doc := range cases {
if _, err := Parse([]byte(doc)); err == nil {
if _, err := ParseMap([]byte(doc)); err == nil {
t.Errorf("%s: expected an inline-table extension error", name)
}
}
@@ -532,7 +538,7 @@ func TestArrayOfTablesFreshScopePerElement(t *testing.T) {
"dotted key": "[[a]]\nb.c = 1\n[[a]]\n[a.b]\nd = 2\n",
}
for name, doc := range cases {
tree, err := Parse([]byte(doc))
tree, err := ParseMap([]byte(doc))
if err != nil {
t.Errorf("%s: %v", name, err)
continue
@@ -547,7 +553,7 @@ func TestArrayOfTablesFreshScopePerElement(t *testing.T) {
"header over dotted in one element": "[[a]]\nb.c = 1\n[a.b]\nd = 2\n",
"table over nested array": "[[a]]\n[[a.b]]\n[a.b]\nx = 1\n",
} {
if _, err := Parse([]byte(doc)); err == nil {
if _, err := ParseMap([]byte(doc)); err == nil {
t.Errorf("%s: expected an error, got none", name)
}
}
@@ -565,14 +571,14 @@ func TestRejectsSpecInvalid(t *testing.T) {
// makes the seconds optional.
}
for name, doc := range cases {
if _, err := Parse([]byte(doc)); err == nil {
if _, err := ParseMap([]byte(doc)); err == nil {
t.Errorf("%s: expected an error", name)
}
}
}
func TestArrayOfTablesPerElementSubtable(t *testing.T) {
tree, err := Parse([]byte(`
tree, err := ParseMap([]byte(`
[[forms]]
name = "a"
@@ -603,7 +609,7 @@ host = "h2"
// --- TOML 1.1 --------------------------------------------------------------
func TestParseAcceptsNoSecondsDatetimes(t *testing.T) {
tree, err := Parse([]byte(`t = 13:37
tree, err := ParseMap([]byte(`t = 13:37
dt = 1979-05-27T07:32
odt1 = 1979-05-27 07:32Z
odt2 = 1979-05-27 07:32-07:00
@@ -611,26 +617,28 @@ odt2 = 1979-05-27 07:32-07:00
if err != nil {
t.Fatalf("parse: %v", err)
}
if got := tree["t"].(LocalTime).String(); got != "13:37:00" {
t.Errorf("t = %q, want %q", got, "13:37:00")
// A value written without seconds comes back without them: the seconds are
// only written when the value carries them.
if got := tree["t"].(LocalTime).String(); got != "13:37" {
t.Errorf("t = %q, want %q", got, "13:37")
}
if got := tree["dt"].(LocalDateTime).String(); got != "1979-05-27T07:32:00" {
t.Errorf("dt = %q, want %q", got, "1979-05-27T07:32:00")
if got := tree["dt"].(LocalDateTime).String(); got != "1979-05-27T07:32" {
t.Errorf("dt = %q, want %q", got, "1979-05-27T07:32")
}
if got := tree["odt1"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00Z" {
if got := tree["odt1"].(OffsetDateTime).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00Z" {
t.Errorf("odt1 = %q", got)
}
if got := tree["odt2"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00-07:00" {
if got := tree["odt2"].(OffsetDateTime).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00-07:00" {
t.Errorf("odt2 = %q", got)
}
// The fraction still requires the seconds it belongs to.
if _, err := Parse([]byte("a = 07:32.5\n")); err == nil {
if _, err := ParseMap([]byte("a = 07:32.5\n")); err == nil {
t.Error("07:32.5: expected an error, got none")
}
}
func TestParseAcceptsEscapeAndHexEscapes(t *testing.T) {
tree, err := Parse([]byte(`esc = "\e"
tree, err := ParseMap([]byte(`esc = "\e"
hex = "\x20\x7f\xf8"
nul = "\x00"
multi = """\x68\x65"""
@@ -657,14 +665,14 @@ lit = '\x20'
}
// Two digits exactly; a short or non-hex escape is an error.
for _, doc := range []string{`a = "\x4"`, `a = "\x"`, `a = "\xgg"`} {
if _, err := Parse([]byte(doc)); err == nil {
if _, err := ParseMap([]byte(doc)); err == nil {
t.Errorf("%s: expected an error, got none", doc)
}
}
}
func TestParseAcceptsMultilineInlineTables(t *testing.T) {
tree, err := Parse([]byte("tbl = {\n\thello = \"world\",\n\tarr = [1,\n\t\t2,\n\t],\n\tsub = {\n\t\tk = 1,\n\t},\n\tbare = 2}\n"))
tree, err := ParseMap([]byte("tbl = {\n\thello = \"world\",\n\tarr = [1,\n\t\t2,\n\t],\n\tsub = {\n\t\tk = 1,\n\t},\n\tbare = 2}\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
@@ -679,7 +687,7 @@ func TestParseAcceptsMultilineInlineTables(t *testing.T) {
t.Errorf("sub = %#v", tbl["sub"])
}
// Comments inside the table, and a trailing comma at both depths.
tree, err = Parse([]byte("m = { # one\n\t# two\n\ta = 1, # three\n\t# four\n}\n"))
tree, err = ParseMap([]byte("m = { # one\n\t# two\n\ta = 1, # three\n\t# four\n}\n"))
if err != nil {
t.Fatalf("parse with comments: %v", err)
}
@@ -687,7 +695,7 @@ func TestParseAcceptsMultilineInlineTables(t *testing.T) {
t.Errorf("m = %#v", m)
}
// The old single-line shapes keep working, with and without the comma.
if _, err := Parse([]byte("a = { b = 1, c = 2 }\n")); err != nil {
if _, err := ParseMap([]byte("a = { b = 1, c = 2 }\n")); err != nil {
t.Errorf("single line: %v", err)
}
// Still rejected: two commas, a missing value, and an unclosed table.
@@ -696,8 +704,302 @@ func TestParseAcceptsMultilineInlineTables(t *testing.T) {
"missing value": "a = {\n\tb =\n}\n",
"unterminated": "a = { b = 1,\n",
} {
if _, err := Parse([]byte(doc)); err == nil {
if _, err := ParseMap([]byte(doc)); err == nil {
t.Errorf("%s: expected an error, got none", name)
}
}
}
func TestParseNestingLimit(t *testing.T) {
// The parser is a recursive descent, so a document that nests without bound
// is rejected instead of exhausting the stack.
deep := func(n int) []byte {
return []byte("v = " + strings.Repeat("[", n) + strings.Repeat("]", n) + "\n")
}
if _, err := ParseMap(deep(100)); err != nil {
t.Fatalf("a document well inside the limit: %v", err)
}
_, err := ParseMap(deep(maxNestingDepth + 1))
if err == nil {
t.Fatal("expected a nesting error")
}
var se *SyntaxError
if !errors.As(err, &se) {
t.Fatalf("expected a *SyntaxError, got %T: %v", err, err)
}
if !strings.Contains(se.Msg, "nesting") {
t.Errorf("Msg = %q, want it to name the nesting limit", se.Msg)
}
}
func TestParseFile(t *testing.T) {
path := filepath.Join(t.TempDir(), "config.toml")
if err := os.WriteFile(path, []byte("port = 8080\n"), 0o644); err != nil {
t.Fatal(err)
}
doc, err := ParseFile(path)
if err != nil {
t.Fatal(err)
}
if got := doc.Map()["port"]; got != int64(8080) {
t.Errorf("port = %v, want 8080", got)
}
_, err = ParseFile(filepath.Join(t.TempDir(), "missing.toml"))
if err == nil || !strings.Contains(err.Error(), "missing.toml") {
t.Errorf("read error = %v, want it to name the file", err)
}
bad := filepath.Join(t.TempDir(), "broken.toml")
if err := os.WriteFile(bad, []byte("port =\n"), 0o644); err != nil {
t.Fatal(err)
}
_, err = ParseFile(bad)
if err == nil || !strings.Contains(err.Error(), "broken.toml") {
t.Errorf("parse error = %v, want it to name the file", err)
}
s, ok := errors.AsType[*SyntaxError](err)
if !ok || s.Line != 1 {
t.Errorf("parse error = %v, want a SyntaxError with line 1 inside", err)
}
}
func TestValid(t *testing.T) {
if err := Valid([]byte("a = 1\n[t]\nb = 2\n")); err != nil {
t.Errorf("Valid(valid) = %v, want nil", err)
}
err := Valid([]byte("a = \n"))
if err == nil {
t.Fatal("Valid(invalid) = nil, want an error")
}
if _, ok := errors.AsType[*SyntaxError](err); !ok {
t.Errorf("Valid(invalid) = %v, want a SyntaxError", err)
}
}
func TestParseAsAndNewSchema(t *testing.T) {
type Config struct {
Host string `toml:"host"`
Port int `toml:"port"`
}
cfg, err := ParseAs[Config]([]byte("host = \"db\"\nport = 5432\n"))
if err != nil {
t.Fatal(err)
}
if cfg.Host != "db" || cfg.Port != 5432 {
t.Errorf("decoded %+v", cfg)
}
if _, err := ParseAs[Config]([]byte("port =\n")); err == nil {
t.Error("ParseAs(invalid) succeeded, want an error and the zero value")
}
NewSchema[Config]()
if _, ok := structSchemaCache.Load(reflect.TypeFor[Config]()); !ok {
t.Error("NewSchema left no schema in the cache")
}
NewSchema[map[string]any]() // must not panic
}
func TestZeroOffsetRoundTrip(t *testing.T) {
// A document may write a zero offset as +00:00; the tree must hold the
// same value after a round trip, because the written form is "Z" either
// way.
src := []byte("a = 1979-05-27T07:32:00+00:00\n")
tree, err := ParseMap(src)
if err != nil {
t.Fatal(err)
}
out, err := Marshal(tree)
if err != nil {
t.Fatal(err)
}
re, err := ParseMap(out)
if err != nil {
t.Fatal(err)
}
if !reflect.DeepEqual(tree, re) {
t.Errorf("round trip changed the tree: %#v vs %#v", tree, re)
}
if got := tree["a"].(OffsetDateTime).String(); got != "1979-05-27T07:32Z" {
t.Errorf("a = %q, want 1979-05-27T07:32Z", got)
}
}
func TestStatements(t *testing.T) {
src := strings.NewReader(`title = "demo"
port = 8080
[server]
host = "127.0.0.1"
[[items]]
name = "a"
[[items]]
name = "b"
`)
var lines []string
for stmt, err := range Statements(src) {
if err != nil {
t.Fatal(err)
}
switch {
case stmt.Index >= 0:
lines = append(lines, fmt.Sprintf("%s #%d", stmt.Key, stmt.Index))
case stmt.Table != nil:
lines = append(lines, fmt.Sprintf("[%s] %v", stmt.Key, stmt.Table.Keys()))
default:
lines = append(lines, fmt.Sprintf("%s = %v", stmt.Key, stmt.Value))
}
}
want := []string{
`title = demo`,
`port = 8080`,
`[server] [host]`,
`items #0`,
`items #1`,
}
if !slices.Equal(lines, want) {
t.Errorf("statements =\n%v\nwant:\n%v", lines, want)
}
t.Run("breaking stops the iteration", func(t *testing.T) {
src := strings.NewReader("a = 1\nb = 2\nc = 3\n")
count := 0
for range Statements(src) {
count++
break
}
if count != 1 {
t.Errorf("iterated %d statements after break, want 1", count)
}
})
t.Run("a parse error arrives as the second value", func(t *testing.T) {
for stmt, err := range Statements(strings.NewReader("broken =\n")) {
if err == nil {
t.Fatalf("statement %+v without an error", stmt)
}
if _, ok := errors.AsType[*SyntaxError](err); !ok {
t.Errorf("err = %v, want a SyntaxError", err)
}
break
}
})
}
func TestParseCRLFDocument(t *testing.T) {
tree, err := ParseMap([]byte("a = 1\r\nb = 2\r\n[t]\r\nc = \"x\"\r\n"))
if err != nil {
t.Fatal(err)
}
if tree["a"] != int64(1) || tree["b"] != int64(2) {
t.Errorf("tree = %v", tree)
}
}
// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U:
// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the
// greatest scalar value parses.
func TestParseUnicodeEscapeBoundaries(t *testing.T) {
bad := []struct {
name string
in string
}{
{"high surrogate", `a = "\ud800"`},
{"low surrogate", `a = "\udfff"`},
{"past the greatest scalar", `a = "\U00110000"`},
{"signed short escape", `a = "\u+041"`},
{"negative long escape", `a = "\U-0000001"`},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := Parse([]byte(tt.in))
if err == nil {
t.Fatalf("Parse accepted %q", tt.in)
}
})
}
tree, err := ParseMap([]byte("a = \"\\U0010FFFF\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "􏿿" {
t.Errorf("a = %q", tree["a"])
}
}
// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a
// date-time with a component out of range is rejected as a date-time, not
// left to the number decoder's complaint.
func TestParseRejectsOutOfRangeDateTimes(t *testing.T) {
bad := []struct {
name string
in string
}{
{"hour 24", "a = 1979-05-27T24:00:00Z"},
{"minute 60", "a = 1979-05-27T07:60:00Z"},
{"second 60", "a = 1979-05-27T07:32:60Z"},
{"month 13", "a = 1979-13-27T07:32:00Z"},
{"day 32", "a = 1979-05-32T07:32:00Z"},
{"february the thirtieth", "a = 1979-02-30"},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseMap([]byte(tt.in))
if err == nil {
t.Fatalf("ParseMap accepted %q", tt.in)
}
if !strings.Contains(err.Error(), "invalid date-time") {
t.Errorf("err = %v, want the date-time complaint", err)
}
})
}
}
// TestParseMultilineStringEdges pins the carriage-return and delimiter rules
// of multi-line strings: a bare CR right after the opening delimiter is the
// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the
// content survives.
func TestParseMultilineStringEdges(t *testing.T) {
_, err := ParseMap([]byte("a = \"\"\"\rX\"\"\""))
if err == nil || !strings.Contains(err.Error(), "bare carriage return") {
t.Errorf("err = %v, want the bare-CR error after the delimiter", err)
}
tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "X\r\nY" {
t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"])
}
}
// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes
// before the closing delimiter of a basic multi-line string are content, and
// more than five are the error.
func TestParseMultilineBasicDelimiterRuns(t *testing.T) {
tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != `end"` {
t.Errorf("a = %q", tree["a"])
}
_, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\""))
if err == nil || !strings.Contains(err.Error(), "too many") {
t.Errorf("err = %v, want the too-many-delimiters error", err)
}
}
// TestParseLineEndingBackslashEdges pins the line-ending backslash at the
// very end of the input and before a bare CR.
func TestParseLineEndingBackslashEdges(t *testing.T) {
bad := []string{
"a = \"\"\"x \\\\",
"a = \"\"\"x \\\\\rZ\"\"\"",
}
for _, in := range bad {
if _, err := ParseMap([]byte(in)); err == nil {
t.Errorf("ParseMap accepted %q", in)
}
}
}
+43 -4
View File
@@ -1,7 +1,7 @@
# interpres.
#
# Everything below the variable block is the standard recipe set from the `justfile`
# skill, identical in every repository; project values live in the variable block only.
# Everything below the variable block is the standard recipe set, identical in
# every repository; project values live in the variable block only.
binary := "interpres-decode"
package := "./cmd/interpres-decode"
@@ -92,14 +92,53 @@ run:
dev:
go run -buildvcs=true {{package}}
# Runs the official toml-test compliance suite against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
# Runs the official toml-test compliance suite in both directions, decoder and encoder, against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
toml-test: build
toml-test test -decoder=bin/interpres-decode -toml=1.1
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
coverage-html: test
go tool cover -html=coverage.out -o coverage.html
# Cross-compile smoke: the library and the command build for the foreign architectures and the browser and edge runtimes; not a gate, it is a hand-run convenience and runs std-lib only.
cross:
GOARCH=arm64 go build ./...
GOARCH=loong64 go build ./...
GOARCH=riscv64 go build ./...
GOOS=js GOARCH=wasm go build ./...
GOOS=wasip1 GOARCH=wasm go build ./...
GOARCH=arm64 CGO_ENABLED=0 go build -o /dev/null {{package}}
GOARCH=loong64 CGO_ENABLED=0 go build -o /dev/null {{package}}
GOARCH=riscv64 CGO_ENABLED=0 go build -o /dev/null {{package}}
# Runs the example program under examples/basic; not standard because `run` runs the adapter, and an example is documentation, not the product.
example:
go run ./examples/basic
# The release pre-flight, in one command: the branch, a clean tree, a sync with origin, the gates, and a CHANGELOG section ready to release. Not a gate, it is the checklist before a release may even be discussed.
release-check version:
#!/usr/bin/env perl
# The version arrives through the recipe interpolation: just does not hand
# positional arguments to a shebang script's @ARGV.
my $version = "{{version}}";
$version =~ m{\Av?\d+\.\d+\.\d+\z} or die qq{usage: just release-check X.Y.Z\n};
my $branch = qx{git rev-parse --abbrev-ref HEAD};
chomp $branch;
$branch eq q{development} or die qq{release-check: on '$branch', cut releases from development\n};
my $dirty = qx{git status --porcelain};
$dirty eq q{} or die qq{release-check: the working tree is dirty\n};
system(qw{git fetch origin}) == 0 or die qq{release-check: git fetch failed\n};
my $local = qx{git rev-parse development};
my $remote = qx{git rev-parse origin/development};
$local eq $remote or die qq{release-check: development is out of sync with origin\n};
my $changelog = do { open(my $fh, q{<}, q{CHANGELOG.md}) or die qq{release-check: cannot read CHANGELOG.md: $!\n}; local $/; <$fh> };
$changelog =~ m{## \[development\]\n\n### \w+} or die qq{release-check: the [development] section of CHANGELOG.md is missing or empty\n};
print qq{branch, tree, sync and changelog verified; running the gates\n};
system(qw{just gates}) == 0 or die qq{release-check: the gates failed\n};
print qq{release-check: ready to release $version\n};
print qq{after tagging, verify the /v2 module resolves through the proxy:\n};
print qq{ cd \$(mktemp -d) && go mod init t && GOPRIVATE= GOPROXY=https://proxy.golang.org go get sourcedock.dev/petrbalvin/interpres/v2\@$version\n};
# Compares the toml-test counts the documentation names with the live suite run; not standard, it exists because a corpus change used to be corrected by hand.
docs-drift:
perl scripts/docs-drift.pl
+140
View File
@@ -0,0 +1,140 @@
.TH INTERPRES-DECODE 1 "2026-09-22" "interpres 2.0.0" "User Commands"
.SH NAME
interpres-decode \- TOML validator and toml-test harness adapter
.SH SYNOPSIS
.B interpres-decode
[\fIFLAGS\fR]
.br
.B interpres-decode
.B \-\-encode
.br
.B interpres-decode
.B \-\-validate
[\fIFILE\fR...]
.br
.B interpres-decode
.B \-\-validate
[\fIDIRECTORY\fR...]
.br
.B interpres-decode
.B \-\-json
.br
.B interpres-decode
.B \-\-struct
.br
.B interpres-decode
.B \-\-schema
\fITYPE\fR
\fIFILE.go\fR
.br
.B interpres-decode
.B \-\-version
.SH DESCRIPTION
.B interpres-decode
is the toml-test harness adapter in both directions and a TOML validator.
Without a mode flag it reads one TOML document from standard input and writes
the toml-test tagged-JSON representation to standard output.
.B \-\-encode
reads a tagged-JSON description from standard input and writes the TOML
document it describes.
.B \-\-validate
parses each named file, or standard input when none are named, and prints one
line per invalid document to standard error; a named directory is walked for
.B .toml
files, every one validated, and the walk closes with a summary on standard
error naming the counts. The name
.B \-
means standard input.
.B \-\-json
prints plain indented JSON instead of the tagged form; it shapes the decoding
output only, so it is rejected together with the mode flags.
.B \-\-struct
prints a Go struct definition inferred from the document on standard input.
.B \-\-schema
writes a TOML template for the struct type
\fITYPE\fR
declared in the Go source file
\fIFILE.go\fR,
taking the key names, comments and defaults from the fields' tags.
.B \-\-version
prints the binary's version and exits.
.PP
The mode flags
.BR \-\-validate ,
.BR \-\-encode ,
.B \-\-struct
and
.B \-\-schema
cannot be combined.
.SH OPTIONS
.TP
.B \-\-validate
Validate the documents instead of emitting tagged JSON.
.TP
.B \-\-encode
Read tagged JSON from standard input and write TOML instead.
.TP
.B \-\-json
With the default mode, print plain indented JSON instead of tagged JSON.
.TP
.B \-\-struct
Infer a Go struct definition from the document on standard input and print it.
.TP
.BI \-\-schema " TYPE"
Write a TOML template for the struct type \fITYPE\fR; the Go source file
follows as the first argument.
.TP
.B \-\-version
Print the version and exit.
.TP
.B \-\-help
Print the usage.
.SH EXIT STATUS
.TP
.B 0
The document parsed and the output was written; in validate mode, every
document parsed.
.TP
.B 1
Adapter: a parse error. Validate: at least one document is invalid. Struct:
the document on standard input failed to parse.
.TP
.B 2
A usage error, a read or write failure, malformed tagged JSON, or a value
with no TOML representation.
.SH EXAMPLES
Decode a document into tagged JSON:
.PP
.nf
.RS
echo 'title = "hello"' | interpres-decode
.RE
.fi
.PP
Validate a directory of configuration, with the summary:
.PP
.nf
.RS
interpres-decode \-\-validate configs/
.RE
.fi
.PP
Infer a Go type from a document:
.PP
.nf
.RS
interpres-decode \-\-struct < config.toml > config.go
.RE
.fi
.PP
Write the template back from the type:
.PP
.nf
.RS
interpres-decode \-\-schema Config config.go
.RE
.fi
.SH SEE ALSO
The repository's
.B docs/CLI.md
carries the full reference, including the tagged-JSON wire format.
+54
View File
@@ -10,6 +10,51 @@ import (
"strings"
)
// A Number holds a TOML number as the literal the document wrote it with:
// 0x1f, 1_000, +1.0, inf. The NumbersAsLiterals option decodes integers and
// floats into
// it, so a round trip through the value tree keeps the spelling instead of a
// normalised one, and Marshal writes the literal back as it is.
//
// Number is a string type, the shape encoding/json.Number has: the literal is
// carried, not evaluated. Float64 and Int64 evaluate it on demand, and a
// destination of another numeric kind takes the evaluated value through the
// ordinary conversion rules.
type Number string
// Float64 returns the value as a float64. An integer or radix literal
// converts; a literal that is not a valid TOML number is an error.
func (n Number) Float64() (float64, error) {
v, err := decodeNumber(string(n))
if err != nil {
return 0, fmt.Errorf("interpres: %w", err)
}
switch v := v.(type) {
case float64:
return v, nil
case int64:
return float64(v), nil
}
return 0, fmt.Errorf("interpres: %q is not a number", n)
}
// Int64 returns the value as an int64. A float literal is an error, however
// whole its value, and so is a literal that is not a valid TOML number.
func (n Number) Int64() (int64, error) {
v, err := decodeNumber(string(n))
if err != nil {
return 0, fmt.Errorf("interpres: %w", err)
}
i, ok := v.(int64)
if !ok {
return 0, fmt.Errorf("interpres: %q is not an integer", n)
}
return i, nil
}
// String returns the literal itself.
func (n Number) String() string { return string(n) }
// decodeNumber parses a bare numeric token under strict TOML rules: no leading
// zeros, underscores only between digits, prefixed radixes without a sign, and
// floats with explicit fraction/exponent digits.
@@ -41,6 +86,15 @@ func decodeDecimalInt(tok string) (any, error) {
if err := checkNoLeadingZero(digits); err != nil {
return nil, err
}
// An unsigned token parses in place; only a sign needs the concatenated
// copy, and concatenating an empty sign still allocated.
if sign == "" {
i, err := strconv.ParseInt(digits, 10, 64)
if err != nil {
return nil, fmt.Errorf("integer %q out of range", tok)
}
return i, nil
}
i, err := strconv.ParseInt(sign+digits, 10, 64)
if err != nil {
return nil, fmt.Errorf("integer %q out of range", tok)
+199
View File
@@ -0,0 +1,199 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"fmt"
"maps"
"reflect"
"slices"
"sync"
)
// An OrderedMap is a string-keyed table that remembers the order its keys
// were set in, the shape a map[string]any cannot carry. Marshal writes a
// table of its own kind in that order, and decoding a document into one
// fills it in the order the document wrote the keys, where a map
// destination carries no order at all. The values are untyped, the shape
// the parser produces, so a nested table inside an OrderedMap is a plain
// map[string]any; the order is kept at the level the OrderedMap sits at.
//
// The zero value is an empty table ready for use.
type OrderedMap struct {
keys []string
values map[string]any
}
var orderedMapType = reflect.TypeFor[OrderedMap]()
// NewOrderedMap returns an empty OrderedMap.
func NewOrderedMap() *OrderedMap { return &OrderedMap{} }
// Set stores value under key. A key the table already has keeps its position
// and takes the new value; a new one joins the end.
func (m *OrderedMap) Set(key string, value any) {
if m.values == nil {
m.values = make(map[string]any, 4)
}
if _, ok := m.values[key]; !ok {
m.keys = append(m.keys, key)
}
m.values[key] = value
}
// Get returns the value under key, and whether the table has one.
func (m *OrderedMap) Get(key string) (any, bool) {
v, ok := m.values[key]
return v, ok
}
// Delete removes key. A later Set of the same key appends it to the end
// again.
func (m *OrderedMap) Delete(key string) {
if _, ok := m.values[key]; !ok {
return
}
delete(m.values, key)
m.keys = slices.DeleteFunc(m.keys, func(k string) bool { return k == key })
}
// Keys returns the keys in the order they were set.
func (m *OrderedMap) Keys() []string { return m.keys }
// Len returns the number of keys.
func (m *OrderedMap) Len() int { return len(m.keys) }
// Range calls f for every key in order, stopping when f returns false.
func (m *OrderedMap) Range(f func(key string, value any) bool) {
for _, k := range m.keys {
if !f(k, m.values[k]) {
return
}
}
}
// Map returns the values as a plain map, which carries no order. It is the
// view Marshal's Document-free callers need.
func (m *OrderedMap) Map() map[string]any { return m.values }
// --- decode: the order the document wrote ----------------------------------
// wantsOrderCache holds whether a destination type mentions OrderedMap
// anywhere a decode can reach. One computed answer per type, the same
// trade-off structSchemaCache makes.
var wantsOrderCache sync.Map // reflect.Type -> bool
// typeWantsOrder reports whether decoding into t can reach an OrderedMap, in
// which case the parse has to build the node tree the key order is read
// from. Structs walk their exported fields, and pointers, slices, arrays and
// maps walk their element; anything else holds no OrderedMap.
func typeWantsOrder(t reflect.Type) bool {
if t == nil {
return false
}
if v, ok := wantsOrderCache.Load(t); ok {
return v.(bool)
}
r := scanWantsOrder(t, make(map[reflect.Type]bool))
v, _ := wantsOrderCache.LoadOrStore(t, r)
return v.(bool)
}
func scanWantsOrder(t reflect.Type, seen map[reflect.Type]bool) bool {
for {
if t == orderedMapType {
return true
}
if seen[t] {
return false
}
seen[t] = true
switch t.Kind() {
case reflect.Pointer, reflect.Slice, reflect.Array, reflect.Map:
t = t.Elem()
case reflect.Struct:
for f := range t.Fields() {
if f.PkgPath != "" {
continue
}
if scanWantsOrder(f.Type, seen) {
return true
}
}
return false
default:
return false
}
}
}
// nodes maps a table's value map to its node, the index the decoder reads
// the written key order from. The key is the map header's runtime pointer,
// the one identity a map value offers; the nodes share their maps with the
// value tree, so one lookup per table is exact.
type nodeIndex map[uintptr]*Table
// indexNodeIndex walks a document's node tree into an index. A nil tree
// gives a nil index, which every lookup answers with nil.
func indexNodes(t *Table) nodeIndex {
if t == nil {
return nil
}
idx := nodeIndex{}
var walk func(t *Table)
walk = func(t *Table) {
idx[reflect.ValueOf(t.values).Pointer()] = t
for _, e := range t.entries {
if e.child != nil {
walk(e.child)
}
// The elements of a value array carry a node only where an element
// is an inline table; the rest are nil.
for _, el := range e.elements {
if el != nil {
walk(el)
}
}
}
}
walk(t)
return idx
}
// nodeOf returns the node a value table was parsed into, or nil when the
// parse built no node tree, which is the ordinary decode's shape. A tree
// built by hand carries no nodes either.
func (d *decoder) nodeOf(tbl map[string]any) *Table {
return d.nodes[reflect.ValueOf(tbl).Pointer()]
}
// fillOrderedMap decodes a parsed table into an OrderedMap destination,
// taking the keys in the order the document wrote them. A table with no
// node, which is what a hand-built tree or a ParseMap result offers, fills
// in sorted key order, the deterministic order a map can offer.
func (d *decoder) fillOrderedMap(tbl map[string]any, dst reflect.Value) error {
if !dst.CanAddr() {
return fmt.Errorf("interpres: cannot decode into an OrderedMap that is not addressable")
}
om := dst.Addr().Interface().(*OrderedMap)
if om.values == nil {
om.values = make(map[string]any, len(tbl))
}
keys := slices.Sorted(maps.Keys(tbl))
if node := d.nodeOf(tbl); node != nil {
keys = node.Keys()
}
for _, key := range keys {
val, ok := tbl[key]
if !ok {
continue
}
elem := reflect.New(reflect.TypeFor[any]()).Elem()
if err := d.assign(val, elem); err != nil {
return newDecodeError(key, err)
}
om.Set(key, elem.Interface())
}
return nil
}
+214
View File
@@ -0,0 +1,214 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"context"
"slices"
"testing"
)
func TestOrderedMapBasics(t *testing.T) {
m := NewOrderedMap()
if m.Len() != 0 {
t.Fatalf("fresh map holds %d keys", m.Len())
}
m.Set("b", 1)
m.Set("a", 2)
m.Set("c", 3)
if got := m.Keys(); !slices.Equal(got, []string{"b", "a", "c"}) {
t.Errorf("keys = %v, want [b a c]", got)
}
if v, ok := m.Get("a"); !ok || v != 2 {
t.Errorf("a = %v, %v", v, ok)
}
m.Set("a", 9)
if got := m.Keys(); !slices.Equal(got, []string{"b", "a", "c"}) {
t.Errorf("keys after replace = %v, want the position kept", got)
}
if v, _ := m.Get("a"); v != 9 {
t.Errorf("a = %v, want 9", v)
}
seen := ""
m.Range(func(key string, value any) bool {
seen += key
return key != "a"
})
if seen != "ba" {
t.Errorf("range visited %q, want \"ba\"", seen)
}
m.Delete("b")
m.Delete("missing")
if got := m.Keys(); !slices.Equal(got, []string{"a", "c"}) {
t.Errorf("keys after delete = %v, want [a c]", got)
}
m.Delete("c")
m.Set("c", 3)
if got := m.Keys(); !slices.Equal(got, []string{"a", "c"}) {
t.Errorf("re-set key = %v, want it appended as [a c]", got)
}
}
func TestMarshalOrderedMap(t *testing.T) {
t.Run("top level keeps the order", func(t *testing.T) {
m := NewOrderedMap()
m.Set("zebra", int64(1))
m.Set("alpha", "x")
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
want := "zebra = 1\nalpha = \"x\"\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
t.Run("a pointer top level does the same", func(t *testing.T) {
m := &OrderedMap{}
m.Set("second", true)
m.Set("first", int64(2))
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
if string(out) != "second = true\nfirst = 2\n" {
t.Errorf("output %q", out)
}
})
t.Run("a struct field keeps the order as a table", func(t *testing.T) {
type Cfg struct {
Title string `toml:"title"`
Extra *OrderedMap `toml:"extra"`
}
m := &OrderedMap{}
m.Set("late", int64(1))
m.Set("early", int64(2))
out, err := Marshal(Cfg{Title: "t", Extra: m})
if err != nil {
t.Fatal(err)
}
want := "title = \"t\"\n\n[extra]\nlate = 1\nearly = 2\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
t.Run("inline form keeps the order too", func(t *testing.T) {
m := NewOrderedMap()
m.Set("zebra", int64(1))
m.Set("alpha", int64(2))
out, err := Marshal(map[string]any{"t": m}, InlineTables(60))
if err != nil {
t.Fatal(err)
}
if string(out) != "t = {zebra = 1, alpha = 2}\n" {
t.Errorf("output %q", out)
}
})
t.Run("an array of tables keeps each element's order", func(t *testing.T) {
type Cfg struct {
Items []*OrderedMap `toml:"items"`
}
a, b := NewOrderedMap(), NewOrderedMap()
a.Set("y", int64(1))
a.Set("x", int64(2))
b.Set("n", int64(3))
out, err := Marshal(Cfg{Items: []*OrderedMap{a, b}})
if err != nil {
t.Fatal(err)
}
want := "[[items]]\ny = 1\nx = 2\n\n[[items]]\nn = 3\n"
if string(out) != want {
t.Errorf("output:\n%q\nwant:\n%q", out, want)
}
})
t.Run("a nil value is skipped", func(t *testing.T) {
m := NewOrderedMap()
m.Set("gone", nil)
m.Set("here", int64(1))
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
if string(out) != "here = 1\n" {
t.Errorf("output %q", out)
}
})
}
func TestDecodeOrderedMap(t *testing.T) {
t.Run("keys come back in written order", func(t *testing.T) {
doc := []byte("zebra = 1\nmiddle = \"m\"\nalpha = true\n")
var m OrderedMap
if err := Unmarshal(doc, &m); err != nil {
t.Fatal(err)
}
if got := m.Keys(); !slices.Equal(got, []string{"zebra", "middle", "alpha"}) {
t.Fatalf("keys = %v", got)
}
if v, _ := m.Get("middle"); v != "m" {
t.Errorf("middle = %#v", v)
}
})
t.Run("a nested table keeps the table order", func(t *testing.T) {
type Cfg struct {
Ports []int `toml:"ports"`
DB *OrderedMap `toml:"db"`
}
doc := []byte("ports = [1, 2]\n\n[db]\nslow = 1\nfast = 2\n")
var cfg Cfg
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if got := cfg.DB.Keys(); !slices.Equal(got, []string{"slow", "fast"}) {
t.Errorf("db keys = %v", got)
}
})
t.Run("an array of tables fills in order", func(t *testing.T) {
var m OrderedMap
doc := []byte("b = 1\n[[items]]\nname = \"x\"\n[[items]]\nname = \"y\"\na = 2\n")
if err := Unmarshal(doc, &m); err != nil {
t.Fatal(err)
}
if got := m.Keys(); !slices.Equal(got, []string{"b", "items"}) {
t.Errorf("keys = %v, want [b items]", got)
}
elems, ok := m.values["items"].([]map[string]any)
if !ok || len(elems) != 2 {
t.Fatalf("items = %#v", m.values["items"])
}
if elems[1]["name"] != "y" {
t.Errorf("second element = %#v", elems[1])
}
})
t.Run("the sorted fallback needs a tree without nodes", func(t *testing.T) {
// Unmarshal and Decode build the node tree whenever the destination can
// reach an OrderedMap, so the sorted fallback is only reachable from a
// tree that never had one.
tree, _, err := parseWithOptions(context.Background(), []byte("b = 1\na = 2\n"), parseOptions{}, false)
if err != nil {
t.Fatal(err)
}
var m OrderedMap
if err := newDecoder().decode(tree, &m); err != nil {
t.Fatal(err)
}
if got := m.Keys(); !slices.Equal(got, []string{"a", "b"}) {
t.Errorf("keys = %v, want the sorted [a b]", got)
}
})
t.Run("the order survives a round trip", func(t *testing.T) {
doc := []byte("z = 1\na = 2\nm = 3\n")
var m OrderedMap
if err := Unmarshal(doc, &m); err != nil {
t.Fatal(err)
}
out, err := Marshal(m)
if err != nil {
t.Fatal(err)
}
if string(out) != "z = 1\na = 2\nm = 3\n" {
t.Errorf("output:\n%q", out)
}
})
}
+681 -116
View File
File diff suppressed because it is too large Load Diff
+47
View File
@@ -0,0 +1,47 @@
#!/usr/bin/env perl
# Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
# SPDX-License-Identifier: MIT
# docs-drift compares the toml-test suite counts the documentation names with
# the run this repository produces now. A corpus change moves the counts, and
# README.md and docs/ARCHITECTURE.md quote them; this is the check that keeps
# the quotation honest. Builtins only, and the toml-test binary on PATH.
use v5.40;
my $out = qx{toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode --encode' -toml=1.1 2>&1};
die "docs-drift: toml-test failed to run; build the adapter first (just build)\n"
if !defined $out || $out =~ /not found|No such file/;
# A red run is a failure to answer, not a drift: the counts it prints describe
# a suite that did not pass, and sending the maintainer to correct counts that
# are correct would be the wrong diagnosis.
die "docs-drift: the toml-test run failed; fix the suite before comparing counts\n"
if $? != 0;
my %live;
for my $kind (qw(valid invalid encoder)) {
my ($passed) = $out =~ /\b$kind tests:\s+(\d+) passed/;
die "docs-drift: could not read the $kind count from the toml-test output\n"
unless defined $passed;
my ($failed) = $out =~ /\b$kind tests:\s+\d+ passed, (\d+) failed/;
die "docs-drift: the $kind run has $failed failures; fix the suite first\n"
if defined $failed && $failed != 0;
$live{$kind} = $passed;
}
print "docs-drift: the suite now stands at $live{valid} valid, $live{invalid} invalid and $live{encoder} encoder cases\n";
my $drift = 0;
for my $file ('README.md', 'docs/ARCHITECTURE.md') {
open(my $fh, '<', $file) or die "docs-drift: cannot read $file: $!\n";
my $text = do { local $/; <$fh> };
close($fh);
while ($text =~ /(\d+)\s+(valid|invalid|encoder)/g) {
my ($quoted, $kind) = ($1, $2);
if ($quoted != $live{$kind}) {
print "docs-drift: $file quotes $quoted $kind cases, the suite says $live{$kind}\n";
$drift = 1;
}
}
}
die "docs-drift: the documentation has drifted from the suite\n" if $drift;
print "docs-drift: the documentation matches the suite\n";
+1396
View File
File diff suppressed because it is too large Load Diff
+874
View File
@@ -0,0 +1,874 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"errors"
"maps"
"net"
"reflect"
"strings"
"sync/atomic"
"testing"
"time"
)
type targetNested struct {
X int `toml:"x"`
Y string `toml:"y"`
}
type targetCfg struct {
Num int `toml:"num"`
Small uint8 `toml:"small"`
Tags []string `toml:"tags"`
Lims map[string]any `toml:"lims"`
Tab targetNested `toml:"tab"`
Arr []targetNested `toml:"arr"`
Other string `toml:"other"`
}
// TestTargetedStrictFindings pins the strict findings of the targeted parse
// to the tree decode's own texts, paths included. Every case here was first
// surfaced by FuzzTargetedDecode.
func TestTargetedStrictFindings(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "unknown key in a header table",
doc: "[tab]\nother = \"o\"\n",
want: `interpres: tab: unknown field "other" for interpres.targetNested`,
},
{
name: "unknown nested header without the parent header",
doc: "[tab.nested]\nx = 1\n",
want: `interpres: tab: unknown field "nested" for interpres.targetNested`,
},
{
name: "unknown key in an array-of-tables element",
doc: "[[arr]]\nother = \"o\"\n",
want: `interpres: arr[0]: unknown field "other" for interpres.targetNested`,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg, RejectUnknownFields(true))
if err == nil {
t.Fatalf("no error, want %q", tt.want)
}
if err.Error() != tt.want {
t.Errorf("message = %q, want %q", err.Error(), tt.want)
}
})
}
}
// TestTargetedParseErrors pins the parse-stage errors the targeted skeleton
// raises, whose texts and lines are the tree parser's own.
func TestTargetedParseErrors(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "header on an assigned scalar",
doc: "zz = 1\n[zz]\nx = 4\n",
want: "interpres: line 2: key \"zz\" is not a table",
},
{
name: "dotted key on an assigned scalar",
doc: "zz = 1\nzz.x = 2\n",
want: "interpres: line 2: key \"zz\" is not a table",
},
{
name: "duplicate unknown keys",
doc: "zz = 1\nzz = 2\n",
want: "interpres: line 2: duplicate key \"zz\"",
},
{
name: "duplicate inside an unknown table",
doc: "[zz]\nk = 1\nk = 2\n",
want: "interpres: line 3: duplicate key \"k\"",
},
{
name: "duplicate across a sink's dotted keys",
doc: "[zz]\na.b = 1\na.b = 2\n",
want: "interpres: line 3: duplicate key \"b\"",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg)
if err == nil {
t.Fatalf("no error, want %q", tt.want)
}
if err.Error() != tt.want {
t.Errorf("message = %q, want %q", err.Error(), tt.want)
}
})
}
}
// TestTargetedSilentShapes covers the documents the targeted parse accepts
// with the values the tree decode gives.
func TestTargetedSilentShapes(t *testing.T) {
t.Run("dotted key after an unknown nested header", func(t *testing.T) {
// [tab.nested] is unknown and sinks; tab.x then lands in tab, and the
// sink's own x is a different key, the tree's shape exactly.
var cfg, ref targetCfg
in := []byte("[tab]\nx = 1\n[tab.nested]\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if err := treeDecodeInto(in, &ref); err != nil {
t.Fatalf("reference: %v", err)
}
if !reflect.DeepEqual(cfg, ref) {
t.Errorf("values disagree: targeted %+v, tree %+v", cfg, ref)
}
if cfg.Tab.X != 1 {
t.Errorf("tab.x = %d, want 1", cfg.Tab.X)
}
})
t.Run("unknown keys are ignored without strict", func(t *testing.T) {
var cfg, ref targetCfg
in := []byte("num = 5\nz1 = 1\n[zz]\nk = 1\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if err := treeDecodeInto(in, &ref); err != nil {
t.Fatalf("reference: %v", err)
}
if !reflect.DeepEqual(cfg, ref) {
t.Errorf("values disagree: targeted %+v, tree %+v", cfg, ref)
}
if cfg.Num != 5 {
t.Errorf("num = %d, want 5", cfg.Num)
}
})
t.Run("an inline table into a map field", func(t *testing.T) {
var cfg targetCfg
in := []byte("lims = { cpu = 4, deep = { a = true } }\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if cfg.Lims["cpu"] != int64(4) {
t.Errorf("lims = %v", cfg.Lims)
}
})
t.Run("an overflow falls back to the decode error", func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("small = 300\n"), &cfg)
want := "interpres: small: integer 300 overflows uint8"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("too many array-of-tables elements falls back", func(t *testing.T) {
type Item struct {
N int `toml:"n"`
}
var cfg struct {
Items [2]Item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n[[items]]\nn = 2\n[[items]]\nn = 3\n"), &cfg)
want := "interpres: items: cannot assign 3 elements to [2]interpres.Item"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("a UseNumber tree keeps literals in the targeted path", func(t *testing.T) {
var cfg struct {
Rate Number `toml:"rate"`
}
if err := Unmarshal([]byte("rate = 1_000\n"), &cfg, NumbersAsLiterals(true)); err != nil {
t.Fatal(err)
}
if cfg.Rate != "1_000" {
t.Errorf("rate = %q, want 1_000", cfg.Rate)
}
})
t.Run("dotted keys fill a map field", func(t *testing.T) {
var cfg targetCfg
in := []byte("lims.a.b = true\nlims.c = 3\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatalf("decode: %v", err)
}
if cfg.Lims["c"] != int64(3) {
t.Errorf("lims = %v", cfg.Lims)
}
})
t.Run("an inline table cannot be extended", func(t *testing.T) {
var cfg targetCfg
in := []byte("lims = { a = 1 }\n[lims.deep]\nb = 2\n")
err := Unmarshal(in, &cfg)
if err == nil || !strings.Contains(err.Error(), "cannot extend inline table") {
t.Errorf("err = %v, want the inline-table extension error", err)
}
})
}
// TestTargetedShapesMatrix walks a document per destination shape, both
// through the targeted path and the tree reference, so the two agree on
// every branch the skeleton carries.
func TestTargetedShapesMatrix(t *testing.T) {
docs := []string{
// Scalars of every kind, arrays, maps, tables, arrays of tables.
"num = 7\nflt = 1.25\nstr = \"s\"\nflag = false\nsmall = 9\ntags = [\"a\"]\nlims = { a = 1 }\n\n[tab]\nx = 1\ny = \"t\"\n\n[[arr]]\nx = 2\ny = \"u\"\n\n[[arr]]\nx = 3\ny = \"v\"\n",
// Dotted keys through nested tables and maps.
"tab.x = 1\ntab.y = \"s\"\nlims.a.b = true\nlims.c = 3\nnum = 2\n",
// Inline tables nested in arrays, mixed value arrays.
"lims = { a = { b = 1 } }\ntags = []\nother = \"o\"\n",
// A sub-table of an array of tables, then a second element.
"[[arr]]\nx = 1\n[arr.nested]\ny = \"n\"\n[[arr]]\ny = \"m\"\n",
// Negative and signed numbers, exponents, radix forms into floats.
"flt = -3.5e2\nnum = -42\nflt = +1.0\n",
// A quoted key and a defined-string-shaped value.
"\"quoted key\" = 1\nstr = \"multi\"\n",
}
for i, doc := range docs {
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
tgtErr := Unmarshal([]byte(doc), &tgt)
if (refErr == nil) != (tgtErr == nil) {
t.Errorf("doc %d: error presence disagrees: tree %v, targeted %v", i, refErr, tgtErr)
continue
}
if refErr != nil {
continue
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("doc %d: values disagree:\ntree: %#v\ntargeted: %#v", i, ref, tgt)
}
}
}
// TestTargetedFallbackContracts pins the documents that must fall back and
// produce the tree decode's exact error.
func TestTargetedFallbackContracts(t *testing.T) {
type Item struct {
N int `toml:"n"`
}
tests := []struct {
name string
doc string
want string
}{
{
name: "uint8 overflow",
doc: "small = 300\n",
want: "interpres: small: integer 300 overflows uint8",
},
{
name: "negative into uint",
doc: "small = -1\n",
want: "interpres: small: cannot assign negative -1 to uint8",
},
{
name: "a table into a scalar",
doc: "num = { a = 1 }\n",
want: "interpres: num: cannot assign table to int",
},
{
name: "an integer into a string field",
doc: "other = 5\n",
want: "interpres: other: cannot assign integer to string",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg)
if err == nil || err.Error() != tt.want {
t.Errorf("err = %v, want %q", err, tt.want)
}
})
}
_ = Item{}
}
// TestTargetedHeaderOnAssignedScalar pins the parse error a header raises
// when the key already holds a scalar, before any fallback can happen.
func TestTargetedHeaderOnAssignedScalarArray(t *testing.T) {
var cfg targetCfg
in := []byte("arr = []\n[[arr]]\nx = 1\n")
err := Unmarshal(in, &cfg)
want := "interpres: line 2: key \"arr\" is not an array of tables"
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
}
// TestTargetedBranchParity walks the fallback branches of the targeted
// skeleton: every document here takes the tree path on a rerun, and must
// carry the tree decode's exact error text.
func TestTargetedBranchParity(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "a header over a value array",
doc: "tags = [\"x\"]\n[tags]\na = 1\n",
want: "interpres: line 2: key \"tags\" is not a table",
},
{
name: "an array header over a value array",
doc: "tags = [\"x\"]\n[[tags]]\na = 1\n",
want: "interpres: line 2: key \"tags\" is not an array of tables",
},
{
name: "an array header over a datetime field",
doc: "when = 1979-05-27T07:32:00Z\n[[when]]\nx = 1\n",
want: "interpres: line 2: key \"when\" is not an array of tables",
},
{
name: "a boolean into a string field",
doc: "other = true\n",
want: "interpres: other: cannot assign bool to string",
},
{
name: "a leading-zero integer",
doc: "num = 01\n",
want: "interpres: line 1: leading zeros are not allowed in numbers",
},
{
name: "an int64-overflowing integer",
doc: "num = 99999999999999999999\n",
want: "interpres: line 1: integer \"99999999999999999999\" out of range",
},
{
name: "a malformed boolean",
doc: "flag = tru\n",
want: "interpres: line 1: invalid value",
},
{
name: "a negative number into an unsigned field",
doc: "small = -5\n",
want: "interpres: small: cannot assign negative -5 to uint8",
},
{
name: "an integer into a string field via the generic path",
doc: "other = 5\n",
want: "interpres: other: cannot assign integer to string",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte(tt.doc), &cfg, RejectUnknownFields(true))
if tt.want == "" {
if err != nil {
t.Fatalf("err = %v, want nil", err)
}
return
}
if err == nil || err.Error() != tt.want {
t.Errorf("err = %v, want %q", err, tt.want)
}
})
}
}
// TestTargetedDecodeHookFields keeps the custom decode hooks of scalar-typed
// fields working in the targeted path.
func TestTargetedDecodeHookFields(t *testing.T) {
type Cfg struct {
IP net.IP `toml:"ip"`
Dur time.Duration `toml:"dur"`
Unm *scalarUnmarshaler `toml:"unm"`
}
var cfg Cfg
in := []byte("ip = \"192.0.2.1\"\ndur = \"1h30m\"\nunm = \"hello\"\n")
if err := Unmarshal(in, &cfg); err != nil {
t.Fatal(err)
}
if cfg.IP.String() != "192.0.2.1" {
t.Errorf("ip = %v", cfg.IP)
}
if cfg.Dur != 90*time.Minute {
t.Errorf("dur = %v", cfg.Dur)
}
if cfg.Unm == nil || cfg.Unm.val != "hello" {
t.Errorf("unm = %+v", cfg.Unm)
}
}
// TestTargetedOddShapes pins the fallback and value shapes the matrix does
// not reach: space-separated date-times, non-string map keys and repeated
// dotted map keys.
func TestTargetedOddShapes(t *testing.T) {
t.Run("a space-separated date-time", func(t *testing.T) {
type Cfg struct {
When time.Time `toml:"when"`
}
var cfg, ref Cfg
doc := []byte("when = 1979-05-27 07:32:00Z\n")
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatal(err)
}
if !cfg.When.Equal(ref.When) {
t.Errorf("when = %v, want %v", cfg.When, ref.When)
}
})
t.Run("a map with a non-string key falls back", func(t *testing.T) {
type Cfg struct {
M map[int]string `toml:"m"`
}
var cfg, ref Cfg
doc := []byte("m = { a = 1 }\n")
err := Unmarshal(doc, &cfg)
refErr := treeDecodeInto(doc, &ref)
if err == nil || refErr == nil {
t.Fatalf("err = %v, refErr = %v, want both to fail", err, refErr)
}
if err.Error() != refErr.Error() {
t.Errorf("errors disagree: targeted %q, tree %q", err, refErr)
}
})
t.Run("a repeated dotted map key is a duplicate", func(t *testing.T) {
var cfg targetCfg
doc := []byte("lims.a.b = 1\nlims.a.b = 2\n")
err := Unmarshal(doc, &cfg)
want := "interpres: line 2: duplicate key \"b\""
if err == nil || err.Error() != want {
t.Errorf("err = %v, want %q", err, want)
}
})
t.Run("an underscored integer takes the token path", func(t *testing.T) {
var cfg targetCfg
doc := []byte("num = 1_000\n")
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Num != 1000 {
t.Errorf("num = %d, want 1000", cfg.Num)
}
})
t.Run("an 18-digit integer takes the fast path", func(t *testing.T) {
var cfg struct {
Big int64 `toml:"big"`
}
doc := []byte("big = 999999999999999999\n")
if err := Unmarshal(doc, &cfg); err != nil {
t.Fatal(err)
}
if cfg.Big != 999999999999999999 {
t.Errorf("big = %d", cfg.Big)
}
})
}
// TestTargetedMapTableShapes covers the map-entry branches of the targeted
// skeleton: entries that become tables, entries that refuse them, and the
// duplicate checks across them.
func TestTargetedMapTableShapes(t *testing.T) {
tests := []struct {
name string
doc string
want string
}{
{
name: "a header opens a map entry table",
doc: "lims.c = 1\n[lims.d]\nk = 1\n",
want: "",
},
{
name: "a header over an assigned map entry",
doc: "lims.a = 1\n[lims.a]\nk = 1\n",
want: "interpres: line 2: key \"a\" is not a table",
},
{
name: "a dotted key over an assigned map entry",
doc: "lims.a = 1\nlims.a.b = 2\n",
want: "interpres: line 2: key \"a\" is not a table",
},
{
name: "a duplicate plain map entry",
doc: "lims.a = 1\nlims.a = 2\n",
want: "interpres: line 2: duplicate key \"a\"",
},
{
name: "an array of tables inside a map entry",
doc: "lims.c = 1\n[[lims.items]]\nk = 1\n",
want: "",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var cfg, ref targetCfg
err := Unmarshal([]byte(tt.doc), &cfg)
refErr := treeDecodeInto([]byte(tt.doc), &ref)
if (err == nil) != (refErr == nil) {
t.Fatalf("error presence disagrees: tree %v, targeted %v", refErr, err)
}
if err != nil {
if err.Error() != refErr.Error() {
t.Fatalf("errors disagree:\ntree: %v\ntargeted: %v", refErr, err)
}
return
}
if !reflect.DeepEqual(cfg, ref) {
t.Errorf("values disagree: targeted %+v, tree %+v", cfg, ref)
}
})
}
}
// TestTargetedNestedMapDescents pins the descents into a map of maps that
// meet entries the document built earlier: a dotted key twice through the
// same sub-table, a header into a dotted-built sub-table, and a typed array
// under a map key. Each shape once panicked on a reflect Elem of a map.
func TestTargetedNestedMapDescents(t *testing.T) {
t.Run("dotted key through one sub-table twice", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\nm.a.c = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["c"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("header under a dotted-built sub-table", func(t *testing.T) {
var cfg struct {
M map[string]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("m.a.b = 1\n[m.a.deep]\nx = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.M["a"]["b"] != int64(1) || cfg.M["a"]["deep"].(map[string]any)["x"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
t.Run("typed array under a map key", func(t *testing.T) {
var cfg struct {
M map[string][]map[string]any `toml:"m"`
}
err := Unmarshal([]byte("[[m.arr]]\nx = 1\n\n[[m.arr]]\ny = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.M["arr"]) != 2 || cfg.M["arr"][1]["y"] != int64(2) {
t.Errorf("m = %#v", cfg.M)
}
})
}
// TestTargetedPointerElementSlice pins that an array of tables over a slice
// of pointer elements fills the pointed-to structs.
func TestTargetedPointerElementSlice(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items []*item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n\n[[items]]\nn = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 2 || cfg.Items[0] == nil || cfg.Items[1].N != 2 {
t.Errorf("items = %#v", cfg.Items)
}
}
// TestTargetedArrayScopeResets pins that a new element of an array of tables
// starts a fresh definition scope, the contract the changelog documents.
func TestTargetedArrayScopeResets(t *testing.T) {
doc := "[[a]]\nb.c = 1\n\n[[a]]\n\n[a.b]\nx = 1\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
if refErr != nil {
t.Fatalf("tree decode: %v", refErr)
}
if err := Unmarshal([]byte(doc), &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
}
// TestTargetedUnknownArrayElements pins that every element of an unknown
// array of tables is a fresh namespace, and a sub-table header reaches the
// last element the way the tree parser's does.
func TestTargetedUnknownArrayElements(t *testing.T) {
doc := "[[zz]]\nk = 1\n\n[[zz]]\nk = 2\n\n[zz.sub]\nx = 3\n"
var ref, tgt targetCfg
refErr := treeDecodeInto([]byte(doc), &ref)
tgtErr := Unmarshal([]byte(doc), &tgt)
if (refErr == nil) != (tgtErr == nil) {
t.Fatalf("error presence disagrees: tree %v, targeted %v", refErr, tgtErr)
}
if refErr != nil {
return
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
// A dotted key may not enter the array: the tree's own rule.
var dotted targetCfg
dErr := Unmarshal([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
refDotted := treeDecodeInto([]byte("[[zz]]\nk = 1\nzz.x = 2\n"), &dotted)
if (dErr == nil) != (refDotted == nil) {
t.Errorf("dotted into an array: targeted %v, tree %v", dErr, refDotted)
}
}
// TestTargetedFixedArrayUnderFill pins that a fixed-size array the document
// under-fills is the length mismatch the tree decode raises, with the
// field's path.
func TestTargetedFixedArrayUnderFill(t *testing.T) {
type item struct {
N int `toml:"n"`
}
var cfg struct {
Items [2]item `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nn = 1\n"), &cfg)
if err == nil {
t.Fatal("unmarshal accepted an under-filled array")
}
want := `interpres: items: cannot assign 1 elements to [2]interpres.item`
if err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedPrefilledSliceReplaced pins that a prefilled slice is replaced
// by the document's elements on both paths, not appended to.
func TestTargetedPrefilledSliceReplaced(t *testing.T) {
type item struct {
N int `toml:"n"`
}
doc := []byte("[[items]]\nn = 1\n")
var ref struct {
Items []item `toml:"items"`
}
ref.Items = []item{{N: 9}}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
var tgt struct {
Items []item `toml:"items"`
}
tgt.Items = []item{{N: 9}}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if len(tgt.Items) != 1 || tgt.Items[0].N != 1 {
t.Errorf("items = %#v, want the prefilled element replaced", tgt.Items)
}
}
// TestTargetedHeaderOverValueArrayKeepsCase pins that a value array assigned
// under a differently cased key than the field's name still blocks the
// array-of-tables header over it, the tree parse error.
func TestTargetedHeaderOverValueArrayKeepsCase(t *testing.T) {
var cfg struct {
Arr []targetNested `toml:"arr"`
}
err := Unmarshal([]byte("Arr = [{x = 1}]\n[[Arr]]\nx = 2\n"), &cfg)
if err == nil || err.Error() != `interpres: line 2: key "Arr" is not an array of tables` {
t.Errorf("err = %v, want the parse error over the assigned field", err)
}
}
// TestTargetedDottedInlineFreezePath pins that an inline table assigned by a
// dotted key freezes the whole path the statement wrote: a later header
// under that path is the extension error, and a key outside it stays free.
func TestTargetedDottedInlineFreezePath(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("m.a.b = {x = 1}\nb.y = 2\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
err = Unmarshal([]byte("m.a.b = {x = 1}\n[m.a.b]\ny = 2\n"), &cfg)
want := `interpres: line 2: cannot extend inline table "m.a.b"`
if err == nil || err.Error() != want {
t.Errorf("err = %v\nwant %q", err, want)
}
}
// TestTargetedStrictThroughDottedKeys pins that strict and required findings
// survive the transient tables a dotted descent builds.
func TestTargetedStrictThroughDottedKeys(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("tab.zz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the strict failure through the dotted key", err)
}
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the path through the dotted key", err)
}
}
// TestTargetedRequiredThroughDottedKeys pins that a required tag is honoured
// when the table is reached only through dotted keys.
func TestTargetedRequiredThroughDottedKeys(t *testing.T) {
type nested struct {
X int `toml:"x,required"`
Y int `toml:"y"`
}
var cfg struct {
Tab nested `toml:"tab"`
}
err := Unmarshal([]byte("tab.y = 1\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), `missing required key "x"`) {
t.Errorf("err = %v, want the missing required key through the dotted key", err)
}
}
// TestTargetedOrderedMapSliceFallsBack pins that a slice of OrderedMap
// elements takes the tree path, whose fill keeps the written order.
func TestTargetedOrderedMapSliceFallsBack(t *testing.T) {
var cfg struct {
Items []OrderedMap `toml:"items"`
}
err := Unmarshal([]byte("[[items]]\nk = \"v\"\n"), &cfg)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(cfg.Items) != 1 || cfg.Items[0].Keys()[0] != "k" {
t.Errorf("items = %#v, want the element filled in written order", cfg.Items)
}
}
// hookMap is a named map type whose decode hook counts its calls.
type hookMap map[string]any
var hookMapCalls atomic.Int32
func (h *hookMap) UnmarshalTOML(data any) error {
hookMapCalls.Add(1)
m, _ := data.(map[string]any)
if *h == nil {
*h = hookMap{}
}
maps.Copy((*h), m)
return nil
}
// TestTargetedMapFieldHookGetsWholeTable pins that a named map field with a
// decode hook receives the whole parsed table, even in its header form.
func TestTargetedMapFieldHookGetsWholeTable(t *testing.T) {
type cfg struct {
M hookMap `toml:"m"`
}
var c cfg
hookMapCalls.Store(0)
err := Unmarshal([]byte("[m]\na = 1\nb = 2\n"), &c)
if err != nil {
t.Fatalf("unmarshal: %v", err)
}
if hookMapCalls.Load() != 1 {
t.Errorf("hook calls = %d, want exactly one with the whole table", hookMapCalls.Load())
}
if c.M["a"] != int64(1) || c.M["b"] != int64(2) {
t.Errorf("m = %#v", c.M)
}
}
// errHook fails every decode with a fixed error and counts its calls.
type errHook struct{ calls *int }
func (e *errHook) UnmarshalTOML(any) error {
if e.calls != nil {
*e.calls++
}
return errors.New("boom")
}
// TestTargetedHookErrorRunsOnce pins that a failing hook's error is the
// tree path's own, wrapped with the key, and that the hook is not run a
// second time by a fallback.
func TestTargetedHookErrorRunsOnce(t *testing.T) {
calls := 0
cfg := struct {
F errHook `toml:"f"`
}{F: errHook{calls: &calls}}
err := Unmarshal([]byte("f = 1\n"), &cfg)
if err == nil || err.Error() != "interpres: f: unmarshal: boom" {
t.Errorf("err = %v, want the wrapped hook failure", err)
}
if calls != 1 {
t.Errorf("hook calls = %d, want one", calls)
}
}
// TestTargetedUnknownBeforeRequired pins the report order the tree decode
// produces: an unknown key wins over a missing required one.
func TestTargetedUnknownBeforeRequired(t *testing.T) {
type inner struct {
X int `toml:"x,required"`
}
var cfg struct {
Tab inner `toml:"tab"`
}
err := Unmarshal([]byte("[tab]\nzz = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.Contains(err.Error(), `unknown field "zz"`) {
t.Errorf("err = %v, want the unknown key reported before the required one", err)
}
}
// TestTargetedStrictPathStableAcrossHeaders pins that the path a strict
// finding wraps does not alias the parser's key buffer: the table that owns
// the unknown key keeps its name after a later header.
func TestTargetedStrictPathStableAcrossHeaders(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("[tab]\nzz = 1\n\n[lims]\nx = 1\n"), &cfg, RejectUnknownFields(true))
if err == nil || !strings.HasPrefix(err.Error(), "interpres: tab:") {
t.Errorf("err = %v, want the finding on tab, not the later header", err)
}
}
// TestTargetedPrefilledMapFieldMergesUnderHeader pins that a prefilled map
// field merges the document's header-form table into it on both paths, the
// rule the root map has always followed.
func TestTargetedPrefilledMapFieldMergesUnderHeader(t *testing.T) {
doc := []byte("[lims]\nnew = 3\n")
var ref, tgt targetCfg
ref.Lims = map[string]any{"keep": "yes"}
if err := treeDecodeInto(doc, &ref); err != nil {
t.Fatalf("tree decode: %v", err)
}
tgt.Lims = map[string]any{"keep": "yes"}
if err := Unmarshal(doc, &tgt); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if !reflect.DeepEqual(ref, tgt) {
t.Errorf("targeted = %#v, tree = %#v", tgt, ref)
}
if tgt.Lims["keep"] != "yes" || tgt.Lims["new"] != int64(3) {
t.Errorf("lims = %#v, want the merge", tgt.Lims)
}
}
// TestTargetedNumberTokenValidatesUTF8 pins that the token route the
// targeted parse takes reports invalid UTF-8 with the scanner's own message
// and position.
func TestTargetedNumberTokenValidatesUTF8(t *testing.T) {
var cfg targetCfg
err := Unmarshal([]byte("num = 12\xff\n"), &cfg)
if err == nil || !strings.Contains(err.Error(), "invalid UTF-8 in value at byte offset 8") {
t.Errorf("err = %v, want the UTF-8 complaint on the invalid byte", err)
}
}
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("0=00:00\n1=0000-01-01 00:00:00.0+00:00#000000000000")
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("e = \"\\\\e[0m\\\\x41\\\\x7f\\\\x00\"\n")
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("m = {\n\ttitle = \"one\",\n\tnums = [1, 2,],\n\tinner = { deep = true }, # trailing\n}\n")
+2
View File
@@ -0,0 +1,2 @@
go test fuzz v1
[]byte("t = 13:37\nbig = 1979-05-27 07:32:00.5+01:00\nshort = 1979-05-27 07:32\n")