Compare commits
58
Commits
v1.0.0
...
582222738b
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
582222738b | ||
|
|
3a4bd74bf0 | ||
|
|
b4d564c682 | ||
|
|
adf189aa2c | ||
|
|
7596754180 | ||
|
|
72f8b21ac4 | ||
|
|
a6e3e3fe31 | ||
|
|
b695b69768 | ||
|
|
959eaba4b0 | ||
|
|
8f85bb68fa | ||
|
|
bccaf087c8 | ||
|
|
0149a5b4d1 | ||
|
|
815141440e | ||
|
|
9023784da3 | ||
|
|
942c4b1489 | ||
|
|
8f0eae6604 | ||
|
|
1c7329aeea | ||
|
|
ad6c32d0c6 | ||
|
|
17574a0d15 | ||
|
|
8aa2b1b9c0 | ||
|
|
6a043e2824 | ||
|
|
81033bb27c | ||
|
|
dfd5d240d2 | ||
|
|
78946578d1 | ||
|
|
d365729b37 | ||
|
|
53102d70e6 | ||
|
|
f1a757ec5c | ||
|
|
3c8ac859c0 | ||
|
|
4def1b3e8b | ||
|
|
d5327568fb | ||
|
|
c485aab227 | ||
|
|
30b28fe7fc | ||
|
|
aaea68efc9 | ||
|
|
a8d69d90d5 | ||
|
|
bb238c98c3 | ||
|
|
54c6032a9a | ||
|
|
ec0d7a0023 | ||
|
|
feef4fe9ea | ||
|
|
3c1f65038b | ||
|
|
830f840f44 | ||
|
|
696f117c22 | ||
|
|
d2fc31d260 | ||
|
|
18f1cd51e9 | ||
|
|
3cd538fad6 | ||
|
|
e19a6f35f1 | ||
|
|
5a270d0879 | ||
|
|
3f41266710 | ||
|
|
1e3198c8b6 | ||
|
|
cdb42de561 | ||
|
|
0f6d81fe3e | ||
|
|
274b8a488c | ||
|
|
b061c97a81 | ||
|
|
fc50e3c49a | ||
|
|
93c36cf376 | ||
|
|
58e7dfb1d0 | ||
|
|
510cfb5182 | ||
|
|
3ac0b1e301 | ||
|
|
2737a5ac87 |
@@ -30,6 +30,14 @@ env:
|
||||
GOFLAGS: -p=1
|
||||
GOMAXPROCS: "2"
|
||||
|
||||
# A superseded run of the same ref is cancelled instead of queueing behind one
|
||||
# that no longer matters. Verified on this Gitea on 2026-09-17: a queued run
|
||||
# whose ref moved on is cancelled before it ever reaches the runner, while a
|
||||
# run already dispatched there runs to completion.
|
||||
concurrency:
|
||||
group: ${{ gitea.workflow }}-${{ gitea.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: fedora
|
||||
@@ -91,10 +99,14 @@ jobs:
|
||||
# output has to be captured into a variable.
|
||||
env:
|
||||
GOBIN: ${{ gitea.workspace }}/bin
|
||||
run: go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0
|
||||
run: go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0
|
||||
|
||||
- name: Build the decoder
|
||||
run: go build -o bin/interpres-decode ./cmd/interpres-decode
|
||||
|
||||
- name: Compliance suite
|
||||
run: bin/toml-test bin/interpres-decode
|
||||
# interpres implements TOML 1.1, and the suite runs both directions: the decoder
|
||||
# on the valid and invalid corpora, the encoder on the tagged JSON of the valid
|
||||
# one. The mode is pinned so an upstream default change cannot silently move the
|
||||
# corpus.
|
||||
run: bin/toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1
|
||||
|
||||
+208
-1
@@ -9,7 +9,214 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Added
|
||||
|
||||
-
|
||||
- `encoding.TextMarshaler` and `encoding.TextUnmarshaler` are honoured by
|
||||
default, with no option to switch them off. A type that implements them is
|
||||
encoded as a TOML string and decoded from one: `net.IP` becomes
|
||||
`"192.0.2.1"`, and a user type with `MarshalText` or `UnmarshalText` follows.
|
||||
`MarshalTOML` and `UnmarshalTOML` still win over the text methods, and the
|
||||
four date-time types keep their bare timestamp form instead of becoming a
|
||||
quoted string. A struct type that implements the interface now encodes as a
|
||||
string where it was a table before, which is the breaking part of the change.
|
||||
- `time.Duration` is encoded in its canonical Go form as a TOML string,
|
||||
`1h30m0s`, because TOML has no duration type; the decoder reads that string
|
||||
back and still accepts a bare integer as the nanosecond count.
|
||||
- `interpres-decode -encode`, the adapter's other direction: it reads the
|
||||
toml-test tagged JSON from stdin and writes the TOML document it describes.
|
||||
The compliance suite now runs the encoder as well as the decoder, 214
|
||||
encoder cases against the tagged JSON of the valid corpus.
|
||||
- `Encoder.InlineTables(threshold)`: a sub-table whose single-line rendering is
|
||||
at most `threshold` bytes is written as an inline table instead of a header
|
||||
section, which shortens a document of small tables. An array of tables keeps
|
||||
its header form, because its inline form would re-parse as a value array.
|
||||
- `Document` and `ParseMap`: `Parse` now returns a `*Document`, which holds the
|
||||
values together with the key order, whether a table was written as an inline
|
||||
table or under a header, and the comments, with `Keys`, `Entries`, `Get`,
|
||||
`Comments` and `SetComments` to read and write them. `ParseMap` returns the
|
||||
plain `map[string]any` tree, the shape `Parse` used to give. `Marshal` does
|
||||
not accept a `Document`; it writes values, so `doc.Map()` is the way through.
|
||||
- `OffsetDateTime`, the Go type of the offset date-time kind, so that all four
|
||||
TOML date-time kinds have one of their own. `Parse` and `Unmarshal` hand it
|
||||
back where they produced a bare `time.Time` before, and `Marshal` accepts it.
|
||||
Unmarshalling into a struct field of type `time.Time` keeps working, because
|
||||
the plain type takes an offset date-time as it always did; code that asserts
|
||||
the tree's type, and `UnmarshalTOML` implementations that expect a
|
||||
`time.Time`, need the new type.
|
||||
- `Decoder.MaxDepth(depth)` and `Decoder.MaxInputSize(size)` bound the parse a
|
||||
`Decode` performs, and every parse carries a nesting limit in any case
|
||||
(10000 levels, which no hand-written document approaches): a document that
|
||||
nests arrays or inline tables deeper used to run the stack out and is now
|
||||
rejected with a `SyntaxError` naming the limit.
|
||||
|
||||
### Changed
|
||||
|
||||
- The output takes the TOML 1.1 form. A date-time writes its seconds only when
|
||||
the value carries them and drops the trailing zeros of a fractional second,
|
||||
so `07:32:00` is written `07:32` and half a second as `00.5`. Both are the
|
||||
same value, and a document written without seconds now comes back without
|
||||
them. `LocalDateTime.String()`, `LocalTime.String()` and the offset date-time
|
||||
rendering follow the same rule.
|
||||
- An inline table that would pass the hundredth column is written across lines
|
||||
with a trailing comma and one tab of indentation per nesting level, the shape
|
||||
TOML 1.1 allows an inline table to take.
|
||||
- `MarshalTOML` reaches every array element and every field, whatever the Go
|
||||
kind, and its result is normalised like any other value: an element rendering
|
||||
itself as a table keeps the `[[header]]` form, one rendering itself as a
|
||||
scalar turns the array into a value array, and the method runs once per
|
||||
element. It is found on the addressable pointer as well, so a
|
||||
pointer-receiver `MarshalTOML` is called for a field or an element, exactly
|
||||
as `MarshalText` is.
|
||||
- TOML 1.1 is the acceptance contract, and TOML 1.0 is not. The compliance
|
||||
suite runs the 1.1 corpus alone, and the promise that every 1.0 document
|
||||
parses exactly as before is withdrawn. Nothing that parses today stops
|
||||
parsing: the 1.0 valid corpus still passes in full. The documents whose
|
||||
verdict changes are the ones 1.1 relaxed, such as the `\xHH` escape
|
||||
sequences 1.0 rejected.
|
||||
- The module path carries the /v2 suffix the Go toolchain requires of
|
||||
every major version 2 module: imports change to
|
||||
`sourcedock.dev/petrbalvin/interpres/v2`.
|
||||
- Input that is not valid UTF-8 is now rejected where the parser's scan
|
||||
meets the invalid byte, with a `SyntaxError` naming that line, instead of
|
||||
a whole-input check that always reported line 1. Invalid input is still
|
||||
rejected; the reported location is now the byte's own.
|
||||
|
||||
**Performance**
|
||||
|
||||
- Parsing is faster than in 1.1.0 while carrying the new document layer:
|
||||
the suite's representative document decodes at about 79 MB/s with 104
|
||||
allocations per call, and the long array-of-tables document at about
|
||||
106 MB/s against 56 MB/s in 1.1.0, with allocations on that document
|
||||
halved from 67 664 to 31 765. Date-time tokens are validated by a byte
|
||||
scan instead of regular expressions, repeated keys share one string
|
||||
across array-of-tables elements, and per-statement buffers are reused.
|
||||
- Typed decoding is 12 percent faster than in 1.1.0 on the representative
|
||||
document (9792 ns against 11 147 ns) with 24 percent fewer allocations
|
||||
(167 against 220); interface lookups resolve through a cached per-type
|
||||
flag set instead of boxing every value into an interface to ask.
|
||||
- `Marshal` runs at the 1.1.0 speed while emitting the new TOML 1.1 output
|
||||
form, at half the bytes per operation (6170 against 11 348 on the
|
||||
representative document), and writes through a pooled output buffer with
|
||||
a 1 MiB retention cap; repeated marshals keep the live heap flat.
|
||||
- Two benchmarks measure the shapes that drove the work:
|
||||
`BenchmarkStrictDecodeLong` and `BenchmarkMarshalLong` run the 2000-entry
|
||||
document at about 3.8 ms and 3.4 ms per call, at 63 772 and 63 660
|
||||
allocations.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Decoding into a defined type whose underlying kind is string or bool, such
|
||||
as `type Name string`, panicked instead of storing the value, because a
|
||||
value of the predeclared type is not assignable to a defined type and the
|
||||
decoder assigned it without a conversion.
|
||||
- A top-level value the encoder could not normalise reported its path with
|
||||
a leading dot, `interpres: .port: ...`; the message now reads
|
||||
`interpres: port: ...`, the shape `EncodeError.Path` already used.
|
||||
|
||||
## [1.1.0] - 2026-09-18
|
||||
|
||||
### Added
|
||||
|
||||
- TOML 1.1 support, on by default: date-times and times without seconds
|
||||
(`07:32`, `1979-05-27T07:32`, normalised to full seconds on output), the
|
||||
`\e` and `\xHH` escape sequences, and multi-line inline tables with
|
||||
comments and trailing commas. The compliance suite runs in TOML 1.1 mode:
|
||||
214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document
|
||||
parses exactly as before.
|
||||
- `interpres-decode -validate [file ...]`: a validate mode beside the
|
||||
toml-test adapter. It parses each named file, or stdin when none are named,
|
||||
prints one line per invalid document to stderr, and exits 0 when all are
|
||||
valid, 1 when one is not, and 2 on a usage or read failure. Install it with
|
||||
`go install .../cmd/interpres-decode@latest`; releases still ship no
|
||||
binaries.
|
||||
- `DecodeError` and `EncodeError`: decode and encode failures are wrapped in
|
||||
typed errors carrying the key path, read with `errors.AsType` instead of
|
||||
parsing the message text. The rendered messages keep their shape; the only
|
||||
visible change is that an encode failure on a top-level field no longer
|
||||
gains a meaningless leading dot in its path.
|
||||
- `omitzero` and `omitempty` tag options on encode: `toml:"name,omitzero"`
|
||||
skips a field whose value is the zero value of its type (a type with an
|
||||
`IsZero() bool` method decides through the method), and
|
||||
`toml:"name,omitempty"` skips a nil or empty slice, array, or map. The
|
||||
decoder ignores both options.
|
||||
|
||||
### Changed
|
||||
|
||||
- The compliance suite is [toml-test](https://github.com/toml-lang/toml-test)
|
||||
v2.2.0, up from v1.6.0. Its TOML 1.0 corpus holds 205 valid and 474 invalid
|
||||
cases (185 and 371 before), and it caught the two documents the parser
|
||||
still accepted, fixed below.
|
||||
- The flattened struct layout the decoder consults is cached per struct type
|
||||
and shared with the encoder, which now resolves duplicate field keys with
|
||||
it. Strict decoding of an array of tables of structs runs about a quarter
|
||||
faster; marshalling structs gained the same layout without measurable cost.
|
||||
- The parser scans the input bytes in place instead of building a `[]rune`
|
||||
copy of the document: every character that drives the grammar is ASCII and
|
||||
the input is validated UTF-8 up front, so the conversion pass and its four
|
||||
bytes per rune were pure overhead. Parsing a large array-of-tables document
|
||||
runs about a fifth faster and allocates about half the memory.
|
||||
- Numeric tokens without underscores skip the normalising rebuild: digits are
|
||||
validated in place in `joinDigits`, and a float whose token is already
|
||||
clean goes to `strconv.ParseFloat` directly. One allocation per integer
|
||||
atom and two per float atom disappear.
|
||||
|
||||
### Fixed
|
||||
|
||||
- A `MarshalTOML` result of `nil` with a nil error fails the marshal with
|
||||
`MarshalTOML returned a nil value`. The field silently vanished before, and
|
||||
inside a value array the nil result reached reflection as a zero value and
|
||||
panicked.
|
||||
- Strict decoding reports the smallest unknown key. Several unknown keys in
|
||||
one table made the message depend on Go's random map iteration order, so
|
||||
the same document reported different keys across runs.
|
||||
- Decoding into a struct that embeds a pointer to itself terminates. The
|
||||
schema walk recursed through the embedded type forever, so such a
|
||||
`Unmarshal` call hung the process; the walk now tracks the struct types on
|
||||
the current path and stops when one repeats.
|
||||
- An array-of-tables header whose path runs through an inline table
|
||||
(`a = {b = {}}` followed by `[[a.b.c]]`) is rejected. The frozen-inline-table
|
||||
check covered `[table]` headers and dotted keys but not the intermediate
|
||||
steps of an array-of-tables header, so such a document silently extended the
|
||||
inline table.
|
||||
- A new element of an array of tables starts a fresh scope for dotted-key paths
|
||||
and nested arrays of tables: `[[a]]`, `b.c = 1`, `[[a]]`, `[a.b]` parses, as
|
||||
the TOML examples in the spec shape it. The records of the previous element
|
||||
falsely rejected the same paths in the next one.
|
||||
- `Marshal` emits exactly one key when two struct fields resolve to the same
|
||||
TOML name, picking the field the decoder would fill (the shallower one, the
|
||||
later declaration at equal depth). Such a struct previously marshalled into
|
||||
a duplicate key, and the output never re-parsed, breaking the round-trip
|
||||
guarantee.
|
||||
- `Marshal` returns an error for a table header key or an inline-table key that
|
||||
is not valid UTF-8, the way scalar keys already did, instead of silently
|
||||
emitting corrupt TOML (a header that lost its key, an inline table with a
|
||||
missing key).
|
||||
- `UseLiteralMultiline` falls back to the escaped basic string when the value
|
||||
cannot be carried verbatim by the literal form: a run of three single quotes,
|
||||
a control character, or a lone carriage return. Such values previously
|
||||
produced output that did not re-parse.
|
||||
- A `[]any` holding only tables marshals in the value-array form with inline
|
||||
tables, keeping the type `Parse` produces for such an array. It previously
|
||||
took the `[[header]]` form, so a round-trip changed the value's type from
|
||||
`[]any` to `[]map[string]any`.
|
||||
- Decoding into a `uint` destination checks the type's platform width instead
|
||||
of only the fixed widths, so a 32-bit `uint` no longer truncates silently;
|
||||
decoding a finite float beyond the `float32` range is an overflow error
|
||||
instead of a silent infinity.
|
||||
- Struct fields that resolve to one key at equal depth decode through the
|
||||
field declared later, matching the documented rule; the first one won before.
|
||||
- A float with an exponent marker but no digits (`1e`, `0.0E`) is rejected;
|
||||
the exponent requires at least one digit.
|
||||
- A date-time offset outside 00:00 through 23:59 is rejected; such offsets
|
||||
were accepted and silently rolled over (`+00:60` decoded as `+01:00`).
|
||||
- Untagged embedded fields now decode symmetrically with encode: an embedded
|
||||
struct receives its keys inline (a nil embedded pointer struct is
|
||||
allocated), an embedded map catches the keys no field claims, and a name
|
||||
clash resolves in favour of the shallower field. A struct with an untagged
|
||||
embedded field previously decoded with all inline keys dropped and did not
|
||||
round-trip.
|
||||
- `Marshal` re-emits arrays that mix tables with scalars: the table elements
|
||||
render as inline tables inside the value array. A tree that `Parse` accepts
|
||||
from such a document previously failed with
|
||||
`cannot encode map[string]interface {}`.
|
||||
|
||||
## [1.0.0] - 2026-08-20
|
||||
|
||||
|
||||
+29
-8
@@ -1,10 +1,29 @@
|
||||
# Contributing
|
||||
|
||||
Thanks for contributing to **interpres**.
|
||||
Contributions to **interpres** are governed by the Contributor terms
|
||||
below; submitting one means you accept them.
|
||||
|
||||
## Contributor terms
|
||||
|
||||
1. This project belongs to its owner alone. The owner decides what is
|
||||
accepted, in what form and when; the decision is final and needs no
|
||||
justification.
|
||||
2. By submitting a contribution you assign to Petr Balvín
|
||||
<opensource@petrbalvin.org> all present and future copyright and
|
||||
related rights in it, worldwide, for the full term of the rights,
|
||||
with the right to relicense and sublicense without restriction,
|
||||
including under proprietary terms.
|
||||
3. Where that assignment is not effective, it counts as a perpetual,
|
||||
irrevocable, royalty-free licence with the same scope.
|
||||
4. To the fullest extent permitted by law, you waive any right of
|
||||
attribution and integrity in the contribution. The project names no
|
||||
contributors and keeps no credits list.
|
||||
5. By submitting you represent that the work is yours and that you
|
||||
hold the rights to assign it as above.
|
||||
|
||||
## Development setup
|
||||
|
||||
Requirements: Go 1.27.0, the version `go.mod` declares, and
|
||||
Requirements: Go 1.27.1, the version `go.mod` declares, and
|
||||
[just](https://github.com/casey/just) for the recipes.
|
||||
|
||||
```sh
|
||||
@@ -26,17 +45,19 @@ just test
|
||||
formatting pass are three commits, never one.
|
||||
4. Record every user-visible change in `CHANGELOG.md` under `## [development]`.
|
||||
5. Add or update tests. Coverage stays at 80 percent or more; it is a hard
|
||||
gate. Parser and decoder changes must also keep the toml-test suite at zero
|
||||
failures, checked with `just toml-test`.
|
||||
gate. Parser, decoder and encoder changes must also keep both directions of
|
||||
the toml-test suite at zero failures, checked with `just toml-test`.
|
||||
6. Update the documentation when the public API, the configuration or the
|
||||
behaviour changes; the documents move in the same commit as the behaviour
|
||||
they describe.
|
||||
7. Open a pull request against `development`.
|
||||
|
||||
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The
|
||||
release workflow runs the full gate set including the race detector and
|
||||
publishes the Gitea release with the matching `CHANGELOG.md` section as its
|
||||
notes.
|
||||
release workflow validates the tag, runs the static gates and the test suite
|
||||
with the coverage floor, and publishes the Gitea release with the matching
|
||||
`CHANGELOG.md` section as its notes. The race detector is not in that set: race
|
||||
never runs on a push path, and the local `just gates` raced the tree before the
|
||||
tag was cut.
|
||||
|
||||
## Code style
|
||||
|
||||
@@ -95,7 +116,7 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
|
||||
|---|---|---|
|
||||
| Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite |
|
||||
| Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs |
|
||||
| Release | a `v*` tag | the same gates plus the race detector, then the Gitea release created from the `CHANGELOG.md` section |
|
||||
| Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector |
|
||||
|
||||
The local equivalent is `just gates`, which is the same set plus the race
|
||||
detector.
|
||||
|
||||
@@ -1,37 +1,44 @@
|
||||
# interpres
|
||||
|
||||
A TOML 1.0 parser and encoder for Go, written with the standard library alone.
|
||||
`interpres` (Latin for *interpreter*) gives zero-dependency programs an
|
||||
`encoding/json`-style API for reading and writing TOML, and passes the entire
|
||||
official [toml-test](https://github.com/toml-lang/toml-test) suite: 185 valid
|
||||
and 371 invalid cases, zero failures.
|
||||
A TOML 1.1 parser and encoder for Go, written with the standard library
|
||||
alone. `interpres` (Latin for *interpreter*) gives zero-dependency
|
||||
programs an `encoding/json`-style API for reading and writing TOML, and passes
|
||||
the entire official [toml-test](https://github.com/toml-lang/toml-test) suite:
|
||||
214 valid, 467 invalid and 214 encoder cases, zero failures.
|
||||
|
||||
## Features
|
||||
|
||||
- **Full TOML 1.0**: bare, quoted and dotted keys; tables and arrays of tables;
|
||||
basic and literal strings including multiline; integers in the four radixes
|
||||
with `_` separators; floats with exponents, `inf` and `nan`; booleans; the
|
||||
four date-time kinds; arrays and inline tables.
|
||||
- **Full TOML 1.1**: bare, quoted and dotted keys; tables and arrays of
|
||||
tables; basic and literal strings including multiline, with the 1.1 `\e` and
|
||||
`\xHH` escapes; integers in the four radixes with `_` separators; floats with
|
||||
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds
|
||||
optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
|
||||
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and
|
||||
`Marshal` for structs and maps, mirroring `encoding/json`.
|
||||
- **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that
|
||||
match no destination field, at every struct depth.
|
||||
- **Custom types**: `Marshaler` and `Unmarshaler` let a type control its own
|
||||
TOML representation in both directions.
|
||||
TOML representation in both directions, and `encoding.TextMarshaler` and
|
||||
`TextUnmarshaler` are honoured by default, so `net.IP`, `time.Duration` and
|
||||
user types with text methods need no configuration.
|
||||
- **Cancellation**: every entry point has a `*Context` sibling that honours a
|
||||
`context.Context`.
|
||||
- **Ordered documents**: `Parse` gives a `*Document` that keeps the key order,
|
||||
tells an inline table from a header one, and carries the comments; `ParseMap`
|
||||
gives the plain `map[string]any` tree.
|
||||
- **Configurable emission**: `Encoder` options for declaration-order output,
|
||||
omitting empty arrays, and literal multiline strings.
|
||||
omitting empty arrays, literal multiline strings, and inlining small
|
||||
sub-tables.
|
||||
|
||||
## Install
|
||||
|
||||
As a library:
|
||||
|
||||
```sh
|
||||
go get sourcedock.dev/petrbalvin/interpres
|
||||
go get sourcedock.dev/petrbalvin/interpres/v2
|
||||
```
|
||||
|
||||
Requires Go 1.27.0 or newer. The module imports only the standard library.
|
||||
Requires Go 1.27.1 or newer. The module imports only the standard library.
|
||||
|
||||
## Quick start
|
||||
|
||||
@@ -63,8 +70,9 @@ err := interpres.Unmarshal(data, &cfg)
|
||||
```
|
||||
|
||||
Fields match by the `toml:"name"` tag, or by the lower-cased field name when no
|
||||
tag is present; `toml:"-"` skips a field. `Parse` returns the untyped
|
||||
`map[string]any` tree instead, and `UnmarshalContext` accepts a context.
|
||||
tag is present; `toml:"-"` skips a field. `Parse` returns a `*Document` that
|
||||
also carries the key order and the comments, `ParseMap` returns the plain
|
||||
`map[string]any` tree, and `UnmarshalContext` accepts a context.
|
||||
|
||||
### Encode from a struct
|
||||
|
||||
@@ -160,7 +168,7 @@ See [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the full workflow, and
|
||||
|
||||
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
|
||||
- [docs/API.md](docs/API.md): the API reference, decoding and encoding rules
|
||||
- [docs/CLI.md](docs/CLI.md): the interpres-decode toml-test adapter
|
||||
- [docs/CLI.md](docs/CLI.md): the interpres-decode adapter and validator
|
||||
|
||||
## Licence
|
||||
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@ releases do not receive them.
|
||||
|
||||
| Version | Supported |
|
||||
|---|---|
|
||||
| 1.0.0 | yes |
|
||||
| 1.1.0 | yes |
|
||||
| older releases | no |
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
+170
@@ -0,0 +1,170 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package interpres
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// benchDoc is a representative configuration document: every scalar kind, an
|
||||
// inline table, sub-tables, and an array of tables.
|
||||
var benchDoc = []byte(`title = "benchmark configuration"
|
||||
replicas = 3
|
||||
ratio = 0.75
|
||||
enabled = true
|
||||
when = 2026-09-17T12:00:00Z
|
||||
local = 2026-09-17T12:00:00
|
||||
tags = ["alpha", "beta", "gamma"]
|
||||
limits = { cpu = 4, memory = 1024 }
|
||||
|
||||
[server]
|
||||
host = "localhost"
|
||||
port = 8080
|
||||
hosts = ["a.example", "b.example"]
|
||||
|
||||
[server.tls]
|
||||
enabled = true
|
||||
cert = "/etc/cert.pem"
|
||||
|
||||
[[items]]
|
||||
name = "first"
|
||||
weight = 10
|
||||
flags = ["x", "y"]
|
||||
|
||||
[[items]]
|
||||
name = "second"
|
||||
weight = 20
|
||||
flags = ["z"]
|
||||
`)
|
||||
|
||||
// longDoc is generated once so the large-input benchmarks measure parsing,
|
||||
// not document construction. Roughly 2000 array-of-tables entries.
|
||||
var longDoc = func() []byte {
|
||||
var b strings.Builder
|
||||
b.WriteString("title = \"long\"\n")
|
||||
for i := range 2000 {
|
||||
fmt.Fprintf(&b, "[[entry]]\nname = \"entry-%d\"\nweight = %d\nwhen = 2026-09-17T12:00:00Z\nratio = 0.5\ntags = [\"a\", \"b\", \"c\"]\n\n", i, i)
|
||||
}
|
||||
return []byte(b.String())
|
||||
}()
|
||||
|
||||
type benchTLS struct {
|
||||
Enabled bool `toml:"enabled"`
|
||||
Cert string `toml:"cert"`
|
||||
}
|
||||
|
||||
type benchServer struct {
|
||||
Host string `toml:"host"`
|
||||
Port int `toml:"port"`
|
||||
Hosts []string `toml:"hosts"`
|
||||
TLS benchTLS `toml:"tls"`
|
||||
}
|
||||
|
||||
type benchItem struct {
|
||||
Name string `toml:"name"`
|
||||
Weight int `toml:"weight"`
|
||||
Flags []string `toml:"flags"`
|
||||
}
|
||||
|
||||
type benchConfig struct {
|
||||
Title string `toml:"title"`
|
||||
Replicas int `toml:"replicas"`
|
||||
Ratio float64 `toml:"ratio"`
|
||||
Enabled bool `toml:"enabled"`
|
||||
When time.Time `toml:"when"`
|
||||
Local LocalDateTime `toml:"local"`
|
||||
Tags []string `toml:"tags"`
|
||||
Limits map[string]any `toml:"limits"`
|
||||
Server benchServer `toml:"server"`
|
||||
Items []benchItem `toml:"items"`
|
||||
}
|
||||
|
||||
func BenchmarkParse(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
b.SetBytes(int64(len(benchDoc)))
|
||||
for b.Loop() {
|
||||
if _, err := ParseMap(benchDoc); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkMarshal(b *testing.B) {
|
||||
tree, err := ParseMap(benchDoc)
|
||||
if err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
b.ReportAllocs()
|
||||
b.SetBytes(int64(len(benchDoc)))
|
||||
for b.Loop() {
|
||||
if _, err := Marshal(tree); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkStrictDecode(b *testing.B) {
|
||||
dec := NewDecoder().DisallowUnknownFields()
|
||||
b.ReportAllocs()
|
||||
for b.Loop() {
|
||||
var cfg benchConfig
|
||||
if err := dec.Decode(benchDoc, &cfg); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkParseLong(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
b.SetBytes(int64(len(longDoc)))
|
||||
for b.Loop() {
|
||||
if _, err := ParseMap(longDoc); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// benchLongEntry mirrors one [[entry]] element of longDoc for the typed
|
||||
// decode of the long document.
|
||||
type benchLongEntry struct {
|
||||
Name string `toml:"name"`
|
||||
Weight int `toml:"weight"`
|
||||
When time.Time `toml:"when"`
|
||||
Ratio float64 `toml:"ratio"`
|
||||
Tags []string `toml:"tags"`
|
||||
}
|
||||
|
||||
type benchLongDoc struct {
|
||||
Title string `toml:"title"`
|
||||
Entry []benchLongEntry `toml:"entry"`
|
||||
}
|
||||
|
||||
func BenchmarkStrictDecodeLong(b *testing.B) {
|
||||
dec := NewDecoder().DisallowUnknownFields()
|
||||
b.ReportAllocs()
|
||||
b.SetBytes(int64(len(longDoc)))
|
||||
for b.Loop() {
|
||||
var doc benchLongDoc
|
||||
if err := dec.Decode(longDoc, &doc); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkMarshalLong(b *testing.B) {
|
||||
tree, err := ParseMap(longDoc)
|
||||
if err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
b.ReportAllocs()
|
||||
b.SetBytes(int64(len(longDoc)))
|
||||
for b.Loop() {
|
||||
if _, err := Marshal(tree); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
+263
-12
@@ -1,17 +1,25 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
// Command interpres-decode reads a TOML document from standard input and writes
|
||||
// the toml-test "tagged JSON" representation to standard output.
|
||||
// Command interpres-decode is the toml-test harness adapter and a TOML
|
||||
// validator. Without flags it reads a TOML document from standard input and
|
||||
// writes the toml-test "tagged JSON" representation to standard output. With
|
||||
// -encode it is the reverse: it reads tagged JSON and writes the TOML document
|
||||
// it describes. With -validate it checks the named documents, or standard
|
||||
// input when none are named, and exits non-zero on the first invalid one:
|
||||
//
|
||||
// It exits non-zero on a parse error, which is how the toml-test harness checks
|
||||
// that invalid documents are rejected. Run the official suite against it with:
|
||||
// interpres-decode -validate config.toml
|
||||
// interpres-decode -encode < case.json
|
||||
//
|
||||
// toml-test ./interpres-decode
|
||||
// Run the official suite in both directions against the adapter with:
|
||||
//
|
||||
// toml-test test -decoder=./interpres-decode -encoder='./interpres-decode -encode'
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
@@ -19,23 +27,47 @@ import (
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"sourcedock.dev/petrbalvin/interpres"
|
||||
"sourcedock.dev/petrbalvin/interpres/v2"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(Run(os.Stdin, os.Stdout, os.Stderr))
|
||||
os.Exit(Run(os.Args[1:], os.Stdin, os.Stdout, os.Stderr))
|
||||
}
|
||||
|
||||
// Run reads a TOML document from stdin, emits the toml-test tagged-JSON form
|
||||
// on stdout, and returns the process exit code (0 success, 1 parse error,
|
||||
// 2 I/O, encoding, or unsupported-value error).
|
||||
func Run(stdin io.Reader, stdout, stderr io.Writer) int {
|
||||
// Run runs the command line and returns the process exit code: 0 success,
|
||||
// 1 an invalid document, 2 a usage, reading, encoding, or
|
||||
// unsupported-value error.
|
||||
func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON")
|
||||
encode := fs.Bool("encode", false, "read tagged JSON from stdin and write TOML instead")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if *validate && *encode {
|
||||
fmt.Fprintln(stderr, "interpres-decode: -validate and -encode cannot be combined")
|
||||
return 2
|
||||
}
|
||||
if *validate {
|
||||
return validatePaths(fs.Args(), stdin, stderr)
|
||||
}
|
||||
if fs.NArg() > 0 {
|
||||
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with -validate")
|
||||
return 2
|
||||
}
|
||||
if *encode {
|
||||
return encodeJSON(stdin, stdout, stderr)
|
||||
}
|
||||
data, err := io.ReadAll(stdin)
|
||||
if err != nil {
|
||||
fmt.Fprintln(stderr, "read stdin:", err)
|
||||
return 2
|
||||
}
|
||||
tree, err := interpres.Parse(data)
|
||||
tree, err := interpres.ParseMap(data)
|
||||
if err != nil {
|
||||
fmt.Fprintln(stderr, err)
|
||||
return 1
|
||||
@@ -54,6 +86,223 @@ func Run(stdin io.Reader, stdout, stderr io.Writer) int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// validatePaths parses every named file, or standard input when none are
|
||||
// named, and reports each invalid document on stderr. It returns 0 when all
|
||||
// documents parse, 1 when one does not, and 2 on a usage or read failure.
|
||||
func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
|
||||
if len(paths) == 0 {
|
||||
paths = []string{"-"}
|
||||
}
|
||||
valid := true
|
||||
for _, p := range paths {
|
||||
name := p
|
||||
var data []byte
|
||||
var err error
|
||||
if p == "-" {
|
||||
data, err = io.ReadAll(stdin)
|
||||
name = "<stdin>"
|
||||
} else {
|
||||
data, err = os.ReadFile(p)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
|
||||
return 2
|
||||
}
|
||||
if _, err := interpres.ParseMap(data); err != nil {
|
||||
fmt.Fprintf(stderr, "%s: %v\n", name, err)
|
||||
valid = false
|
||||
}
|
||||
}
|
||||
if !valid {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// encodeJSON reads a toml-test tagged JSON description from standard input and
|
||||
// writes the TOML document it describes to standard output.
|
||||
func encodeJSON(stdin io.Reader, stdout, stderr io.Writer) int {
|
||||
data, err := io.ReadAll(stdin)
|
||||
if err != nil {
|
||||
fmt.Fprintln(stderr, "read stdin:", err)
|
||||
return 2
|
||||
}
|
||||
var desc any
|
||||
if err := json.Unmarshal(data, &desc); err != nil {
|
||||
fmt.Fprintln(stderr, "decode JSON:", err)
|
||||
return 2
|
||||
}
|
||||
tree, err := untag(desc)
|
||||
if err != nil {
|
||||
fmt.Fprintln(stderr, err)
|
||||
return 2
|
||||
}
|
||||
doc, ok := tree.(map[string]any)
|
||||
if !ok {
|
||||
fmt.Fprintln(stderr, "interpres-decode: the description must be a JSON object at the top level")
|
||||
return 2
|
||||
}
|
||||
out, err := interpres.Marshal(doc)
|
||||
if err != nil {
|
||||
fmt.Fprintln(stderr, err)
|
||||
return 2
|
||||
}
|
||||
if _, err := stdout.Write(out); err != nil {
|
||||
fmt.Fprintln(stderr, "write stdout:", err)
|
||||
return 2
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// untag converts a toml-test JSON description into the value tree Marshal
|
||||
// expects: a JSON object becomes a map[string]any, a JSON array becomes a
|
||||
// []any, and an object carrying exactly the keys "type" and "value" becomes
|
||||
// the Go value for that TOML type.
|
||||
func untag(v any) (any, error) {
|
||||
switch x := v.(type) {
|
||||
case map[string]any:
|
||||
if typ, val, ok := taggedValue(x); ok {
|
||||
return decodeTagged(typ, val)
|
||||
}
|
||||
out := make(map[string]any, len(x))
|
||||
for k, e := range x {
|
||||
u, err := untag(e)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", k, err)
|
||||
}
|
||||
out[k] = u
|
||||
}
|
||||
return out, nil
|
||||
case []any:
|
||||
out := make([]any, len(x))
|
||||
for i, e := range x {
|
||||
u, err := untag(e)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("[%d]: %w", i, err)
|
||||
}
|
||||
out[i] = u
|
||||
}
|
||||
return asTables(out), nil
|
||||
default:
|
||||
return nil, fmt.Errorf("unsupported JSON value %T", v)
|
||||
}
|
||||
}
|
||||
|
||||
// asTables returns the elements as a []map[string]any when there is at least
|
||||
// one and every element is a table, the shape the encoder renders as an array
|
||||
// of tables. The tagged JSON cannot tell an array of tables from a value array
|
||||
// of inline tables, and both parse back to the same value, so the header form
|
||||
// is chosen because it is the one the encoder otherwise never exercises. An
|
||||
// empty array stays a []any, because TOML has no empty array of tables.
|
||||
func asTables(items []any) any {
|
||||
if len(items) == 0 {
|
||||
return items
|
||||
}
|
||||
tbls := make([]map[string]any, len(items))
|
||||
for i, e := range items {
|
||||
tbl, ok := e.(map[string]any)
|
||||
if !ok {
|
||||
return items
|
||||
}
|
||||
tbls[i] = tbl
|
||||
}
|
||||
return tbls
|
||||
}
|
||||
|
||||
// taggedValue reports whether m is a toml-test value object: a JSON object of
|
||||
// exactly the two string keys "type" and "value", carrying a type this adapter
|
||||
// knows. Any other object is a table.
|
||||
func taggedValue(m map[string]any) (typ, val string, ok bool) {
|
||||
if len(m) != 2 {
|
||||
return "", "", false
|
||||
}
|
||||
ts, ok := m["type"].(string)
|
||||
if !ok || !knownType(ts) {
|
||||
return "", "", false
|
||||
}
|
||||
vs, ok := m["value"].(string)
|
||||
if !ok {
|
||||
return "", "", false
|
||||
}
|
||||
return ts, vs, true
|
||||
}
|
||||
|
||||
func knownType(typ string) bool {
|
||||
switch typ {
|
||||
case "string", "integer", "float", "bool",
|
||||
"datetime", "datetime-local", "date-local", "time-local":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// decodeTagged returns the Go value for one tagged JSON value. Every type but
|
||||
// string is parsed by the library itself, so the adapter and the library agree
|
||||
// on what an integer, a float or a date-time is.
|
||||
func decodeTagged(typ, val string) (any, error) {
|
||||
if typ == "string" {
|
||||
return val, nil
|
||||
}
|
||||
v, err := parseAtom(val)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s %q: %w", typ, val, err)
|
||||
}
|
||||
// A float with no fractional part and no exponent is described by a bare
|
||||
// integer literal, so here the tag decides and not the literal.
|
||||
if n, ok := v.(int64); ok && typ == "float" {
|
||||
return float64(n), nil
|
||||
}
|
||||
if !typeMatches(typ, v) {
|
||||
return nil, fmt.Errorf("%s %q parsed as %T", typ, val, v)
|
||||
}
|
||||
return v, nil
|
||||
}
|
||||
|
||||
// parseAtom parses one bare TOML value, by handing `v = <val>` to the library's
|
||||
// parser and requiring the result to hold exactly that one statement, so a
|
||||
// value carrying a newline or a comment cannot smuggle a second one in.
|
||||
func parseAtom(val string) (any, error) {
|
||||
tree, err := interpres.ParseMap([]byte("v = " + val + "\n"))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(tree) != 1 {
|
||||
return nil, errors.New("not a single bare value")
|
||||
}
|
||||
return tree["v"], nil
|
||||
}
|
||||
|
||||
// typeMatches reports whether v is the Go value the tagged type names.
|
||||
func typeMatches(typ string, v any) bool {
|
||||
switch typ {
|
||||
case "integer":
|
||||
_, ok := v.(int64)
|
||||
return ok
|
||||
case "float":
|
||||
_, ok := v.(float64)
|
||||
return ok
|
||||
case "bool":
|
||||
_, ok := v.(bool)
|
||||
return ok
|
||||
case "datetime":
|
||||
switch v.(type) {
|
||||
case time.Time, interpres.OffsetDateTime:
|
||||
return true
|
||||
}
|
||||
return false
|
||||
case "datetime-local":
|
||||
_, ok := v.(interpres.LocalDateTime)
|
||||
return ok
|
||||
case "date-local":
|
||||
_, ok := v.(interpres.LocalDate)
|
||||
return ok
|
||||
case "time-local":
|
||||
_, ok := v.(interpres.LocalTime)
|
||||
return ok
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// tag converts an interpres value into its toml-test tagged-JSON form. Tables
|
||||
// become JSON objects and arrays become JSON arrays; scalars are wrapped in a
|
||||
// {"type", "value"} object. An error is returned for value types the encoder
|
||||
@@ -100,6 +349,8 @@ func tag(v any) (any, error) {
|
||||
return tagged("float", formatFloat(x)), nil
|
||||
case time.Time:
|
||||
return tagged("datetime", x.Format(time.RFC3339Nano)), nil
|
||||
case interpres.OffsetDateTime:
|
||||
return tagged("datetime", x.Format(time.RFC3339Nano)), nil
|
||||
case interpres.LocalDateTime:
|
||||
return tagged("datetime-local", x.Format("2006-01-02T15:04:05.999999999")), nil
|
||||
case interpres.LocalDate:
|
||||
|
||||
@@ -7,11 +7,13 @@ import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"os"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"sourcedock.dev/petrbalvin/interpres"
|
||||
"sourcedock.dev/petrbalvin/interpres/v2"
|
||||
)
|
||||
|
||||
func TestRunParsesValidTOML(t *testing.T) {
|
||||
@@ -20,7 +22,7 @@ func TestRunParsesValidTOML(t *testing.T) {
|
||||
port = 8080
|
||||
enabled = true
|
||||
`))
|
||||
if code := Run(in, &stdout, &stderr); code != 0 {
|
||||
if code := Run(nil, in, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
var got map[string]any
|
||||
@@ -41,7 +43,7 @@ enabled = true
|
||||
func TestRunRejectsInvalidInput(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
in := bytes.NewReader([]byte("v = \n"))
|
||||
code := Run(in, &stdout, &stderr)
|
||||
code := Run(nil, in, &stdout, &stderr)
|
||||
if code != 1 {
|
||||
t.Errorf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String())
|
||||
}
|
||||
@@ -52,7 +54,7 @@ func TestRunRejectsInvalidInput(t *testing.T) {
|
||||
|
||||
func TestRunReadErrorReturnsTwo(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := Run(errorReader{}, &stdout, &stderr)
|
||||
code := Run(nil, errorReader{}, &stdout, &stderr)
|
||||
if code != 2 {
|
||||
t.Errorf("Run returned %d, want 2 (read error); stderr = %q", code, stderr.String())
|
||||
}
|
||||
@@ -70,7 +72,7 @@ func TestRunEncodeErrorReturnsTwo(t *testing.T) {
|
||||
var stderr bytes.Buffer
|
||||
w := errorWriter{}
|
||||
in := bytes.NewReader([]byte(`k = "v"` + "\n"))
|
||||
code := Run(in, w, &stderr)
|
||||
code := Run(nil, in, w, &stderr)
|
||||
if code != 2 {
|
||||
t.Errorf("Run returned %d, want 2 (encode error); stderr = %q", code, stderr.String())
|
||||
}
|
||||
@@ -210,3 +212,237 @@ func TestTaggedHelper(t *testing.T) {
|
||||
t.Errorf("tagged = %#v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateStdinAcceptsValidDocument(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
in := bytes.NewReader([]byte("title = \"ok\"\n"))
|
||||
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
if stdout.Len() != 0 || stderr.Len() != 0 {
|
||||
t.Fatalf("validate should be quiet on success, stdout %q stderr %q", stdout.String(), stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateStdinRejectsInvalidDocument(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
in := bytes.NewReader([]byte("title = \"unterminated\n"))
|
||||
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") {
|
||||
t.Fatalf("stderr = %q, want the name and the line", stderr.String())
|
||||
}
|
||||
if stdout.Len() != 0 {
|
||||
t.Fatalf("stdout should stay empty, got %q", stdout.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateFiles(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
good := dir + "/good.toml"
|
||||
bad := dir + "/bad.toml"
|
||||
if err := os.WriteFile(good, []byte("a = 1\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := os.WriteFile(bad, []byte("a =\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := Run([]string{"-validate", good}, nil, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
if code := Run([]string{"-validate", good, bad}, nil, &stdout, &stderr); code != 1 {
|
||||
t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") {
|
||||
t.Fatalf("stderr = %q, want the file name and the line", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateMissingFileReturnsTwo(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := Run([]string{"-validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestAdapterModeRejectsPositionalArgument(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
in := bytes.NewReader([]byte("a = 1\n"))
|
||||
if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "-validate") {
|
||||
t.Fatalf("stderr = %q, want it to point at -validate", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnknownFlagReturnsTwo(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := Run([]string{"-nope"}, nil, &stdout, &stderr); code != 2 {
|
||||
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
// --- encoder mode ----------------------------------------------------------
|
||||
|
||||
func TestRunEncoderScalars(t *testing.T) {
|
||||
in := `{
|
||||
"s": {"type": "string", "value": "quote \" and backslash \\"},
|
||||
"nl": {"type": "string", "value": "line1\nline2"},
|
||||
"i": {"type": "integer", "value": "-9223372036854775808"},
|
||||
"g": {"type": "float", "value": "1.5"},
|
||||
"f": {"type": "float", "value": "inf"},
|
||||
"b": {"type": "bool", "value": "false"},
|
||||
"dt": {"type": "datetime", "value": "1979-05-27T07:32:00-07:00"},
|
||||
"ldt": {"type": "datetime-local", "value": "1979-05-27T07:32:00"},
|
||||
"ld": {"type": "date-local", "value": "1979-05-27"},
|
||||
"lt": {"type": "time-local", "value": "07:32:00.999"}
|
||||
}
|
||||
`
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
|
||||
if code != 0 {
|
||||
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
want := "b = false\n" +
|
||||
"dt = 1979-05-27T07:32-07:00\n" +
|
||||
"f = inf\n" +
|
||||
"g = 1.5\n" +
|
||||
"i = -9223372036854775808\n" +
|
||||
"ld = 1979-05-27\n" +
|
||||
"ldt = 1979-05-27T07:32\n" +
|
||||
"lt = 07:32:00.999\n" +
|
||||
"nl = \"line1\\nline2\"\n" +
|
||||
"s = \"quote \\\" and backslash \\\\\"\n"
|
||||
if stdout.String() != want {
|
||||
t.Errorf("output mismatch:\ngot: %q\nwant: %q", stdout.String(), want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunEncoderNested(t *testing.T) {
|
||||
in := `{
|
||||
"tbl": {"x": {"type": "bool", "value": "true"},
|
||||
"sub": {"y": {"type": "integer", "value": "1"}}},
|
||||
"items": [{"n": {"type": "string", "value": "a"}},
|
||||
{"n": {"type": "string", "value": "b"}}],
|
||||
"list": [{"type": "integer", "value": "1"}, {"type": "string", "value": "two"}],
|
||||
"emptyTbl": {},
|
||||
"emptyArr": []
|
||||
}
|
||||
`
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
|
||||
if code != 0 {
|
||||
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
want := "emptyArr = []\n" +
|
||||
"list = [1, \"two\"]\n" +
|
||||
"\n[emptyTbl]\n" +
|
||||
"\n[tbl]\nx = true\n" +
|
||||
"\n[tbl.sub]\ny = 1\n" +
|
||||
"\n[[items]]\nn = \"a\"\n" +
|
||||
"\n[[items]]\nn = \"b\"\n"
|
||||
if stdout.String() != want {
|
||||
t.Errorf("output mismatch:\ngot: %q\nwant: %q", stdout.String(), want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunEncoderFloatTagDecides(t *testing.T) {
|
||||
// A float with no fraction is described by a bare integer literal, so the
|
||||
// tag decides the type; the output must stay a float.
|
||||
var stdout, stderr bytes.Buffer
|
||||
in := `{"whole": {"type": "float", "value": "1"}, "exp": {"type": "float", "value": "5e+22"}}`
|
||||
code := Run([]string{"-encode"}, strings.NewReader(in), &stdout, &stderr)
|
||||
if code != 0 {
|
||||
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
if want := "exp = 5e+22\nwhole = 1.0\n"; stdout.String() != want {
|
||||
t.Errorf("output mismatch:\ngot: %q\nwant: %q", stdout.String(), want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunEncoderRejectsBadInput(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in string
|
||||
want string
|
||||
}{
|
||||
{"not-json", "not json", "decode JSON"},
|
||||
{"top-level-array", `[{"type": "integer", "value": "1"}]`, "must be a JSON object"},
|
||||
{"untagged-scalar", `{"x": 1}`, "unsupported JSON value"},
|
||||
{"literal-mismatch", `{"x": {"type": "integer", "value": "1.5"}}`, "parsed as float64"},
|
||||
{"offset-for-local", `{"x": {"type": "datetime-local", "value": "1979-05-27T07:32:00Z"}}`, "parsed as interpres.OffsetDateTime"},
|
||||
{"bad-literal", `{"x": {"type": "date-local", "value": "nope"}}`, "date-local"},
|
||||
{"smuggled-statement", `{"x": {"type": "integer", "value": "1\nx = 2"}}`, "not a single bare value"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
var stdout, stderr bytes.Buffer
|
||||
code := Run([]string{"-encode"}, strings.NewReader(c.in), &stdout, &stderr)
|
||||
if code != 2 {
|
||||
t.Errorf("%s: Run returned %d, want 2; stderr = %q", c.name, code, stderr.String())
|
||||
continue
|
||||
}
|
||||
if !strings.Contains(stderr.String(), c.want) {
|
||||
t.Errorf("%s: stderr = %q, want it to mention %q", c.name, stderr.String(), c.want)
|
||||
}
|
||||
if stdout.Len() != 0 {
|
||||
t.Errorf("%s: stdout should be empty, got %q", c.name, stdout.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunEncoderFlagConflicts(t *testing.T) {
|
||||
var stdout, stderr bytes.Buffer
|
||||
if code := Run([]string{"-encode", "-validate"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
|
||||
t.Errorf("Run returned %d, want 2 for the two modes together", code)
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "cannot be combined") {
|
||||
t.Errorf("stderr = %q, want it to explain the conflict", stderr.String())
|
||||
}
|
||||
|
||||
stdout.Reset()
|
||||
stderr.Reset()
|
||||
if code := Run([]string{"-encode", "file.json"}, strings.NewReader(""), &stdout, &stderr); code != 2 {
|
||||
t.Errorf("Run returned %d, want 2 for an argument", code)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEncodeAfterDecodeRoundTrip(t *testing.T) {
|
||||
doc := `title = "x"
|
||||
flt = 1.5
|
||||
whole = 7.0
|
||||
big = 9223372036854775807
|
||||
when = 1979-05-27T07:32:00-07:00
|
||||
day = 1979-05-27
|
||||
clock = 07:32:00.999
|
||||
list = [1, "two"]
|
||||
multi = "a\nb"
|
||||
|
||||
[tbl]
|
||||
x = true
|
||||
|
||||
[[items]]
|
||||
n = "a"
|
||||
`
|
||||
var tagged, stderr bytes.Buffer
|
||||
if code := Run(nil, strings.NewReader(doc), &tagged, &stderr); code != 0 {
|
||||
t.Fatalf("decode returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
var out bytes.Buffer
|
||||
if code := Run([]string{"-encode"}, bytes.NewReader(tagged.Bytes()), &out, &stderr); code != 0 {
|
||||
t.Fatalf("encode returned %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
want, err := interpres.ParseMap([]byte(doc))
|
||||
if err != nil {
|
||||
t.Fatalf("parse of the original: %v", err)
|
||||
}
|
||||
got, err := interpres.ParseMap(out.Bytes())
|
||||
if err != nil {
|
||||
t.Fatalf("parse of the encoder output (%q): %v", out.String(), err)
|
||||
}
|
||||
if !reflect.DeepEqual(want, got) {
|
||||
t.Errorf("round trip changed the document:\noriginal: %#v\nencoded: %#v\noutput: %q", want, got, out.String())
|
||||
}
|
||||
}
|
||||
|
||||
+233
-58
@@ -5,14 +5,19 @@ package interpres
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TOML distinguishes four date-time kinds. interpres decodes an offset
|
||||
// date-time to a plain time.Time (it carries a zone), and uses the wrapper
|
||||
// types below for the local variants so callers can tell them apart.
|
||||
// TOML distinguishes four date-time kinds, and each has its own Go type:
|
||||
// OffsetDateTime for the offset kind, and the local wrappers below for the
|
||||
// three that carry no offset. A plain time.Time is accepted wherever an
|
||||
// offset date-time is, on both the encoding and the decoding side, so a
|
||||
// timestamp field does not have to name the wrapper.
|
||||
|
||||
// OffsetDateTime is a TOML offset date-time, e.g. 1979-05-27T07:32:00-07:00.
|
||||
// The embedded time.Time is the instant, with the offset the document wrote.
|
||||
type OffsetDateTime struct{ time.Time }
|
||||
|
||||
// LocalDateTime is a TOML local date-time with no offset, e.g.
|
||||
// 1979-05-27T07:32:00. The embedded time.Time is in UTC.
|
||||
@@ -26,60 +31,210 @@ type LocalDate struct{ time.Time }
|
||||
// The embedded time.Time uses the zero date.
|
||||
type LocalTime struct{ time.Time }
|
||||
|
||||
// String returns the TOML-canonical rendering of the offset date-time, e.g.
|
||||
// "1979-05-27T07:32Z" or "1979-05-27T07:32:00-07:00". The seconds appear only
|
||||
// when the value carries them, a fractional second drops its trailing zeros,
|
||||
// and an offset of zero is written "Z".
|
||||
func (odt OffsetDateTime) String() string { return offsetString(odt.Time) }
|
||||
|
||||
// String returns the TOML-canonical rendering of the local date-time, e.g.
|
||||
// "1979-05-27T07:32:00" or "...:00.000000123" when the time has a fractional
|
||||
// second. The fractional component is zero-padded to nanosecond precision.
|
||||
// "1979-05-27T07:32" or "1979-05-27T07:32:00.5" when the time carries a
|
||||
// fractional second. TOML 1.1 makes the seconds optional, so they appear only
|
||||
// when they are non-zero, and a fraction drops its trailing zeros.
|
||||
func (ldt LocalDateTime) String() string {
|
||||
base := ldt.Format("2006-01-02T15:04:05")
|
||||
if ns := ldt.Nanosecond(); ns > 0 {
|
||||
return base + "." + fmt.Sprintf("%09d", ns)
|
||||
}
|
||||
return base
|
||||
buf := ldt.Time.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
|
||||
return string(appendClock(buf, ldt.Time))
|
||||
}
|
||||
|
||||
// String returns the TOML-canonical rendering of the local date, e.g.
|
||||
// "1979-05-27".
|
||||
func (ld LocalDate) String() string { return ld.Format("2006-01-02") }
|
||||
|
||||
// String returns the TOML-canonical rendering of the local time, e.g.
|
||||
// "07:32:00" or "...:00.000000123" when the time has a fractional second.
|
||||
// The fractional component is zero-padded to nanosecond precision.
|
||||
func (lt LocalTime) String() string {
|
||||
base := lt.Format("15:04:05")
|
||||
if ns := lt.Nanosecond(); ns > 0 {
|
||||
return base + "." + fmt.Sprintf("%09d", ns)
|
||||
// String returns the TOML-canonical rendering of the local time, e.g. "07:32"
|
||||
// or "07:32:00.5" when the time carries a fractional second.
|
||||
func (lt LocalTime) String() string { return clockString(lt.Time) }
|
||||
|
||||
// appendClock appends the clock part of a TOML time to buf: HH:MM, seconds
|
||||
// only when the value carries them, and a fraction with its trailing zeros
|
||||
// dropped, so half a second is ".5" and not ".500000000". Both are the same
|
||||
// value either way; the shorter form is the one TOML 1.1 allows. The whole
|
||||
// rendering is built in one buffer, because the encoder writes a date-time
|
||||
// per entry of a large document.
|
||||
func appendClock(buf []byte, t time.Time) []byte {
|
||||
buf = t.AppendFormat(buf, "15:04")
|
||||
if t.Second() != 0 || t.Nanosecond() != 0 {
|
||||
buf = t.AppendFormat(buf, ":05")
|
||||
}
|
||||
return base
|
||||
if ns := t.Nanosecond(); ns > 0 {
|
||||
buf = append(buf, '.')
|
||||
buf = append(buf, strings.TrimRight(fmt.Sprintf("%09d", ns), "0")...)
|
||||
}
|
||||
return buf
|
||||
}
|
||||
|
||||
var (
|
||||
offsetDateTimeLayouts = []string{
|
||||
"2006-01-02T15:04:05.999999999Z07:00",
|
||||
"2006-01-02T15:04:05Z07:00",
|
||||
"2006-01-02 15:04:05.999999999Z07:00",
|
||||
"2006-01-02 15:04:05Z07:00",
|
||||
}
|
||||
localDateTimeLayouts = []string{
|
||||
"2006-01-02T15:04:05.999999999",
|
||||
"2006-01-02T15:04:05",
|
||||
"2006-01-02 15:04:05.999999999",
|
||||
"2006-01-02 15:04:05",
|
||||
}
|
||||
localTimeLayouts = []string{
|
||||
"15:04:05.999999999",
|
||||
"15:04:05",
|
||||
}
|
||||
// clockString renders a time of day the way TOML writes it.
|
||||
func clockString(t time.Time) string {
|
||||
return string(appendClock(make([]byte, 0, 16), t))
|
||||
}
|
||||
|
||||
// offsetString renders an offset date-time, the fourth TOML kind, in the same
|
||||
// shape: no zero seconds, no trailing zeros in the fraction, and the offset
|
||||
// written as "Z" when it is zero.
|
||||
func offsetString(t time.Time) string {
|
||||
buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
|
||||
buf = appendClock(buf, t)
|
||||
buf = t.AppendFormat(buf, "Z07:00")
|
||||
return string(buf)
|
||||
}
|
||||
|
||||
// dateTimeKind names the date-time shape a bare token has, as the scanner
|
||||
// below classifies it.
|
||||
type dateTimeKind int
|
||||
|
||||
const (
|
||||
dateTimeNone dateTimeKind = iota
|
||||
dateTimeOffset
|
||||
dateTimeLocal
|
||||
dateTimeDate
|
||||
dateTimeClock
|
||||
)
|
||||
|
||||
// dateTimeShape enforces the strict TOML grammar (two-digit components) that
|
||||
// time.Parse would otherwise accept loosely (e.g. a single-digit hour).
|
||||
var dateTimeShape = regexp.MustCompile(
|
||||
`^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}:\d{2}(\.\d+)?([Zz]|[+-]\d{2}:\d{2})?)?$` +
|
||||
`|^\d{2}:\d{2}:\d{2}(\.\d+)?$`,
|
||||
// The layouts the time package parses each shape with. Parsing accepts a
|
||||
// fractional second even when the layout does not carry one, so each shape
|
||||
// needs a single layout, chosen by whether the token has seconds.
|
||||
const (
|
||||
offsetDateTimeLayout = "2006-01-02T15:04:05Z07:00"
|
||||
offsetClockLayout = "2006-01-02T15:04Z07:00"
|
||||
localDateTimeLayout = "2006-01-02T15:04:05"
|
||||
localClockLayout = "2006-01-02T15:04"
|
||||
localTimeLayout = "15:04:05"
|
||||
localTimeClockLayout = "15:04"
|
||||
localDateOnlyLayout = "2006-01-02"
|
||||
)
|
||||
|
||||
// scanDateTimeShape validates a bare token against the strict TOML date-time
|
||||
// grammar and reports which kind it is: two-digit components, seconds
|
||||
// optional since TOML 1.1, a fraction only after seconds, an offset only
|
||||
// after a time, and an offset bounded to 00:00 through 23:59. The grammar is
|
||||
// a fixed byte shape, so the scan is a byte walk; the regular expressions
|
||||
// this replaced cost the parser measurably per token, and a shape that fails
|
||||
// the scan is simply not a date-time.
|
||||
func scanDateTimeShape(tok string) (kind dateTimeKind, seconds bool) {
|
||||
// A local clock on its own: HH:MM[:SS[.fraction]].
|
||||
if len(tok) >= 5 && tok[2] == ':' {
|
||||
n, secs, ok := scanClock(tok, 0)
|
||||
if !ok || n != len(tok) {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
return dateTimeClock, secs
|
||||
}
|
||||
// A date, optionally followed by a time and an offset.
|
||||
if len(tok) < 10 || tok[4] != '-' || tok[7] != '-' {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
for _, i := range [8]int{0, 1, 2, 3, 5, 6, 8, 9} {
|
||||
if !isDecDigit(tok[i]) {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
}
|
||||
if len(tok) == 10 {
|
||||
return dateTimeDate, false
|
||||
}
|
||||
if sep := tok[10]; sep != 'T' && sep != 't' && sep != ' ' {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
n, secs, ok := scanClock(tok, 11)
|
||||
if !ok {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
if n == len(tok) {
|
||||
return dateTimeLocal, secs
|
||||
}
|
||||
// The offset: Z/z, or a signed HH:MM bounded as the ABNF requires.
|
||||
switch c := tok[n]; {
|
||||
case c == 'Z' || c == 'z':
|
||||
if n+1 != len(tok) {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
case c == '+' || c == '-':
|
||||
if n+6 != len(tok) || tok[n+3] != ':' ||
|
||||
!isDecDigit(tok[n+1]) || !isDecDigit(tok[n+2]) ||
|
||||
!isDecDigit(tok[n+4]) || !isDecDigit(tok[n+5]) ||
|
||||
tok[n+1] > '2' || (tok[n+1] == '2' && tok[n+2] > '3') ||
|
||||
tok[n+4] > '5' {
|
||||
return dateTimeNone, false
|
||||
}
|
||||
default:
|
||||
return dateTimeNone, false
|
||||
}
|
||||
return dateTimeOffset, secs
|
||||
}
|
||||
|
||||
// scanClock validates HH:MM[:SS[.fraction]] starting at i and returns the
|
||||
// position after the clock, whether seconds were present, and whether the
|
||||
// shape is valid.
|
||||
func scanClock(tok string, i int) (pos int, seconds bool, ok bool) {
|
||||
if i+5 > len(tok) || tok[i+2] != ':' ||
|
||||
!isDecDigit(tok[i]) || !isDecDigit(tok[i+1]) ||
|
||||
!isDecDigit(tok[i+3]) || !isDecDigit(tok[i+4]) {
|
||||
return 0, false, false
|
||||
}
|
||||
i += 5
|
||||
if i == len(tok) || tok[i] != ':' {
|
||||
return i, false, true
|
||||
}
|
||||
if i+3 > len(tok) || !isDecDigit(tok[i+1]) || !isDecDigit(tok[i+2]) {
|
||||
return 0, false, false
|
||||
}
|
||||
i += 3
|
||||
if i == len(tok) || tok[i] != '.' {
|
||||
return i, true, true
|
||||
}
|
||||
i++
|
||||
digits := i
|
||||
for i < len(tok) && isDecDigit(tok[i]) {
|
||||
i++
|
||||
}
|
||||
if i == digits {
|
||||
return 0, false, false
|
||||
}
|
||||
return i, true, true
|
||||
}
|
||||
|
||||
// normaliseDateTimeToken rewrites the date/time separator to 'T' and the
|
||||
// offset marker to 'Z', the characters the layouts above carry. A token that
|
||||
// already has them is returned as it is, without a copy.
|
||||
func normaliseDateTimeToken(tok string, kind dateTimeKind) string {
|
||||
if kind == dateTimeDate || kind == dateTimeClock {
|
||||
return tok
|
||||
}
|
||||
needs := false
|
||||
for i := range len(tok) {
|
||||
c := tok[i]
|
||||
if c == 't' || c == 'z' || (c == ' ' && i == 10) {
|
||||
needs = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !needs {
|
||||
return tok
|
||||
}
|
||||
b := []byte(tok)
|
||||
for i, c := range b {
|
||||
switch {
|
||||
case c == 't':
|
||||
b[i] = 'T'
|
||||
case c == 'z':
|
||||
b[i] = 'Z'
|
||||
case c == ' ' && i == 10:
|
||||
b[i] = 'T'
|
||||
}
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// parseDateTime classifies and parses a bare token as a TOML date-time value.
|
||||
// It returns the decoded value (time.Time, LocalDateTime, LocalDate, or
|
||||
// It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or
|
||||
// LocalTime) and whether the token was a date-time at all.
|
||||
func parseDateTime(tok string) (any, bool) {
|
||||
if tok == "" || tok[0] < '0' || tok[0] > '9' {
|
||||
@@ -88,28 +243,48 @@ func parseDateTime(tok string) (any, bool) {
|
||||
if !strings.ContainsAny(tok, "-:") {
|
||||
return nil, false
|
||||
}
|
||||
if !dateTimeShape.MatchString(tok) {
|
||||
kind, seconds := scanDateTimeShape(tok)
|
||||
if kind == dateTimeNone {
|
||||
return nil, false
|
||||
}
|
||||
// The ABNF accepts lowercase "t"/"z"; time.Parse only matches uppercase.
|
||||
norm := strings.ToUpper(tok)
|
||||
for _, layout := range offsetDateTimeLayouts {
|
||||
if t, err := time.Parse(layout, norm); err == nil {
|
||||
return t, true
|
||||
norm := normaliseDateTimeToken(tok, kind)
|
||||
switch kind {
|
||||
case dateTimeOffset:
|
||||
layout := offsetClockLayout
|
||||
if seconds {
|
||||
layout = offsetDateTimeLayout
|
||||
}
|
||||
}
|
||||
for _, layout := range localDateTimeLayouts {
|
||||
if t, err := time.Parse(layout, norm); err == nil {
|
||||
return LocalDateTime{t}, true
|
||||
t, err := time.Parse(layout, norm)
|
||||
if err != nil {
|
||||
return nil, false
|
||||
}
|
||||
return OffsetDateTime{t}, true
|
||||
case dateTimeLocal:
|
||||
layout := localClockLayout
|
||||
if seconds {
|
||||
layout = localDateTimeLayout
|
||||
}
|
||||
t, err := time.Parse(layout, norm)
|
||||
if err != nil {
|
||||
return nil, false
|
||||
}
|
||||
return LocalDateTime{t}, true
|
||||
case dateTimeDate:
|
||||
t, err := time.Parse(localDateOnlyLayout, norm)
|
||||
if err != nil {
|
||||
return nil, false
|
||||
}
|
||||
}
|
||||
if t, err := time.Parse("2006-01-02", norm); err == nil {
|
||||
return LocalDate{t}, true
|
||||
}
|
||||
for _, layout := range localTimeLayouts {
|
||||
if t, err := time.Parse(layout, norm); err == nil {
|
||||
return LocalTime{t}, true
|
||||
case dateTimeClock:
|
||||
layout := localTimeClockLayout
|
||||
if seconds {
|
||||
layout = localTimeLayout
|
||||
}
|
||||
t, err := time.Parse(layout, norm)
|
||||
if err != nil {
|
||||
return nil, false
|
||||
}
|
||||
return LocalTime{t}, true
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
@@ -4,10 +4,13 @@
|
||||
package interpres
|
||||
|
||||
import (
|
||||
"encoding"
|
||||
"fmt"
|
||||
"math"
|
||||
"reflect"
|
||||
"slices"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
|
||||
@@ -20,6 +23,97 @@ func newDecoder() *decoder { return &decoder{} }
|
||||
|
||||
var timeType = reflect.TypeFor[time.Time]()
|
||||
|
||||
var (
|
||||
unmarshalerType = reflect.TypeFor[Unmarshaler]()
|
||||
textUnmarshalerType = reflect.TypeFor[encoding.TextUnmarshaler]()
|
||||
)
|
||||
|
||||
// The per-type flags record which interface lookups a decode into that type
|
||||
// can succeed at, so the hot path consults the cache instead of boxing every
|
||||
// value into an interface to ask. The bits name the receiver the method is
|
||||
// found on: the value itself, or its address.
|
||||
const (
|
||||
flagUnmarshaler uint8 = 1 << iota
|
||||
flagAddrUnmarshaler
|
||||
flagTextUnmarshaler
|
||||
flagAddrTextUnmarshaler
|
||||
)
|
||||
|
||||
// typeFlagCache holds one flag entry per destination type. A set is immutable
|
||||
// once published, the same trade-off structSchemaCache makes; the cache grows
|
||||
// with the number of distinct types decoded, never per document. The hint
|
||||
// below re-points at these published entries, so a hot lookup allocates
|
||||
// nothing.
|
||||
var typeFlagCache sync.Map // reflect.Type -> *flagHintEntry
|
||||
|
||||
// flagHintEntry pairs a type with its cached flags for the monomorphic hint
|
||||
// below. Both caches share the entry shape.
|
||||
type flagHintEntry struct {
|
||||
typ reflect.Type
|
||||
flags uint8
|
||||
}
|
||||
|
||||
// typeFlagHint remembers the entry resolved last, because a decode walks one
|
||||
// type across consecutive fields and elements. A lost race loses only the
|
||||
// hint: every value it can hold came from the cache.
|
||||
var typeFlagHint atomic.Pointer[flagHintEntry]
|
||||
|
||||
func typeFlags(t reflect.Type) uint8 {
|
||||
if e := typeFlagHint.Load(); e != nil && e.typ == t {
|
||||
return e.flags
|
||||
}
|
||||
if v, ok := typeFlagCache.Load(t); ok {
|
||||
entry := v.(*flagHintEntry)
|
||||
typeFlagHint.Store(entry)
|
||||
return entry.flags
|
||||
}
|
||||
var f uint8
|
||||
if t.Implements(unmarshalerType) {
|
||||
f |= flagUnmarshaler
|
||||
}
|
||||
pt := reflect.PointerTo(t)
|
||||
if pt.Implements(unmarshalerType) {
|
||||
f |= flagAddrUnmarshaler
|
||||
}
|
||||
// The date-time types are excluded from the text path: they carry
|
||||
// time.Time's UnmarshalText through an embedded field while their only
|
||||
// accepted form is a bare timestamp.
|
||||
if !isDateTimeType(t) {
|
||||
if t.Implements(textUnmarshalerType) {
|
||||
f |= flagTextUnmarshaler
|
||||
}
|
||||
if pt.Implements(textUnmarshalerType) {
|
||||
f |= flagAddrTextUnmarshaler
|
||||
}
|
||||
}
|
||||
actual, _ := typeFlagCache.LoadOrStore(t, &flagHintEntry{t, f})
|
||||
published := actual.(*flagHintEntry)
|
||||
typeFlagHint.Store(published)
|
||||
return published.flags
|
||||
}
|
||||
|
||||
// unmarshalerOf resolves the Unmarshaler for dst through the flag cache, so
|
||||
// an interface value is built only where the cache says the assertion can
|
||||
// succeed. An interface destination is asked dynamically, because the value
|
||||
// it will hold may implement the interface even when the interface type
|
||||
// itself does not.
|
||||
func unmarshalerOf(dst reflect.Value) (Unmarshaler, bool) {
|
||||
if dst.Kind() == reflect.Interface {
|
||||
u, ok := dst.Interface().(Unmarshaler)
|
||||
return u, ok
|
||||
}
|
||||
f := typeFlags(dst.Type())
|
||||
if f&flagUnmarshaler != 0 {
|
||||
u, ok := dst.Interface().(Unmarshaler)
|
||||
return u, ok
|
||||
}
|
||||
if f&flagAddrUnmarshaler != 0 && dst.CanAddr() {
|
||||
u, ok := dst.Addr().Interface().(Unmarshaler)
|
||||
return u, ok
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
func (d *decoder) decode(tree map[string]any, v any) error {
|
||||
rv := reflect.ValueOf(v)
|
||||
if rv.Kind() != reflect.Pointer || rv.IsNil() {
|
||||
@@ -62,6 +156,19 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
|
||||
}
|
||||
}
|
||||
|
||||
// A TOML string fills a destination that implements
|
||||
// encoding.TextUnmarshaler, the rule encoding/json follows. Every other
|
||||
// value kind keeps its own rule, so an integer still reaches a numeric
|
||||
// destination.
|
||||
if s, isString := data.(string); isString {
|
||||
if tu, ok := textUnmarshalerOf(dst); ok {
|
||||
if err := tu.UnmarshalText([]byte(s)); err != nil {
|
||||
return fmt.Errorf("unmarshal text: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
switch v := data.(type) {
|
||||
case map[string]any:
|
||||
return d.assignTable(v, dst)
|
||||
@@ -70,6 +177,9 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
|
||||
case []any:
|
||||
return d.assignSlice(v, dst)
|
||||
case string:
|
||||
if dst.Type() == durationType {
|
||||
return setDuration(dst, v)
|
||||
}
|
||||
return setBasic(dst, reflect.ValueOf(v), "string")
|
||||
case bool:
|
||||
return setBasic(dst, reflect.ValueOf(v), "bool")
|
||||
@@ -77,12 +187,10 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
|
||||
return setInt(dst, v)
|
||||
case float64:
|
||||
return setFloat(dst, v)
|
||||
case OffsetDateTime:
|
||||
return setOffsetDateTime(v, dst)
|
||||
case time.Time:
|
||||
if dst.Type() != timeType {
|
||||
return fmt.Errorf("interpres: cannot assign datetime to %s", dst.Type())
|
||||
}
|
||||
dst.Set(reflect.ValueOf(v))
|
||||
return nil
|
||||
return setDateTime(v, dst)
|
||||
default:
|
||||
rv := reflect.ValueOf(data)
|
||||
if rv.IsValid() && dst.Type() == rv.Type() {
|
||||
@@ -93,6 +201,28 @@ func (d *decoder) assign(data any, dst reflect.Value) error {
|
||||
}
|
||||
}
|
||||
|
||||
// textUnmarshalerOf is the same resolution for encoding.TextUnmarshaler,
|
||||
// with the date-time types excluded for the reason typeFlags records.
|
||||
func textUnmarshalerOf(dst reflect.Value) (encoding.TextUnmarshaler, bool) {
|
||||
if !dst.CanInterface() || isDateTimeType(dst.Type()) {
|
||||
return nil, false
|
||||
}
|
||||
if dst.Kind() == reflect.Interface {
|
||||
tu, ok := dst.Interface().(encoding.TextUnmarshaler)
|
||||
return tu, ok
|
||||
}
|
||||
f := typeFlags(dst.Type())
|
||||
if f&flagTextUnmarshaler != 0 {
|
||||
tu, ok := dst.Interface().(encoding.TextUnmarshaler)
|
||||
return tu, ok
|
||||
}
|
||||
if f&flagAddrTextUnmarshaler != 0 && dst.CanAddr() {
|
||||
tu, ok := dst.Addr().Interface().(encoding.TextUnmarshaler)
|
||||
return tu, ok
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
func (d *decoder) assignTable(tbl map[string]any, dst reflect.Value) error {
|
||||
switch dst.Kind() {
|
||||
case reflect.Struct:
|
||||
@@ -105,17 +235,53 @@ func (d *decoder) assignTable(tbl map[string]any, dst reflect.Value) error {
|
||||
}
|
||||
|
||||
func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
|
||||
fields := structFields(dst.Type())
|
||||
schema := cachedStructSchema(dst.Type())
|
||||
if d.disallowUnknown {
|
||||
// Map iteration order is random, so pick the unknown key to report
|
||||
// deterministically: the smallest one.
|
||||
unknown := ""
|
||||
for key := range tbl {
|
||||
if _, ok := schema.byName[key]; ok {
|
||||
continue
|
||||
}
|
||||
if _, ok := schema.byName[strings.ToLower(key)]; ok {
|
||||
continue
|
||||
}
|
||||
if unknown == "" || key < unknown {
|
||||
unknown = key
|
||||
}
|
||||
}
|
||||
if unknown != "" {
|
||||
return fmt.Errorf("interpres: unknown field %q for %s", unknown, dst.Type())
|
||||
}
|
||||
}
|
||||
for key, val := range tbl {
|
||||
field, ok := fields[strings.ToLower(key)]
|
||||
// A key that is already lowercase, which document keys usually are,
|
||||
// hits the map directly; only a miss pays for the case fold.
|
||||
field, ok := schema.byName[key]
|
||||
if !ok {
|
||||
if d.disallowUnknown {
|
||||
return fmt.Errorf("interpres: unknown field %q for %s", key, dst.Type())
|
||||
field, ok = schema.byName[strings.ToLower(key)]
|
||||
}
|
||||
if !ok {
|
||||
if schema.embedMaps != nil {
|
||||
// Leftover keys land in an untagged embedded map, the inverse
|
||||
// of the encoder inlining that map's entries.
|
||||
mv, err := fieldByIndex(dst, schema.embedMaps[0])
|
||||
if err != nil {
|
||||
return newDecodeError(key, err)
|
||||
}
|
||||
if err := d.assignMap(map[string]any{key: val}, mv); err != nil {
|
||||
return newDecodeError(key, err)
|
||||
}
|
||||
}
|
||||
continue
|
||||
}
|
||||
if err := d.assign(val, dst.Field(field)); err != nil {
|
||||
return fmt.Errorf("%s: %w", key, err)
|
||||
fv, err := fieldByIndex(dst, field.index)
|
||||
if err != nil {
|
||||
return newDecodeError(key, err)
|
||||
}
|
||||
if err := d.assign(val, fv); err != nil {
|
||||
return newDecodeError(key, err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
@@ -132,7 +298,7 @@ func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
|
||||
for key, val := range tbl {
|
||||
elem := reflect.New(elemType).Elem()
|
||||
if err := d.assign(val, elem); err != nil {
|
||||
return fmt.Errorf("%s: %w", key, err)
|
||||
return newDecodeError(key, err)
|
||||
}
|
||||
dst.SetMapIndex(reflect.ValueOf(key), elem)
|
||||
}
|
||||
@@ -146,7 +312,7 @@ func (d *decoder) assignSlice(items []any, dst reflect.Value) error {
|
||||
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
|
||||
for i, item := range items {
|
||||
if err := d.assign(item, out.Index(i)); err != nil {
|
||||
return fmt.Errorf("[%d]: %w", i, err)
|
||||
return newDecodeError(fmt.Sprintf("[%d]", i), err)
|
||||
}
|
||||
}
|
||||
dst.Set(out)
|
||||
@@ -160,7 +326,7 @@ func (d *decoder) assignTableSlice(items []map[string]any, dst reflect.Value) er
|
||||
out := reflect.MakeSlice(dst.Type(), len(items), len(items))
|
||||
for i, item := range items {
|
||||
if err := d.assign(item, out.Index(i)); err != nil {
|
||||
return fmt.Errorf("[%d]: %w", i, err)
|
||||
return newDecodeError(fmt.Sprintf("[%d]", i), err)
|
||||
}
|
||||
}
|
||||
dst.Set(out)
|
||||
@@ -169,11 +335,57 @@ func (d *decoder) assignTableSlice(items []map[string]any, dst reflect.Value) er
|
||||
|
||||
// --- low-level setters -----------------------------------------------------
|
||||
|
||||
// setOffsetDateTime stores an offset date-time: in a wrapper destination as it
|
||||
// is, and in a plain time.Time, which takes the instant with the offset the
|
||||
// document wrote, so a timestamp field does not have to name the wrapper.
|
||||
func setOffsetDateTime(v OffsetDateTime, dst reflect.Value) error {
|
||||
switch dst.Type() {
|
||||
case offsetDateTimeType:
|
||||
dst.Set(reflect.ValueOf(v))
|
||||
case timeType:
|
||||
dst.Set(reflect.ValueOf(v.Time))
|
||||
default:
|
||||
return fmt.Errorf("interpres: cannot assign datetime to %s", dst.Type())
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// setDateTime stores a time.Time that reached the tree directly, which is the
|
||||
// shape a tree built by hand carries. Dates the parser produced arrive as
|
||||
// OffsetDateTime instead.
|
||||
func setDateTime(v time.Time, dst reflect.Value) error {
|
||||
switch dst.Type() {
|
||||
case timeType:
|
||||
dst.Set(reflect.ValueOf(v))
|
||||
case offsetDateTimeType:
|
||||
dst.Set(reflect.ValueOf(OffsetDateTime{v}))
|
||||
default:
|
||||
return fmt.Errorf("interpres: cannot assign datetime to %s", dst.Type())
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func setBasic(dst, val reflect.Value, kind string) error {
|
||||
if dst.Kind() != val.Kind() {
|
||||
return fmt.Errorf("interpres: cannot assign %s to %s", kind, dst.Type())
|
||||
}
|
||||
dst.Set(val)
|
||||
// Convert rather than assign: a value of the predeclared type is not
|
||||
// assignable to a defined type of the same kind, so a plain Set panics on
|
||||
// a destination such as `type Name string`.
|
||||
dst.Set(val.Convert(dst.Type()))
|
||||
return nil
|
||||
}
|
||||
|
||||
// setDuration reads a duration literal into a time.Duration destination. TOML
|
||||
// has no duration type, so the encoder writes the canonical Go form and the
|
||||
// decoder reads that back; a bare integer stays the nanosecond count it has
|
||||
// always been, and reaches the destination through setInt.
|
||||
func setDuration(dst reflect.Value, s string) error {
|
||||
d, err := time.ParseDuration(s)
|
||||
if err != nil {
|
||||
return fmt.Errorf("interpres: invalid duration %q", s)
|
||||
}
|
||||
dst.SetInt(int64(d))
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -188,21 +400,21 @@ func setInt(dst reflect.Value, v int64) error {
|
||||
if v < 0 {
|
||||
return fmt.Errorf("interpres: cannot assign negative %d to %s", v, dst.Type())
|
||||
}
|
||||
var max uint64
|
||||
switch dst.Kind() {
|
||||
case reflect.Uint8:
|
||||
max = math.MaxUint8
|
||||
case reflect.Uint16:
|
||||
max = math.MaxUint16
|
||||
case reflect.Uint32:
|
||||
max = math.MaxUint32
|
||||
}
|
||||
if max != 0 && uint64(v) > max {
|
||||
// OverflowUint knows every width, uint included on platforms where it
|
||||
// is narrower than uint64; SetUint would silently truncate instead.
|
||||
if dst.OverflowUint(uint64(v)) {
|
||||
return fmt.Errorf("interpres: integer %d overflows %s", v, dst.Type())
|
||||
}
|
||||
dst.SetUint(uint64(v))
|
||||
case reflect.Float32, reflect.Float64:
|
||||
dst.SetFloat(float64(v))
|
||||
// A finite value beyond the float32 range would silently become ±Inf;
|
||||
// infinities and NaN themselves pass through. An int64 never
|
||||
// overflows either float width.
|
||||
f := float64(v)
|
||||
if dst.OverflowFloat(f) {
|
||||
return fmt.Errorf("interpres: integer %d overflows %s", v, dst.Type())
|
||||
}
|
||||
dst.SetFloat(f)
|
||||
default:
|
||||
return fmt.Errorf("interpres: cannot assign integer to %s", dst.Type())
|
||||
}
|
||||
@@ -212,6 +424,9 @@ func setInt(dst reflect.Value, v int64) error {
|
||||
func setFloat(dst reflect.Value, v float64) error {
|
||||
switch dst.Kind() {
|
||||
case reflect.Float32, reflect.Float64:
|
||||
if dst.OverflowFloat(v) {
|
||||
return fmt.Errorf("interpres: float %g overflows %s", v, dst.Type())
|
||||
}
|
||||
dst.SetFloat(v)
|
||||
return nil
|
||||
default:
|
||||
@@ -219,26 +434,118 @@ func setFloat(dst reflect.Value, v float64) error {
|
||||
}
|
||||
}
|
||||
|
||||
// structFields builds a lower-cased lookup of field name → field index for the
|
||||
// exported fields of t, honouring `toml:"name"` tags.
|
||||
func structFields(t reflect.Type) map[string]int {
|
||||
fields := make(map[string]int, t.NumField())
|
||||
for i := range t.NumField() {
|
||||
f := t.Field(i)
|
||||
if f.PkgPath != "" { // unexported
|
||||
continue
|
||||
}
|
||||
name := f.Name
|
||||
if tag, ok := f.Tag.Lookup("toml"); ok {
|
||||
tag = strings.Split(tag, ",")[0]
|
||||
if tag == "-" {
|
||||
// structFieldLoc locates one destination field by its index path from the
|
||||
// struct root and by the depth the field sits at, which breaks name clashes
|
||||
// in favour of the shallower field.
|
||||
type structFieldLoc struct {
|
||||
index []int
|
||||
depth int
|
||||
}
|
||||
|
||||
// structSchema flattens the exported fields of t for decode, mirroring the
|
||||
// encoder: an untagged embedded struct is inlined, so its own fields match
|
||||
// keys of the same table, and an untagged embedded map is recorded in
|
||||
// embedMaps (first declaration first) as the destination for leftover keys.
|
||||
// When two fields resolve to one name, the shallower wins, then the later
|
||||
// declaration.
|
||||
type structSchema struct {
|
||||
byName map[string]structFieldLoc
|
||||
embedMaps [][]int
|
||||
}
|
||||
|
||||
// structSchemaCache holds one schema per struct type. A schema is immutable
|
||||
// once published, so concurrent callers only race to build an identical value,
|
||||
// the same trade-off encoding/json's field cache makes. The cache grows with
|
||||
// the number of distinct types decoded or encoded, never per document.
|
||||
var structSchemaCache sync.Map // reflect.Type -> structSchema
|
||||
|
||||
func cachedStructSchema(t reflect.Type) structSchema {
|
||||
if s, ok := structSchemaCache.Load(t); ok {
|
||||
return s.(structSchema)
|
||||
}
|
||||
s := newStructSchema(t)
|
||||
actual, _ := structSchemaCache.LoadOrStore(t, s)
|
||||
return actual.(structSchema)
|
||||
}
|
||||
|
||||
func newStructSchema(t reflect.Type) structSchema {
|
||||
s := structSchema{byName: make(map[string]structFieldLoc, t.NumField())}
|
||||
// A struct may embed a pointer to itself, which is legal Go, so the walk
|
||||
// tracks the struct types on the current path and stops when one repeats;
|
||||
// without the guard the recursion never terminates. A self-promoted key
|
||||
// always loses to the shallower original, so skipping it changes nothing.
|
||||
visiting := map[reflect.Type]bool{}
|
||||
var walk func(t reflect.Type, prefix []int, depth int)
|
||||
walk = func(t reflect.Type, prefix []int, depth int) {
|
||||
visiting[t] = true
|
||||
defer delete(visiting, t)
|
||||
for i := range t.NumField() {
|
||||
f := t.Field(i)
|
||||
if f.PkgPath != "" { // unexported
|
||||
continue
|
||||
}
|
||||
if tag != "" {
|
||||
name = tag
|
||||
path := append(append([]int{}, prefix...), i)
|
||||
name := ""
|
||||
if tag, ok := f.Tag.Lookup("toml"); ok {
|
||||
name, _, _ = strings.Cut(tag, ",")
|
||||
if name == "-" {
|
||||
continue
|
||||
}
|
||||
}
|
||||
if f.Anonymous && name == "" {
|
||||
ft := f.Type
|
||||
for ft.Kind() == reflect.Pointer {
|
||||
ft = ft.Elem()
|
||||
}
|
||||
switch {
|
||||
case ft.Kind() == reflect.Struct && !isScalarStruct(ft):
|
||||
if !visiting[ft] {
|
||||
walk(ft, path, depth+1)
|
||||
}
|
||||
continue
|
||||
case ft.Kind() == reflect.Map && ft.Key().Kind() == reflect.String:
|
||||
s.embedMaps = append(s.embedMaps, path)
|
||||
continue
|
||||
}
|
||||
name = f.Name
|
||||
}
|
||||
if name == "" {
|
||||
name = f.Name
|
||||
}
|
||||
key := strings.ToLower(name)
|
||||
if existing, ok := s.byName[key]; !ok || depth <= existing.depth {
|
||||
s.byName[key] = structFieldLoc{index: path, depth: depth}
|
||||
}
|
||||
}
|
||||
fields[strings.ToLower(name)] = i
|
||||
}
|
||||
return fields
|
||||
walk(t, nil, 0)
|
||||
return s
|
||||
}
|
||||
|
||||
// ownsKey reports whether the field at path is the one that resolves key.
|
||||
// The encoder consults it to emit exactly the field the decoder would fill,
|
||||
// so a struct with two fields mapping to one key does not marshal into a
|
||||
// duplicate TOML key.
|
||||
func (s structSchema) ownsKey(key string, path []int) bool {
|
||||
loc, ok := s.byName[key]
|
||||
return ok && slices.Equal(loc.index, path)
|
||||
}
|
||||
|
||||
// fieldByIndex walks an index path from a struct value, allocating nil
|
||||
// pointers along the way so a key can reach through an embedded pointer
|
||||
// struct. Every field on the path is exported, so each step is settable.
|
||||
func fieldByIndex(v reflect.Value, path []int) (reflect.Value, error) {
|
||||
for i, x := range path {
|
||||
v = v.Field(x)
|
||||
if i < len(path)-1 && v.Kind() == reflect.Pointer {
|
||||
if v.IsNil() {
|
||||
if !v.CanSet() {
|
||||
return reflect.Value{}, fmt.Errorf("cannot allocate nil embedded pointer")
|
||||
}
|
||||
v.Set(reflect.New(v.Type().Elem()))
|
||||
}
|
||||
v = v.Elem()
|
||||
}
|
||||
}
|
||||
return v, nil
|
||||
}
|
||||
|
||||
+617
-14
@@ -8,8 +8,11 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"net"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestSyntaxErrorMessage(t *testing.T) {
|
||||
@@ -21,19 +24,38 @@ func TestSyntaxErrorMessage(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestParseRejectsInvalidUTF8(t *testing.T) {
|
||||
_, err := Parse([]byte("v = \"\xff\"\n"))
|
||||
if err == nil {
|
||||
t.Fatal("expected a UTF-8 validation error")
|
||||
// The scan validates UTF-8 where it meets the byte, so the reported line
|
||||
// is the invalid byte's own, wherever in the document it sits.
|
||||
cases := []struct {
|
||||
name string
|
||||
doc string
|
||||
line int
|
||||
}{
|
||||
{"in a basic string", "v = \"\xff\"\n", 1},
|
||||
{"in a literal string", "v = '\xff'\n", 1},
|
||||
{"in a multiline string", "v = \"\"\"\n\xff\"\"\"\n", 2},
|
||||
{"in a comment", "v = 1\n# caf\xe9\xff\n", 2},
|
||||
{"in a bare key", "va\xfflue = 1\n", 1},
|
||||
{"as a statement", "\xff = 1\n", 1},
|
||||
{"in a bare value", "v = \xff1\n", 1},
|
||||
{"after a value", "v = 1 \xff\n", 1},
|
||||
{"after the first line", "a = 1\nb = \"\xff\"\n", 2},
|
||||
}
|
||||
se, ok := err.(*SyntaxError)
|
||||
if !ok {
|
||||
t.Fatalf("err is %T, want *SyntaxError", err)
|
||||
}
|
||||
if !strings.Contains(se.Msg, "UTF-8") {
|
||||
t.Errorf("Msg = %q, want it to mention UTF-8", se.Msg)
|
||||
}
|
||||
if se.Line != 1 {
|
||||
t.Errorf("Line = %d, want 1", se.Line)
|
||||
for _, c := range cases {
|
||||
_, err := ParseMap([]byte(c.doc))
|
||||
if err == nil {
|
||||
t.Fatalf("%s: expected a UTF-8 validation error", c.name)
|
||||
}
|
||||
se, ok := err.(*SyntaxError)
|
||||
if !ok {
|
||||
t.Fatalf("%s: err is %T, want *SyntaxError", c.name, err)
|
||||
}
|
||||
if !strings.Contains(se.Msg, "UTF-8") {
|
||||
t.Errorf("%s: Msg = %q, want it to mention UTF-8", c.name, se.Msg)
|
||||
}
|
||||
if se.Line != c.line {
|
||||
t.Errorf("%s: Line = %d, want %d", c.name, se.Line, c.line)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -127,7 +149,7 @@ func TestUnmarshalIntoNilAny(t *testing.T) {
|
||||
func TestParseContextHonoursCancellation(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
if _, err := ParseContext(ctx, []byte("a = 1\n")); !errors.Is(err, context.Canceled) {
|
||||
if _, err := ParseMapContext(ctx, []byte("a = 1\n")); !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("ParseContext returned %v, want context.Canceled", err)
|
||||
}
|
||||
}
|
||||
@@ -157,7 +179,7 @@ func TestContextRoundTrip(t *testing.T) {
|
||||
in := []byte(`title = "x"
|
||||
count = 3
|
||||
`)
|
||||
if _, err := ParseContext(context.Background(), in); err != nil {
|
||||
if _, err := ParseMapContext(context.Background(), in); err != nil {
|
||||
t.Fatalf("ParseContext: %v", err)
|
||||
}
|
||||
var out struct {
|
||||
@@ -222,6 +244,33 @@ func TestUnmarshalIntToUint64FitsMaxInt64(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalFloat32Overflow(t *testing.T) {
|
||||
// A finite float64 beyond the float32 range must not decode silently as
|
||||
// an infinity.
|
||||
type C struct {
|
||||
X float32 `toml:"x"`
|
||||
}
|
||||
var c C
|
||||
err := Unmarshal([]byte("x = 1e300\n"), &c)
|
||||
if err == nil {
|
||||
t.Fatal("expected overflow error for float32")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "overflow") {
|
||||
t.Errorf("err = %v, want substring 'overflow'", err.Error())
|
||||
}
|
||||
// Infinities themselves pass through, and in-range values are untouched.
|
||||
var ok C
|
||||
if err := Unmarshal([]byte("x = inf\n"), &ok); err != nil {
|
||||
t.Fatalf("inf should decode into float32, got %v", err)
|
||||
}
|
||||
if !math.IsInf(float64(ok.X), 1) {
|
||||
t.Errorf("X = %v, want +Inf", ok.X)
|
||||
}
|
||||
if err := Unmarshal([]byte("x = 1.5\n"), &ok); err != nil || ok.X != 1.5 {
|
||||
t.Fatalf("1.5 should decode into float32, got %v (X=%v)", err, ok.X)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalNegativeIntToUint(t *testing.T) {
|
||||
type C struct {
|
||||
X uint8 `toml:"x"`
|
||||
@@ -494,3 +543,557 @@ field = "y"
|
||||
t.Errorf("Field = %q, want \"y\"", cfg.R.Field)
|
||||
}
|
||||
}
|
||||
|
||||
// --- embedded field symmetry -----------------------------------------------
|
||||
|
||||
type RoundTripBase struct {
|
||||
ID int `toml:"id"`
|
||||
Name string `toml:"name"`
|
||||
}
|
||||
|
||||
type RoundTripDerived struct {
|
||||
RoundTripBase
|
||||
X string `toml:"x"`
|
||||
}
|
||||
|
||||
func TestUnmarshalEmbeddedStructRoundTrip(t *testing.T) {
|
||||
orig := RoundTripDerived{ID: 1, Name: "b", X: "x"}
|
||||
out, err := Marshal(orig)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
var back RoundTripDerived
|
||||
if err := Unmarshal(out, &back); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if back != orig {
|
||||
t.Fatalf("round-trip mismatch:\nwas: %+v\nnow: %+v", orig, back)
|
||||
}
|
||||
}
|
||||
|
||||
type RoundTripPtrCfg struct {
|
||||
*RoundTripBase
|
||||
X string `toml:"x"`
|
||||
}
|
||||
|
||||
func TestUnmarshalEmbeddedPointerStruct(t *testing.T) {
|
||||
var cfg RoundTripPtrCfg
|
||||
if err := Unmarshal([]byte("id = 7\nname = \"n\"\nx = \"x\"\n"), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.RoundTripBase == nil || cfg.ID != 7 || cfg.Name != "n" || cfg.X != "x" {
|
||||
t.Fatalf("decoded: %+v", cfg)
|
||||
}
|
||||
}
|
||||
|
||||
// A struct embedding a pointer to itself is legal Go; decoding into it must
|
||||
// terminate. The schema walk used to recurse through the embedded type
|
||||
// forever.
|
||||
func TestUnmarshalSelfEmbeddedPointerStructTerminates(t *testing.T) {
|
||||
type SelfLink struct {
|
||||
*SelfLink
|
||||
X int `toml:"x"`
|
||||
Y string `toml:"y"`
|
||||
}
|
||||
var n SelfLink
|
||||
if err := Unmarshal([]byte("x = 1\ny = \"s\"\n"), &n); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if n.X != 1 || n.Y != "s" {
|
||||
t.Fatalf("decoded: %+v", n)
|
||||
}
|
||||
|
||||
// A nil self pointer on the encode side stays skippable, as any nil
|
||||
// embedded pointer is.
|
||||
out, err := Marshal(SelfLink{X: 2})
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
if want := "x = 2\ny = \"\"\n"; string(out) != want {
|
||||
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
|
||||
}
|
||||
}
|
||||
|
||||
type RoundTripExtra map[string]int
|
||||
|
||||
type RoundTripMapCfg struct {
|
||||
RoundTripExtra
|
||||
X string `toml:"x"`
|
||||
}
|
||||
|
||||
func TestUnmarshalEmbeddedMap(t *testing.T) {
|
||||
var cfg RoundTripMapCfg
|
||||
if err := Unmarshal([]byte("alpha = 1\nx = \"x\"\n"), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.RoundTripExtra["alpha"] != 1 || cfg.X != "x" {
|
||||
t.Fatalf("decoded: %+v", cfg)
|
||||
}
|
||||
|
||||
orig := RoundTripMapCfg{RoundTripExtra: RoundTripExtra{"a": 1}, X: "x"}
|
||||
out, err := Marshal(orig)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal: %v", err)
|
||||
}
|
||||
var back RoundTripMapCfg
|
||||
if err := Unmarshal(out, &back); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if back.X != "x" || back.RoundTripExtra["a"] != 1 {
|
||||
t.Fatalf("round-trip mismatch: %+v", back)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalEmbeddedNameClashShallowerWins(t *testing.T) {
|
||||
type Inner struct {
|
||||
Name string `toml:"name"`
|
||||
Deep string `toml:"deep"`
|
||||
}
|
||||
type Outer struct {
|
||||
Inner
|
||||
Name string `toml:"name"`
|
||||
}
|
||||
var v Outer
|
||||
if err := Unmarshal([]byte("name = \"outer\"\ndeep = \"d\"\n"), &v); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if v.Name != "outer" || v.Deep != "d" {
|
||||
t.Fatalf("decoded: %+v", v)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalNameClashEqualDepthLaterWins(t *testing.T) {
|
||||
// At equal depth the field declared later resolves the name, matching the
|
||||
// documented rule.
|
||||
type C struct {
|
||||
First string `toml:"v"`
|
||||
Second int `toml:"v"`
|
||||
}
|
||||
var c C
|
||||
if err := Unmarshal([]byte("v = 1\n"), &c); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if c.Second != 1 {
|
||||
t.Fatalf("decoded: %+v, want the later field to take the value", c)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalUnknownKeyWithoutEmbeddedMap(t *testing.T) {
|
||||
var cfg RoundTripDerived
|
||||
if err := Unmarshal([]byte("rogue = 1\n"), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.ID != 0 || cfg.X != "" {
|
||||
t.Fatalf("decoded: %+v", cfg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalStrictEmbeddedMapStaysStrict(t *testing.T) {
|
||||
type Cfg struct {
|
||||
RoundTripExtra
|
||||
Name string `toml:"name"`
|
||||
}
|
||||
dec := NewDecoder().DisallowUnknownFields()
|
||||
err := dec.Decode([]byte("name = \"n\"\nrogue = 1\n"), &Cfg{})
|
||||
if err == nil || !strings.Contains(err.Error(), "unknown field") {
|
||||
t.Fatalf("expected unknown field error, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeErrorCarriesPath(t *testing.T) {
|
||||
type Item struct {
|
||||
Name string `toml:"name"`
|
||||
Weight uint8 `toml:"weight"`
|
||||
}
|
||||
type Cfg struct {
|
||||
Tags []string `toml:"tags"`
|
||||
Items []Item `toml:"items"`
|
||||
}
|
||||
var cfg Cfg
|
||||
err := Unmarshal([]byte("[[items]]\nname = \"a\"\nweight = 300\n"), &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected an overflow error")
|
||||
}
|
||||
de, ok := errors.AsType[*DecodeError](err)
|
||||
if !ok {
|
||||
t.Fatalf("expected a *DecodeError, got %T: %v", err, err)
|
||||
}
|
||||
want := []string{"items", "[0]", "weight"}
|
||||
if !slices.Equal(de.Path, want) {
|
||||
t.Fatalf("Path = %v, want %v", de.Path, want)
|
||||
}
|
||||
if de.Err == nil || !strings.Contains(de.Err.Error(), "overflows uint8") {
|
||||
t.Fatalf("Err = %v", de.Err)
|
||||
}
|
||||
// The rendered message keeps its shape: segments joined with ": ".
|
||||
wantMsg := "items: [0]: weight: interpres: integer 300 overflows uint8"
|
||||
if err.Error() != wantMsg {
|
||||
t.Fatalf("message = %q, want %q", err.Error(), wantMsg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeErrorOnMapDestination(t *testing.T) {
|
||||
var m map[string]uint8
|
||||
err := Unmarshal([]byte("count = -1\n"), &m)
|
||||
if err == nil {
|
||||
t.Fatal("expected an error")
|
||||
}
|
||||
de, ok := errors.AsType[*DecodeError](err)
|
||||
if !ok {
|
||||
t.Fatalf("expected a *DecodeError, got %T: %v", err, err)
|
||||
}
|
||||
if !slices.Equal(de.Path, []string{"count"}) {
|
||||
t.Fatalf("Path = %v", de.Path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalIntoDefinedScalarTypes(t *testing.T) {
|
||||
// A defined type whose underlying kind is string or bool takes the value.
|
||||
// A bare reflect Set panics on such a type, because a string is not
|
||||
// assignable to a defined string type without a conversion.
|
||||
type Name string
|
||||
type Flag bool
|
||||
type Cfg struct {
|
||||
N Name `toml:"n"`
|
||||
F Flag `toml:"f"`
|
||||
}
|
||||
var cfg Cfg
|
||||
if err := Unmarshal([]byte("n = \"x\"\nf = true\n"), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.N != "x" {
|
||||
t.Errorf("N = %q, want \"x\"", cfg.N)
|
||||
}
|
||||
if !cfg.F {
|
||||
t.Error("F = false, want true")
|
||||
}
|
||||
}
|
||||
|
||||
// --- encoding.TextUnmarshaler and time.Duration ----------------------------
|
||||
|
||||
// textReceiver implements encoding.TextUnmarshaler on the pointer receiver.
|
||||
type textReceiver struct{ Text string }
|
||||
|
||||
func (t *textReceiver) UnmarshalText(text []byte) error {
|
||||
t.Text = "got:" + string(text)
|
||||
return nil
|
||||
}
|
||||
|
||||
// upperText is a defined string type whose UnmarshalText transforms the
|
||||
// content, so a plain string assignment would leave the wrong value behind.
|
||||
type upperText string
|
||||
|
||||
func (u *upperText) UnmarshalText(text []byte) error {
|
||||
*u = upperText(strings.ToUpper(string(text)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// failingTextUnmarshaler fails the decode from UnmarshalText.
|
||||
type failingTextUnmarshaler struct{}
|
||||
|
||||
func (f *failingTextUnmarshaler) UnmarshalText(_ []byte) error { return errors.New("text boom") }
|
||||
|
||||
// textAndTOMLReceiver implements both decode interfaces; the TOML method wins.
|
||||
type textAndTOMLReceiver struct{ From string }
|
||||
|
||||
func (t *textAndTOMLReceiver) UnmarshalTOML(any) error { t.From = "toml"; return nil }
|
||||
|
||||
func (t *textAndTOMLReceiver) UnmarshalText([]byte) error { t.From = "text"; return nil }
|
||||
|
||||
func TestTextUnmarshalerByPointer(t *testing.T) {
|
||||
type Cfg struct {
|
||||
R textReceiver `toml:"r"`
|
||||
}
|
||||
var cfg Cfg
|
||||
if err := Unmarshal([]byte(`r = "hello"`), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.R.Text != "got:hello" {
|
||||
t.Errorf("Text = %q, want \"got:hello\"", cfg.R.Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextUnmarshalerWinsOverKindAssignment(t *testing.T) {
|
||||
type Cfg struct {
|
||||
U upperText `toml:"u"`
|
||||
}
|
||||
var cfg Cfg
|
||||
if err := Unmarshal([]byte(`u = "abc"`), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.U != "ABC" {
|
||||
t.Errorf("U = %q, want \"ABC\"", cfg.U)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextUnmarshalerForNetIP(t *testing.T) {
|
||||
type Cfg struct {
|
||||
V4 net.IP `toml:"v4"`
|
||||
V6 net.IP `toml:"v6"`
|
||||
IPs []net.IP `toml:"ips"`
|
||||
}
|
||||
in := "v4 = \"192.0.2.1\"\nv6 = \"2001:db8::68\"\nips = [\"198.51.100.7\", \"203.0.113.9\"]\n"
|
||||
var cfg Cfg
|
||||
if err := Unmarshal([]byte(in), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if got := cfg.V4.String(); got != "192.0.2.1" {
|
||||
t.Errorf("V4 = %q, want \"192.0.2.1\"", got)
|
||||
}
|
||||
if got := cfg.V6.String(); got != "2001:db8::68" {
|
||||
t.Errorf("V6 = %q, want \"2001:db8::68\"", got)
|
||||
}
|
||||
if len(cfg.IPs) != 2 || cfg.IPs[0].String() != "198.51.100.7" || cfg.IPs[1].String() != "203.0.113.9" {
|
||||
t.Errorf("IPs = %v, want two addresses", cfg.IPs)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextUnmarshalerSeesStringsOnly(t *testing.T) {
|
||||
// An integer keeps its own rule: the text method is not consulted, and the
|
||||
// value does not reach the receiver.
|
||||
type Cfg struct {
|
||||
R textReceiver `toml:"r"`
|
||||
}
|
||||
var cfg Cfg
|
||||
err := Unmarshal([]byte("r = 1\n"), &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected an integer to be rejected for a text receiver")
|
||||
}
|
||||
if cfg.R.Text != "" {
|
||||
t.Errorf("Text = %q, want it untouched", cfg.R.Text)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalTOMLWinsOverTextUnmarshaler(t *testing.T) {
|
||||
type Cfg struct {
|
||||
B textAndTOMLReceiver `toml:"b"`
|
||||
}
|
||||
var cfg Cfg
|
||||
if err := Unmarshal([]byte(`b = "x"`), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.B.From != "toml" {
|
||||
t.Errorf("From = %q, want \"toml\"", cfg.B.From)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextUnmarshalerErrorCarriesPath(t *testing.T) {
|
||||
type Inner struct {
|
||||
F failingTextUnmarshaler `toml:"f"`
|
||||
}
|
||||
type Cfg struct {
|
||||
Inner Inner `toml:"inner"`
|
||||
}
|
||||
var cfg Cfg
|
||||
err := Unmarshal([]byte("[inner]\nf = \"x\"\n"), &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected an error from UnmarshalText")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "unmarshal text: text boom") {
|
||||
t.Errorf("err = %v, want the text error wrapped", err)
|
||||
}
|
||||
de, ok := errors.AsType[*DecodeError](err)
|
||||
if !ok {
|
||||
t.Fatalf("expected a *DecodeError, got %T: %v", err, err)
|
||||
}
|
||||
if !slices.Equal(de.Path, []string{"inner", "f"}) {
|
||||
t.Fatalf("Path = %v, want [inner f]", de.Path)
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextUnmarshalerReportsBadText(t *testing.T) {
|
||||
var cfg struct {
|
||||
IP net.IP `toml:"ip"`
|
||||
}
|
||||
err := Unmarshal([]byte(`ip = "not-an-ip"`), &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected an error for a malformed address")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "unmarshal text:") {
|
||||
t.Errorf("err = %v, want it wrapped as a text error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalDurations(t *testing.T) {
|
||||
type Cfg struct {
|
||||
FromText time.Duration `toml:"from_text"`
|
||||
FromInt time.Duration `toml:"from_int"`
|
||||
Fraction time.Duration `toml:"fraction"`
|
||||
}
|
||||
in := "from_text = \"1h30m\"\nfrom_int = 5400000000000\nfraction = \"1.5s\"\n"
|
||||
var cfg Cfg
|
||||
if err := Unmarshal([]byte(in), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.FromText != 90*time.Minute {
|
||||
t.Errorf("FromText = %v, want %v", cfg.FromText, 90*time.Minute)
|
||||
}
|
||||
if cfg.FromInt != 90*time.Minute {
|
||||
t.Errorf("FromInt = %v, want %v", cfg.FromInt, 90*time.Minute)
|
||||
}
|
||||
if cfg.Fraction != 1500*time.Millisecond {
|
||||
t.Errorf("Fraction = %v, want %v", cfg.Fraction, 1500*time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
func TestUnmarshalDurationRejectsMalformedText(t *testing.T) {
|
||||
var cfg struct {
|
||||
D time.Duration `toml:"d"`
|
||||
}
|
||||
err := Unmarshal([]byte("d = \"90\"\n"), &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected an error for a duration without a unit")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "invalid duration") {
|
||||
t.Errorf("err = %v, want an invalid-duration message", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuotedStringNeverBecomesDateTime(t *testing.T) {
|
||||
// The date-time types take a bare timestamp only, so the text path is
|
||||
// excluded for them and a quoted string stays a string.
|
||||
var stamp struct {
|
||||
S time.Time `toml:"s"`
|
||||
}
|
||||
err := Unmarshal([]byte("s = \"2026-06-26T10:00:00Z\"\n"), &stamp)
|
||||
if err == nil {
|
||||
t.Fatal("expected a quoted string to be rejected for time.Time")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "cannot assign string") {
|
||||
t.Errorf("err = %v, want a cannot-assign message", err)
|
||||
}
|
||||
|
||||
var day struct {
|
||||
D LocalDate `toml:"d"`
|
||||
}
|
||||
if err := Unmarshal([]byte("d = \"1979-05-27\"\n"), &day); err == nil {
|
||||
t.Fatal("expected a quoted string to be rejected for LocalDate")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecoderMaxDepth(t *testing.T) {
|
||||
deep := func(n int) []byte {
|
||||
return []byte("v = " + strings.Repeat("[", n) + strings.Repeat("]", n) + "\n")
|
||||
}
|
||||
var cfg struct {
|
||||
V any `toml:"v"`
|
||||
}
|
||||
if err := NewDecoder().MaxDepth(4).Decode(deep(4), &cfg); err != nil {
|
||||
t.Fatalf("at the limit: %v", err)
|
||||
}
|
||||
err := NewDecoder().MaxDepth(4).Decode(deep(5), &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected a nesting error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "limit of 4") {
|
||||
t.Errorf("err = %v, want it to name the limit", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecoderMaxInputSize(t *testing.T) {
|
||||
doc := []byte("v = \"ab\"\n")
|
||||
var cfg struct {
|
||||
V string `toml:"v"`
|
||||
}
|
||||
if err := NewDecoder().MaxInputSize(len(doc)).Decode(doc, &cfg); err != nil {
|
||||
t.Fatalf("at the limit: %v", err)
|
||||
}
|
||||
err := NewDecoder().MaxInputSize(len(doc)-1).Decode(doc, &cfg)
|
||||
if err == nil {
|
||||
t.Fatal("expected a size error")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "over the limit of 8") {
|
||||
t.Errorf("err = %v, want it to name the limit", err)
|
||||
}
|
||||
// Parse carries the nesting default and no size limit.
|
||||
if _, err := ParseMap(doc); err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// --- OffsetDateTime --------------------------------------------------------
|
||||
|
||||
func TestOffsetDateTimeIsTheParsedType(t *testing.T) {
|
||||
// A document's offset date-time arrives as the wrapper, and a plain
|
||||
// time.Time destination still takes it, so a timestamp field needs no
|
||||
// change to keep working.
|
||||
in := []byte("stamp = 2026-06-26T10:00:00-07:00\n")
|
||||
want := time.Date(2026, 6, 26, 10, 0, 0, 0, time.FixedZone("", -7*3600))
|
||||
|
||||
var plain struct {
|
||||
Stamp time.Time `toml:"stamp"`
|
||||
}
|
||||
if err := Unmarshal(in, &plain); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if !plain.Stamp.Equal(want) {
|
||||
t.Errorf("time.Time destination = %v, want %v", plain.Stamp, want)
|
||||
}
|
||||
|
||||
var wrapped struct {
|
||||
Stamp OffsetDateTime `toml:"stamp"`
|
||||
}
|
||||
if err := Unmarshal(in, &wrapped); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if !wrapped.Stamp.Time.Equal(want) {
|
||||
t.Errorf("OffsetDateTime destination = %v, want %v", wrapped.Stamp.Time, want)
|
||||
}
|
||||
if got := wrapped.Stamp.String(); got != "2026-06-26T10:00-07:00" {
|
||||
t.Errorf("String() = %q, want 2026-06-26T10:00-07:00", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOffsetDateTimeFromHandBuiltTree(t *testing.T) {
|
||||
// A tree built by hand may carry a plain time.Time, which is the other
|
||||
// source of the offset kind; both date-time destinations take it.
|
||||
tree := map[string]any{"stamp": time.Date(2026, 6, 26, 10, 0, 0, 0, time.UTC)}
|
||||
want := time.Date(2026, 6, 26, 10, 0, 0, 0, time.UTC)
|
||||
|
||||
var plain struct {
|
||||
Stamp time.Time `toml:"stamp"`
|
||||
}
|
||||
if err := newDecoder().decode(tree, &plain); err != nil {
|
||||
t.Fatalf("decode: %v", err)
|
||||
}
|
||||
if !plain.Stamp.Equal(want) {
|
||||
t.Errorf("time.Time destination = %v, want %v", plain.Stamp, want)
|
||||
}
|
||||
|
||||
var wrapped struct {
|
||||
Stamp OffsetDateTime `toml:"stamp"`
|
||||
}
|
||||
if err := newDecoder().decode(tree, &wrapped); err != nil {
|
||||
t.Fatalf("decode: %v", err)
|
||||
}
|
||||
if !wrapped.Stamp.Time.Equal(want) {
|
||||
t.Errorf("OffsetDateTime destination = %v, want %v", wrapped.Stamp.Time, want)
|
||||
}
|
||||
}
|
||||
|
||||
// dateKindReceiver records the Go type UnmarshalTOML was handed.
|
||||
type dateKindReceiver struct{ Kind string }
|
||||
|
||||
func (r *dateKindReceiver) UnmarshalTOML(data any) error {
|
||||
r.Kind = fmt.Sprintf("%T", data)
|
||||
return nil
|
||||
}
|
||||
|
||||
func TestUnmarshalerReceivesOffsetDateTime(t *testing.T) {
|
||||
// The interface sees the wrapper, which names the date-time kind on its
|
||||
// own; the local kinds keep their own wrappers.
|
||||
var cfg struct {
|
||||
O dateKindReceiver `toml:"o"`
|
||||
L dateKindReceiver `toml:"l"`
|
||||
}
|
||||
in := []byte("o = 2026-06-26T10:00:00Z\nl = 2026-06-26T10:00:00\n")
|
||||
if err := Unmarshal(in, &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if cfg.O.Kind != "interpres.OffsetDateTime" {
|
||||
t.Errorf("offset kind = %q, want interpres.OffsetDateTime", cfg.O.Kind)
|
||||
}
|
||||
if cfg.L.Kind != "interpres.LocalDateTime" {
|
||||
t.Errorf("local kind = %q, want interpres.LocalDateTime", cfg.L.Kind)
|
||||
}
|
||||
}
|
||||
|
||||
+345
-51
@@ -1,31 +1,108 @@
|
||||
# API
|
||||
|
||||
The library exports the surface below from the `sourcedock.dev/petrbalvin/interpres`
|
||||
The library exports the surface below from the `sourcedock.dev/petrbalvin/interpres/v2`
|
||||
package. The snippets assume:
|
||||
|
||||
```go
|
||||
import "sourcedock.dev/petrbalvin/interpres"
|
||||
import "sourcedock.dev/petrbalvin/interpres/v2"
|
||||
```
|
||||
|
||||
The parser implements TOML 1.1: date-times and times without seconds, the
|
||||
`\e` and `\xHH` escape sequences, and multi-line inline tables with comments
|
||||
and trailing commas. The encoder emits TOML 1.1.
|
||||
|
||||
## Functions
|
||||
|
||||
### `func Parse(data []byte) (map[string]any, error)`
|
||||
### `func Parse(data []byte) (*Document, error)`
|
||||
|
||||
Decodes a TOML document into an untyped tree, using the value mapping in the
|
||||
[Decoding](#decoding) section below. Returns `*SyntaxError` on a malformed
|
||||
document. Input that is not valid UTF-8 is rejected before the parser runs.
|
||||
Equivalent to `ParseContext(context.Background(), data)`.
|
||||
Decodes a TOML document into a [Document](#documents): the values, the order the
|
||||
keys were written in, whether a table was written inline, and the comments.
|
||||
The values follow the mapping in the [Decoding](#decoding) section below.
|
||||
Returns `*SyntaxError` on a malformed document. Input that is not valid UTF-8
|
||||
is rejected with a `SyntaxError` naming the line where the invalid byte
|
||||
appears, because validity is checked during the scan. Equivalent to
|
||||
`ParseContext(context.Background(), data)`.
|
||||
|
||||
```go
|
||||
tree, err := interpres.Parse([]byte("title = \"x\"\nport = 8080\n"))
|
||||
doc, err := interpres.Parse([]byte("title = \"x\"\nport = 8080\n"))
|
||||
tree := doc.Map()
|
||||
```
|
||||
|
||||
### `func ParseContext(ctx context.Context, data []byte) (map[string]any, error)`
|
||||
### `func ParseContext(ctx context.Context, data []byte) (*Document, error)`
|
||||
|
||||
The cancellable variant of `Parse`. An already-cancelled context returns
|
||||
`ctx.Err()` before any work. During parsing the context is checked every 64
|
||||
top-level statements, so a long document aborts without running to completion.
|
||||
|
||||
### `func ParseMap(data []byte) (map[string]any, error)`
|
||||
|
||||
Decodes a TOML document into an untyped tree, the shape this package parsed
|
||||
into before [Document](#documents) existed: the order of the keys and the
|
||||
comments are not part of a map, so they are dropped. Use it when only the
|
||||
values matter, or when the extra bookkeeping of a document is not wanted.
|
||||
Equivalent to `ParseMapContext(context.Background(), data)`.
|
||||
|
||||
```go
|
||||
tree, err := interpres.ParseMap([]byte("title = \"x\"\nport = 8080\n"))
|
||||
```
|
||||
|
||||
### `func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error)`
|
||||
|
||||
The cancellable variant of `ParseMap`.
|
||||
|
||||
## Documents
|
||||
|
||||
`Parse` returns a `Document`: the value tree together with what a map cannot
|
||||
carry, which is the order the keys were written in, whether a table was written
|
||||
as an inline table or under a header, and the comments. `ParseMap` gives the
|
||||
plain tree when none of that is wanted.
|
||||
|
||||
```go
|
||||
doc, err := interpres.Parse(data)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
root := doc.Root()
|
||||
for _, key := range root.Keys() { // written order, not sorted
|
||||
entry, _ := root.Get(key)
|
||||
fmt.Println(key, entry.Value())
|
||||
}
|
||||
```
|
||||
|
||||
The values are shared with the tree `ParseMap` returns, so a value read from a
|
||||
document and from `doc.Map()` is the same value.
|
||||
|
||||
| Type | Meaning |
|
||||
|---|---|
|
||||
| `Document` | the parsed document: `Root()` for the top-level table, `Map()` for the value tree, `Footer()` for a comment block at the end |
|
||||
| `Table` | one TOML table: `Keys()` and `Entries()` in written order, `Get(key)`, `Values()` for its part of the value tree, `Inline()` |
|
||||
| `Entry` | one key: `Value()`, `Inline()`, `Table()` when the value is a table, `Elements()` for the tables of an array value |
|
||||
|
||||
`Elements()` holds one node per element of an array value: the tables of an
|
||||
array of tables, and the inline tables inside a value array, with `nil` for the
|
||||
elements that are not tables.
|
||||
|
||||
### Comments
|
||||
|
||||
A comment belongs to the line it precedes or follows, and to the node that line
|
||||
introduced:
|
||||
|
||||
| Written | Carried by |
|
||||
|---|---|
|
||||
| lines above a key | that key's `Entry`, through `Comments()` |
|
||||
| a comment beside a key | that key's `Entry`, through `Trailing()` |
|
||||
| lines above a `[header]` or `[[header]]` | that `Table`, through `Comments()` |
|
||||
| a comment beside a header | that `Table`, through `Trailing()` |
|
||||
| a comment block after the last statement | the `Document`, through `Footer()` |
|
||||
|
||||
`SetComments` and `SetTrailing` replace them. A line carries no leading `#`
|
||||
and no surrounding space, so `# note` is stored as `note` and a bare `#` as
|
||||
`""`.
|
||||
|
||||
A `Document` is not a value to marshal: `Marshal` writes values, so it refuses
|
||||
one and points at `doc.Map()`. Writing a document back, with its order and its
|
||||
comments, belongs with the editing API.
|
||||
|
||||
### `func Unmarshal(data []byte, v any) error`
|
||||
|
||||
Parses `data` and stores the result in the value pointed to by `v`, typically a
|
||||
@@ -46,7 +123,7 @@ The cancellable variant of `Unmarshal`.
|
||||
### `func Marshal(v any) ([]byte, error)`
|
||||
|
||||
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
|
||||
TOML 1.0 document. The emission rules are in the [Encoding](#encoding) section
|
||||
TOML document. The emission rules are in the [Encoding](#encoding) section
|
||||
below. Equivalent to `MarshalContext(context.Background(), v)`.
|
||||
|
||||
```go
|
||||
@@ -70,7 +147,7 @@ and every 64 fields during the reflection walk.
|
||||
| integer | `int64` |
|
||||
| float | `float64` |
|
||||
| boolean | `bool` |
|
||||
| offset date-time | `time.Time` |
|
||||
| offset date-time | `OffsetDateTime` |
|
||||
| local date-time | `LocalDateTime` |
|
||||
| local date | `LocalDate` |
|
||||
| local time | `LocalTime` |
|
||||
@@ -100,16 +177,23 @@ For a struct destination, a TOML key matches a field as follows:
|
||||
1. The `toml:"name"` tag, using the part before any comma. The literal `-`
|
||||
excludes the field.
|
||||
2. Without a tag, the lower-cased field name.
|
||||
3. The key itself is lower-cased before lookup, so the match is
|
||||
3. An anonymous (embedded) field without a tag is inlined: the decoder walks
|
||||
into the embedded struct and matches its own fields against the same keys,
|
||||
mirroring how the encoder flattens it. A nil embedded pointer struct is
|
||||
allocated on demand. An untagged embedded map receives the keys no field
|
||||
claims.
|
||||
4. The key itself is lower-cased before lookup, so the match is
|
||||
case-insensitive on both sides: `DATABASEURL` matches a field named
|
||||
`DatabaseUrl`.
|
||||
|
||||
The match is exact after lower-casing. No separator is inserted, so a TOML key
|
||||
`database_url` does not match a field named `DatabaseUrl`; tag such a field
|
||||
(`toml:"database_url"`) or use the lower-cased name as the key. When two fields
|
||||
resolve to the same name, the one declared later wins.
|
||||
(`toml:"database_url"`) or use the lower-cased name as the key. When two
|
||||
fields resolve to the same name, the shallower one wins; at equal depth, the
|
||||
one declared later wins.
|
||||
|
||||
Unknown keys are ignored by default; see [Strict decoding](#strict-decoding).
|
||||
Unknown keys are ignored by default, landing in an untagged embedded map when
|
||||
the struct has one; [Strict decoding](#strict-decoding) rejects them instead.
|
||||
|
||||
### Numeric conversion
|
||||
|
||||
@@ -119,21 +203,29 @@ The decoder converts to the destination type with explicit overflow checks:
|
||||
| Destination kind | Rule |
|
||||
|---|---|
|
||||
| `int`, `int8`, `int16`, `int32`, `int64` | the `int64` value must not overflow the destination |
|
||||
| `uint`, `uint8`, `uint16`, `uint32`, `uint64` | the value must be non-negative; `uint8`, `uint16` and `uint32` enforce their own maxima; `uint64` accepts any non-negative `int64` |
|
||||
| `float32`, `float64` | copied verbatim; an integer also coerces, so TOML `5` decodes into `5.0` |
|
||||
| `uint`, `uint8`, `uint16`, `uint32`, `uint64` | the value must be non-negative and must not overflow the destination's own width, `uint` on a 32-bit platform included; `uint64` accepts any non-negative `int64` |
|
||||
| `float32`, `float64` | copied verbatim, except that a finite value beyond the `float32` range is an overflow error rather than a silent infinity; an integer also coerces, so TOML `5` decodes into `5.0` |
|
||||
| `bool`, `string` | exact kind match only, no coercion across kinds |
|
||||
| `time.Time` | offset date-times only; no implicit conversion to or from the local variants |
|
||||
| `time.Time`, `OffsetDateTime` | offset date-times only; no implicit conversion to or from the local variants |
|
||||
|
||||
A conversion that the rules do not allow produces an error wrapped with the
|
||||
offending key or index, for example `p: interpres: integer 300 overflows uint8`.
|
||||
|
||||
### Date-time values
|
||||
|
||||
Offset date-times decode into `time.Time` and keep their offset. The local
|
||||
variants decode into `LocalDateTime`, `LocalDate` and `LocalTime`, whose
|
||||
embedded `time.Time` is normalised to UTC (midnight UTC for a local date, the
|
||||
zero date for a local time). There is no implicit conversion between the offset
|
||||
and local kinds; assigning one to the other is an error.
|
||||
Offset date-times decode into `OffsetDateTime`, whose embedded `time.Time` is the
|
||||
instant with the offset the document wrote; a destination of the plain
|
||||
`time.Time` takes the same value, so a timestamp field does not have to name the
|
||||
wrapper. The local variants decode into `LocalDateTime`, `LocalDate` and
|
||||
`LocalTime`, whose embedded `time.Time` is normalised to UTC (midnight UTC for a
|
||||
local date, the zero date for a local time). Every kind may omit the seconds as
|
||||
of TOML 1.1 (`07:32`, `1979-05-27T07:32`); such a value carries a zero second,
|
||||
and the encoder writes the seconds only when the value carries them, so a
|
||||
document written without seconds comes back without them. There is no implicit
|
||||
conversion between the offset and local kinds; assigning one to the other is an
|
||||
error. The date-time types take a bare timestamp and never a quoted string, so a
|
||||
document that writes a date-time with quotes does not decode into them, and
|
||||
neither `encoding.TextUnmarshaler` nor the embedded `time.Time` changes that.
|
||||
|
||||
### Arrays of tables
|
||||
|
||||
@@ -153,7 +245,7 @@ type Unmarshaler interface {
|
||||
```
|
||||
|
||||
`data` is whatever the parser produced for that key: `string`, `bool`, `int64`,
|
||||
`float64`, `time.Time`, `LocalDateTime`, `LocalDate`, `LocalTime`, `[]any`, or
|
||||
`float64`, `OffsetDateTime`, `LocalDateTime`, `LocalDate`, `LocalTime`, `[]any`, or
|
||||
`map[string]any`. The method inspects the value and mutates its own receiver;
|
||||
the decoder keeps whatever state the receiver stored.
|
||||
|
||||
@@ -164,6 +256,38 @@ automatically, and a nil pointer destination is allocated first. An error
|
||||
returned from `UnmarshalTOML` halts the decode and propagates wrapped with the
|
||||
key path, for example `addr: unmarshal: not a string`.
|
||||
|
||||
### Custom decoding: `encoding.TextUnmarshaler`
|
||||
|
||||
A destination type that implements `encoding.TextUnmarshaler` receives a TOML
|
||||
string as its text content, the rule `encoding/json` follows:
|
||||
|
||||
```go
|
||||
func (ip *IP) UnmarshalText(text []byte) error
|
||||
```
|
||||
|
||||
The decoder looks for the method on the destination and on its address, so a
|
||||
pointer-receiver `UnmarshalText` is invoked on an addressable struct field, and
|
||||
the elements of a slice destination are reached the same way. The text path
|
||||
applies to TOML strings only: every other value kind keeps its own rule, so
|
||||
`r = 1` does not reach a receiver that expects text. An error from
|
||||
`UnmarshalText` halts the decode and propagates with the key path and the
|
||||
prefix `unmarshal text:`, for example `addr: unmarshal text: not an address`.
|
||||
|
||||
[`UnmarshalTOML`](#custom-decoding-unmarshaler) wins over `UnmarshalText` when
|
||||
a type implements both, and the four [date-time
|
||||
types](#date-time-values) are excluded: a quoted string stays a string and
|
||||
never becomes an `OffsetDateTime` or one of the local wrappers.
|
||||
|
||||
### Durations
|
||||
|
||||
TOML has no duration type, so `time.Duration` has a rule of its own. The
|
||||
encoder writes the canonical Go form in a TOML string, `1h30m0s`, and the
|
||||
decoder reads that string back with `time.ParseDuration`. A bare integer is
|
||||
still the nanosecond count it has always been, so `from_int = 5400000000000`
|
||||
and `from_text = "1h30m"` decode to the same duration. Text that
|
||||
`time.ParseDuration` rejects, `d = "90"` among it, fails with
|
||||
`interpres: invalid duration "90"`.
|
||||
|
||||
### Strict decoding
|
||||
|
||||
By default unknown keys are dropped silently. A `Decoder` built with
|
||||
@@ -179,7 +303,8 @@ A typo such as `database_urls` then fails with
|
||||
`interpres: unknown field "database_urls" for main.Config` instead of a silent
|
||||
default-zero run. Strictness applies to every struct the decode reaches, at any
|
||||
depth, including struct elements inside slices; map destinations accept every
|
||||
key by nature.
|
||||
key by nature. When several keys are unknown, the message names the smallest
|
||||
one, so it does not depend on map iteration order.
|
||||
|
||||
### Cancellation
|
||||
|
||||
@@ -233,17 +358,40 @@ Keys that match `[A-Za-z0-9_-]+` are emitted bare, all others quoted. A
|
||||
`map[string]V` emits its keys in sorted order for deterministic output, and a
|
||||
nil map emits nothing.
|
||||
|
||||
Note the asymmetry: the encoder inlines untagged embedded structs, while the
|
||||
decoder expects them under their lower-cased type name. A struct with an
|
||||
untagged embedded struct therefore does not round-trip through `Unmarshal` into
|
||||
the same type.
|
||||
### Tag options
|
||||
|
||||
The part of a `toml` tag after the first comma carries options. Both options
|
||||
shape emission only; the decoder ignores them.
|
||||
|
||||
- `omitzero` skips the field when its value is the zero value of its type. A
|
||||
type with an `IsZero() bool` method (time.Time among them) decides through
|
||||
that method, so a zero `time.Time` or an all-zero struct disappears from
|
||||
the output.
|
||||
- `omitempty` skips the field when it holds an empty collection: a nil or
|
||||
empty slice or array, or a nil or empty map. Strings and other scalars are
|
||||
not covered by `omitempty`; use `omitzero` for those.
|
||||
|
||||
```go
|
||||
type Config struct {
|
||||
Host string `toml:"host,omitzero"`
|
||||
Started time.Time `toml:"started,omitzero"`
|
||||
Tags []string `toml:"tags,omitempty"`
|
||||
}
|
||||
```
|
||||
|
||||
Options combine after the name: `toml:"name,omitempty,omitzero"` is valid, and
|
||||
an unknown option is ignored.
|
||||
|
||||
Untagged embedded fields round-trip: the decoder inlines embedded structs and
|
||||
routes unclaimed keys into an embedded map exactly where the encoder flattened
|
||||
them.
|
||||
|
||||
### Group-by-kind layout
|
||||
|
||||
By default every table is emitted with its entries grouped by kind:
|
||||
|
||||
1. scalars (`string`, `int64`, `float64`, `bool`, `time.Time`,
|
||||
`LocalDateTime`, `LocalDate`, `LocalTime`)
|
||||
`OffsetDateTime`, `LocalDateTime`, `LocalDate`, `LocalTime`)
|
||||
2. sub-tables (structs and `map[string]V` values)
|
||||
3. arrays of tables (`[]struct` and `[]map[string]V`)
|
||||
|
||||
@@ -278,8 +426,20 @@ type Marshaler interface {
|
||||
The returned value is encoded as if it had been passed in place of the
|
||||
receiver, so it may be a scalar, a slice, an array of tables, or another
|
||||
struct or map, including the `Marshaler` result of another type; the encoder
|
||||
recurses. An error returned from `MarshalTOML` fails the marshal wrapped with
|
||||
the key path, for example `interpres: server.port: bad timestamp`.
|
||||
recurses, and the result is normalised like any other value, so a method may
|
||||
return a plain `int` or a `time.Duration`.
|
||||
|
||||
An error returned from `MarshalTOML` fails the marshal wrapped with the key
|
||||
path, for example `interpres: server.port: bad timestamp`. A result of `nil` with
|
||||
a nil error fails the same way with `MarshalTOML returned a nil value`: nil has
|
||||
no TOML representation, so dropping the field silently is not an option.
|
||||
|
||||
The method is reached for every value the walk meets, array elements included:
|
||||
an element that renders itself as a table keeps the `[[header]]` form, one that
|
||||
renders itself as a scalar turns the array into a value array, and the method
|
||||
runs once per element. It is looked up on the value and on its address, so a
|
||||
pointer-receiver method is called for a field or an element, exactly as
|
||||
`MarshalText` is.
|
||||
|
||||
```go
|
||||
type Port int
|
||||
@@ -289,6 +449,44 @@ func (p Port) MarshalTOML() (any, error) {
|
||||
}
|
||||
```
|
||||
|
||||
### Custom encoding: `encoding.TextMarshaler`
|
||||
|
||||
A type that implements `encoding.TextMarshaler` is encoded as a TOML string
|
||||
holding the text the method returns, which is the rule `encoding/json` follows:
|
||||
|
||||
```go
|
||||
func (ip IP) MarshalText() ([]byte, error)
|
||||
```
|
||||
|
||||
The encoder looks for the method on the value and on its address, so a
|
||||
pointer-receiver `MarshalText` is found on a struct field of an addressable
|
||||
value (pass a pointer to `Marshal`) and always on a slice element. `net.IP`,
|
||||
`netip.Addr` and user types follow this rule, and a struct that implements the
|
||||
interface becomes a string rather than a table. `MarshalTOML` wins when a type
|
||||
implements both, the four [date-time types](#date-time-values) keep their bare
|
||||
timestamp form, and text that is not valid UTF-8 is an error rather than a
|
||||
replacement character.
|
||||
|
||||
A duration carries no text method of its own; see [Durations](#durations) for
|
||||
its rule.
|
||||
|
||||
### Arrays
|
||||
|
||||
An array whose every element is a table (`[]struct`, `[]map[string]V`, after
|
||||
pointer dereference) emits as an array of tables. TOML also lets one array mix
|
||||
tables with scalars; such an array emits as a plain value array, with the
|
||||
table elements rendered as inline tables:
|
||||
|
||||
```go
|
||||
tree, _ := interpres.Parse([]byte(`arr = [1, {a = 2}, "x"]`))
|
||||
out, _ := interpres.Marshal(tree) // arr = [1, {a = 2}, "x"]
|
||||
```
|
||||
|
||||
A `[]any` holding only tables keeps the value-array form as well, because that
|
||||
is the shape `Parse` gives a value array of inline tables; emitting it as
|
||||
`[[headers]]` would re-parse as `[]map[string]any` and change the value's type
|
||||
across a round-trip.
|
||||
|
||||
### Empty arrays
|
||||
|
||||
A nil slice is always omitted. An empty (length 0) array of tables is always
|
||||
@@ -299,17 +497,61 @@ omitted, because TOML forbids an empty `[[a]]`. Other empty arrays emit as
|
||||
### Long strings
|
||||
|
||||
By default every string is emitted as a basic `"..."` string with the escapes
|
||||
TOML requires, and a string containing a newline is emitted as an escaped
|
||||
multi-line basic string. `UseLiteralMultiline(threshold)` switches strings that
|
||||
contain a newline and are at least `threshold` bytes long to the literal
|
||||
`'''...'''` form, which carries the newlines verbatim:
|
||||
TOML requires, a newline among them as `\n`. `UseLiteralMultiline(threshold)`
|
||||
switches strings that contain a newline and are at least `threshold` bytes long
|
||||
to the literal `'''...'''` form, which carries the newlines verbatim:
|
||||
|
||||
```go
|
||||
out, err := interpres.NewEncoder().UseLiteralMultiline(80).Marshal(cfg)
|
||||
```
|
||||
|
||||
Single-line strings keep the basic form regardless of the threshold, and a
|
||||
threshold of `0` or less disables the option.
|
||||
threshold of `0` or less disables the option. A string the literal form cannot
|
||||
carry verbatim (an embedded run of three single quotes, a control character
|
||||
other than tab or newline, or a carriage return outside a CRLF pair) also keeps
|
||||
the basic form, so the output always re-parses to the same value.
|
||||
|
||||
### Inline tables
|
||||
|
||||
A table element of a value array, and a sub-table inlined by
|
||||
[`InlineTables`](#compact-documents), is written as one `{a = 1, b = 2}` line
|
||||
while it fits. An inline table that would pass the hundredth column carries
|
||||
newlines and a trailing comma instead, which TOML 1.1 allows:
|
||||
|
||||
```toml
|
||||
arr = [1, {
|
||||
n = 1,
|
||||
name = "a value long enough to push this line well past the one hundred column limit",
|
||||
}]
|
||||
```
|
||||
|
||||
The closing brace and the entries are indented one tab per nesting level, a
|
||||
nested table is measured on its own line, and the output re-parses to the same
|
||||
value either way.
|
||||
|
||||
### Compact documents
|
||||
|
||||
`InlineTables(threshold)` writes a sub-table as an inline table when its
|
||||
single-line rendering is at most `threshold` bytes, and as a table header
|
||||
section when it is longer. A document of small tables therefore grows shorter:
|
||||
|
||||
```go
|
||||
out, err := interpres.NewEncoder().InlineTables(60).Marshal(cfg)
|
||||
```
|
||||
|
||||
With `60` and a table of three short entries, the same value is written
|
||||
|
||||
```toml
|
||||
server = {host = "127.0.0.1", port = 9090, tls = {on = false}}
|
||||
```
|
||||
|
||||
instead of three lines under a `[server]` header and a `[server.tls]` section.
|
||||
A nested sub-table takes part in the same way, and the whole option is off at
|
||||
`0` or less. Two limits are deliberate. An array of tables keeps the `[[a]]`
|
||||
header form, because its inline form re-parses as a value array and would change
|
||||
the value's Go type. And because an inlined table is a value line, every one of
|
||||
them precedes the first header of its document, so a table inlined next to a
|
||||
header is not read back as part of that header's section.
|
||||
|
||||
### Cancellation
|
||||
|
||||
@@ -325,6 +567,8 @@ The output is not byte-identical to any document that produced the value:
|
||||
- map keys are emitted in sorted order
|
||||
- the choice between `[table]` headers and inline tables is not preserved
|
||||
- strings use the basic quoted form unless the literal option above applies
|
||||
- a date-time drops its zero seconds and the trailing zeros of its fraction, so
|
||||
`07:32:00` is written `07:32`; both are the same value
|
||||
- floats always carry a `.` or an exponent, so a float `1` is emitted as `1.0`
|
||||
and stays distinguishable from the integer `1` across a round-trip; negative
|
||||
zero is normalised to `0.0`
|
||||
@@ -352,9 +596,11 @@ sequenceDiagram
|
||||
|
||||
### `type SyntaxError struct{ Line int; Msg string }`
|
||||
|
||||
Describes a malformed TOML document; `Line` is 1-based and `Error()` renders as
|
||||
`interpres: line N: msg`. Read the structured fields with a type assertion or
|
||||
`errors.AsType`:
|
||||
Describes a document the parser rejected, with the 1-based `Line` at which it
|
||||
gave up and `Error()` rendering as `interpres: line N: msg`. A malformed
|
||||
document is the usual cause; the nesting limit and an input that is not valid
|
||||
UTF-8 report through the same type. Read the structured fields with a type
|
||||
assertion or `errors.AsType`:
|
||||
|
||||
```go
|
||||
if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
|
||||
@@ -362,6 +608,27 @@ if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
|
||||
}
|
||||
```
|
||||
|
||||
### `type DecodeError struct{ Path []string; Err error }`
|
||||
|
||||
Wraps a decoding failure with the key path at which it happened. `Path` lists
|
||||
one segment per level from the document root, the outermost key first: a key
|
||||
contributes its name, an array element its bracketed index, so the path of the
|
||||
`weight` field in the first item reads `["items", "[0]", "weight"]`. The
|
||||
rendered message is unchanged by the type; read the fields instead of parsing
|
||||
the message:
|
||||
|
||||
```go
|
||||
if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
|
||||
fmt.Println(de.Path, de.Err)
|
||||
}
|
||||
```
|
||||
|
||||
### `type EncodeError struct{ Path string; Err error }`
|
||||
|
||||
Wraps an encoding failure with the key path of the value that failed, in the
|
||||
document's own notation: `server.ports[2]`. Read it with `errors.AsType` the
|
||||
same way.
|
||||
|
||||
### `type Decoder`
|
||||
|
||||
Configurable strictness for decoding, constructed with `NewDecoder`. Set up
|
||||
@@ -369,6 +636,19 @@ with `DisallowUnknownFields`, then call `Decode` or `DecodeContext` any number
|
||||
of times. A configured `Decoder` holds no per-call state and is safe for
|
||||
concurrent use.
|
||||
|
||||
| Method | Default | Effect |
|
||||
|---|---|---|
|
||||
| `DisallowUnknownFields()` | off | a key with no matching struct field is an error |
|
||||
| `MaxDepth(depth int)` | `10000` | bound how deeply arrays and inline tables may nest |
|
||||
| `MaxInputSize(size int)` | no limit | bound the size of the document, in bytes |
|
||||
|
||||
The nesting limit protects the stack, because the parser is a recursive
|
||||
descent: a deeper document is rejected with a `SyntaxError` naming the limit
|
||||
rather than running the stack out. `Parse` and `ParseContext` carry that same
|
||||
default but take no options. The size limit is off by default, because the
|
||||
caller already holds the bytes and the size is therefore a policy, not a
|
||||
protection the library can impose on its own.
|
||||
|
||||
### `type Encoder`
|
||||
|
||||
Configurable emission policy, constructed with `NewEncoder`. The option state
|
||||
@@ -380,12 +660,14 @@ encoder:
|
||||
| `GroupByKind(v bool)` | `true` | group entries as scalars, then sub-tables, then arrays of tables; `false` preserves declaration order |
|
||||
| `OmitEmptyArrays()` | off | skip `key = []` for empty scalar arrays |
|
||||
| `UseLiteralMultiline(threshold int)` | `0` | emit multi-line strings of at least `threshold` bytes as literal `'''...'''` |
|
||||
| `InlineTables(threshold int)` | `0` | write a sub-table inline when its single-line form is at most `threshold` bytes |
|
||||
|
||||
```go
|
||||
out, err := interpres.NewEncoder().
|
||||
GroupByKind(false).
|
||||
OmitEmptyArrays().
|
||||
UseLiteralMultiline(80).
|
||||
InlineTables(60).
|
||||
MarshalContext(ctx, cfg)
|
||||
```
|
||||
|
||||
@@ -393,6 +675,11 @@ A configured `Encoder` holds no per-call state; each `Marshal` or
|
||||
`MarshalContext` call copies the options and is safe for concurrent use, as
|
||||
long as no setter races with a call.
|
||||
|
||||
### `type Document`, `type Table`, `type Entry`
|
||||
|
||||
See [Documents](#documents). A `Document` is what `Parse` returns, and it is
|
||||
not a value `Marshal` accepts.
|
||||
|
||||
### `type Marshaler interface{ MarshalTOML() (any, error) }`
|
||||
|
||||
See [Custom encoding](#custom-encoding-marshaler).
|
||||
@@ -404,25 +691,32 @@ See [Custom decoding](#custom-decoding-unmarshaler).
|
||||
### Date-time wrappers
|
||||
|
||||
```go
|
||||
type LocalDateTime struct{ time.Time } // 1979-05-27T07:32:00
|
||||
type LocalDate struct{ time.Time } // 1979-05-27
|
||||
type LocalTime struct{ time.Time } // 07:32:00.999999
|
||||
type OffsetDateTime struct{ time.Time } // 1979-05-27T07:32:00Z
|
||||
type LocalDateTime struct{ time.Time } // 1979-05-27T07:32:00
|
||||
type LocalDate struct{ time.Time } // 1979-05-27
|
||||
type LocalTime struct{ time.Time } // 07:32:00.999999
|
||||
```
|
||||
|
||||
Each carries a `String()` method returning the TOML-canonical rendering, with
|
||||
the fractional second zero-padded to nanosecond precision when present. The
|
||||
types are produced by `Parse` and accepted by `Marshal`.
|
||||
Each carries a `String()` method returning the TOML-canonical rendering: the
|
||||
seconds appear only when the value carries them, and a fractional second drops
|
||||
its trailing zeros, so `07:32:00` renders as `07:32` and a half second as
|
||||
`00.5`. The types are produced by `Parse` and accepted by `Marshal`, which
|
||||
writes them through `String()`.
|
||||
|
||||
## Errors
|
||||
|
||||
The entry points return:
|
||||
|
||||
- `*SyntaxError` for a malformed document, with the 1-based line
|
||||
- a plain error for everything else: a non-pointer decode target, a type
|
||||
mismatch, an overflow, a marshal policy violation, a cancelled context
|
||||
- `*SyntaxError` for a malformed document, with the 1-based line; the nesting
|
||||
limit reports through it as well
|
||||
- `*DecodeError` for a decoding failure, with the key path in `Path`
|
||||
- `*EncodeError` for an encoding failure, with the key path in `Path`
|
||||
- a plain error for the rest: a non-pointer decode target, a cancelled
|
||||
context, a key that is not valid UTF-8, an input over the size limit
|
||||
|
||||
Decode and encode failures are wrapped with the key path or element index using
|
||||
`fmt.Errorf`, so `errors.Is` and `errors.AsType` see through them.
|
||||
Decode and encode failures carry the key path or element index in the typed
|
||||
wrappers above, so `errors.Is` and `errors.AsType` see through them and the
|
||||
path reads from a field instead of the message text.
|
||||
|
||||
## Notes
|
||||
|
||||
|
||||
+25
-10
@@ -6,10 +6,10 @@ source tree; nothing is aspirational.
|
||||
## Overview
|
||||
|
||||
interpres is one public library package, one command, and one example. The
|
||||
library implements the whole of TOML 1.0, decoding and encoding, in the
|
||||
standard library alone; the command wraps the parser for the toml-test
|
||||
compliance harness, against which it stands at 185 valid and 371 invalid cases
|
||||
with zero failures; the example demonstrates the API.
|
||||
library implements the whole of TOML 1.1, decoding and encoding, in the
|
||||
standard library alone; the command wraps the parser and the encoder for the
|
||||
toml-test compliance harness, against which it stands at 214 valid, 467 invalid
|
||||
and 214 encoder cases with zero failures; the example demonstrates the API.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
@@ -36,14 +36,15 @@ strict validation.
|
||||
| Path | Responsibility |
|
||||
|---|---|
|
||||
| `.` (package `interpres`) | The whole library. `interpres.go` declares the exported surface (`Parse`, `Unmarshal`, `Marshal`, the `*Context` variants, `Decoder`, `Encoder`, `Marshaler`, `Unmarshaler`, `SyntaxError`, the local date-time types); everything below it is unexported. |
|
||||
| `cmd/interpres-decode` | The toml-test adapter. Reads TOML on stdin, writes tagged JSON on stdout. Owns no parsing logic. |
|
||||
| `cmd/interpres-decode` | The toml-test adapter, both directions. Reads TOML on stdin, writes tagged JSON on stdout; with `-encode` it reads tagged JSON and writes TOML. Owns no parsing logic and no emission logic. |
|
||||
| `examples/basic` | A runnable tour of the API. Documentation in executable form, not part of the library. |
|
||||
|
||||
Inside the library package, one file owns one concern:
|
||||
|
||||
| File | Responsibility |
|
||||
|---|---|
|
||||
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 (table redefinitions, dotted keys, arrays of tables). Reports a 1-based line on failure. |
|
||||
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree, records the nodes a [Document](API.md#documents) is built from, and enforces the structural rules of TOML 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
|
||||
| `document.go` | The parsed-document types: `Document`, `Table` and `Entry`, which carry the key order, whether a table was written inline, and the comments. The values they expose are the parser's own tree, not a copy. |
|
||||
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
|
||||
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
|
||||
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
|
||||
@@ -99,11 +100,25 @@ sequenceDiagram
|
||||
`Decode`, `DecodeContext`, `Marshal` and `MarshalContext` call allocates its
|
||||
own unexported worker, so a configured type is safe for concurrent use; the
|
||||
setter methods are not, and must finish before the value is shared.
|
||||
- The parser is allocated per `ParseContext` call; nothing is cached between
|
||||
documents.
|
||||
- The parser is allocated per `ParseContext` call; the parser itself caches
|
||||
nothing between documents.
|
||||
- The shared state is a set of caches and pools whose entries are immutable
|
||||
once published, each growing with the number of distinct types rather than
|
||||
with document size: the struct-schema cache in `decode.go` (a `sync.Map`
|
||||
keyed on `reflect.Type`, holding the flattened field layout the decoder and
|
||||
the encoder both consult), the per-type interface flag caches in `decode.go`
|
||||
and `encode.go` (recording where `Marshaler`, `Unmarshaler` and the text
|
||||
interfaces can be found, so a walk builds an interface value only where the
|
||||
assertion can succeed), each fronted by a monomorphic hint holding the type
|
||||
resolved last, and the encoder's output-buffer pool in `encode.go`
|
||||
(`sync.Pool`, buffers returned to it only within a 1 MiB retention cap). A
|
||||
published schema or flag set never mutates, so concurrent callers only race
|
||||
to build an identical value, the same trade-off `encoding/json`'s field
|
||||
cache makes.
|
||||
- The date-time wrappers are values, not pointers, and are immutable in use.
|
||||
- Nothing in the library starts goroutines or holds locks; concurrency safety
|
||||
comes from having no shared mutable state.
|
||||
- Nothing in the library starts goroutines; apart from the caches and the pool
|
||||
above, which never mutate a published entry, there is no shared mutable
|
||||
state.
|
||||
|
||||
## Dependencies
|
||||
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
# Benchmarking
|
||||
|
||||
How the performance numbers attached to this project are measured, so that a
|
||||
number in a changelog entry or a release note can be reproduced and trusted.
|
||||
|
||||
## The suite
|
||||
|
||||
The benchmarks live in `bench_test.go`, next to the code they measure:
|
||||
|
||||
| Benchmark | What it measures |
|
||||
|---|---|
|
||||
| `BenchmarkParse` | `ParseMap` over a representative configuration document |
|
||||
| `BenchmarkMarshal` | `Marshal` of the tree `ParseMap` produced from the same document |
|
||||
| `BenchmarkStrictDecode` | `Decode` into a struct under `DisallowUnknownFields` |
|
||||
| `BenchmarkParseLong` | `ParseMap` over a generated document with about 2000 array-of-tables entries |
|
||||
| `BenchmarkStrictDecodeLong` | `Decode` into a typed document under `DisallowUnknownFields`, over the same long document |
|
||||
| `BenchmarkMarshalLong` | `Marshal` of the tree `ParseMap` produced from the long document |
|
||||
|
||||
## Running
|
||||
|
||||
```sh
|
||||
just bench
|
||||
```
|
||||
|
||||
The recipe runs the suite with `-benchmem -count=5`. Every benchmark uses
|
||||
`b.Loop`, so setup runs outside the timed region, and `ReportAllocs` records
|
||||
allocations per operation. The parse and marshal benchmarks set `SetBytes`, so
|
||||
their results read as input bytes per second.
|
||||
|
||||
## Method
|
||||
|
||||
- An idle machine only: a loaded box times whatever else is running, and the
|
||||
fastest sample can land on the wrong function.
|
||||
- An A/B comparison runs both variants inside one process, in one binary;
|
||||
separate processes of identical binaries differ by more than the effect
|
||||
being measured.
|
||||
- The five counts are compared through their medians, allocations and bytes
|
||||
per operation alongside the times. Differences within 1 to 2 percent are
|
||||
noise; only a difference beyond that is a result.
|
||||
- When timing is hopeless, the allocation and byte counts are the result.
|
||||
|
||||
## Reports
|
||||
|
||||
The repository stores no benchmark reports. A performance claim in
|
||||
`CHANGELOG.md` is measured with the method above on the change that makes it,
|
||||
and the number travels with the claim.
|
||||
+75
-13
@@ -1,26 +1,53 @@
|
||||
# Command line
|
||||
|
||||
The reference below is taken from the program itself. `interpres-decode` is the
|
||||
toml-test harness adapter, not a general-purpose tool: it takes no flags and no
|
||||
arguments, reads one TOML document from stdin, and writes the toml-test
|
||||
tagged-JSON form to stdout.
|
||||
The reference below is taken from the program itself. `interpres-decode` is
|
||||
the toml-test harness adapter in both directions, decoding TOML into tagged
|
||||
JSON and encoding tagged JSON back into TOML, and it also validates documents.
|
||||
Install it with Go itself, no release assets involved:
|
||||
|
||||
```sh
|
||||
go install sourcedock.dev/petrbalvin/interpres/v2/cmd/interpres-decode@latest
|
||||
```
|
||||
|
||||
## Synopsis
|
||||
|
||||
```sh
|
||||
interpres-decode < document.toml
|
||||
interpres-decode [flags]
|
||||
interpres-decode -encode
|
||||
interpres-decode -validate [file ...]
|
||||
```
|
||||
|
||||
Build it with `just build`, which compiles it into `bin/interpres-decode`, or
|
||||
run it straight from the module directory with `just run`.
|
||||
Without `-validate` or `-encode` the program is the decoding half of the
|
||||
toml-test adapter: it takes no arguments, reads one TOML document from stdin,
|
||||
and writes the toml-test tagged-JSON form to stdout. Build it locally with
|
||||
`just build`, which compiles it into `bin/interpres-decode`, or run it
|
||||
straight from the module directory with `just run`.
|
||||
|
||||
With `-encode` the direction is reversed: the program reads a tagged-JSON
|
||||
description from stdin and writes the TOML document it describes to stdout,
|
||||
which is the shape toml-test expects of an encoder command. It takes no
|
||||
arguments either, and `-validate` and `-encode` cannot be combined.
|
||||
|
||||
With `-validate` the program parses each named file instead, or stdin when no
|
||||
file is named, and prints one line per invalid document to stderr. It is
|
||||
quiet on valid documents, which is the shape a CI step wants. The `-` name
|
||||
means stdin.
|
||||
|
||||
## Flags
|
||||
|
||||
| Flag | Effect |
|
||||
|---|---|
|
||||
| `-validate` | validate the documents instead of emitting tagged JSON |
|
||||
| `-encode` | read tagged JSON from stdin and write TOML instead |
|
||||
| `-h` | print the usage |
|
||||
|
||||
## Exit codes
|
||||
|
||||
| Code | Meaning |
|
||||
|---|---|
|
||||
| `0` | the document parsed, tagged JSON written to stdout |
|
||||
| `1` | parse error, the document is malformed; the message goes to stderr |
|
||||
| `2` | reading stdin failed, or a value has no tagged representation |
|
||||
| `0` | adapter: the document parsed and the tagged JSON was written; encode: the TOML was written; validate: every document parsed |
|
||||
| `1` | adapter: parse error; validate: at least one document is invalid |
|
||||
| `2` | a usage error, a read failure, malformed tagged JSON, or a value with no TOML representation |
|
||||
|
||||
## Wire format
|
||||
|
||||
@@ -47,6 +74,14 @@ wrapped in an object with a `type` and a `value`:
|
||||
| local date | `date-local` | `1979-05-27` |
|
||||
| local time | `time-local` | `07:32:00.999999` |
|
||||
|
||||
The `-encode` mode reads exactly this form back. Two properties of it are
|
||||
worth knowing. A float whose value has no fraction and no exponent is written
|
||||
as a bare integer string, `{"type": "float", "value": "1"}`, so there the tag
|
||||
decides the type and not the literal. And the form cannot tell an array of
|
||||
tables from a value array of inline tables, so the adapter writes the header
|
||||
form, `[[a]]`, for an array whose every element is a JSON object; a mixed
|
||||
array keeps the value form.
|
||||
|
||||
## Examples
|
||||
|
||||
Echo a small document through the adapter:
|
||||
@@ -59,13 +94,40 @@ port = 9090
|
||||
' | ./bin/interpres-decode
|
||||
```
|
||||
|
||||
The output is the equivalent value tree as one JSON object. Run the official
|
||||
compliance suite against the binary:
|
||||
The output is the equivalent value tree as one JSON object. Turn a description
|
||||
back into TOML with `-encode`:
|
||||
|
||||
```sh
|
||||
echo '{"title": {"type": "string", "value": "hello"}}' | ./bin/interpres-decode -encode
|
||||
```
|
||||
|
||||
```toml
|
||||
title = "hello"
|
||||
```
|
||||
|
||||
Validate the TOML files of another repository in CI:
|
||||
|
||||
```sh
|
||||
interpres-decode -validate config.toml deploy/example.toml
|
||||
```
|
||||
|
||||
An invalid document reports the file and the library's line number:
|
||||
|
||||
```sh
|
||||
$ interpres-decode -validate bad.toml
|
||||
bad.toml: interpres: line 1: expected a value
|
||||
$ echo $?
|
||||
1
|
||||
```
|
||||
|
||||
Run the official compliance suite against the adapter:
|
||||
|
||||
```sh
|
||||
just toml-test
|
||||
```
|
||||
|
||||
That recipe needs the `toml-test` binary on `PATH`, installed with
|
||||
`go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0`. The full
|
||||
`go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0`. It runs
|
||||
the suite in both directions: the decoder against the valid and invalid
|
||||
corpora, and the encoder against the tagged JSON of the valid one. The full
|
||||
reference for the library itself is [API.md](API.md).
|
||||
|
||||
+12
-9
@@ -4,10 +4,10 @@ How to work on interpres.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Go 1.27.0, the version the `go` directive in `go.mod` declares.
|
||||
- Go 1.27.1, the version the `go` directive in `go.mod` declares.
|
||||
- [just](https://github.com/casey/just) for the recipes.
|
||||
- The `toml-test` binary on `PATH` for the compliance recipe, installed with
|
||||
`go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0`.
|
||||
`go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0`.
|
||||
|
||||
The module has no third-party dependencies, so there is nothing else to fetch.
|
||||
|
||||
@@ -41,7 +41,7 @@ prints the same list.
|
||||
| `just run` | `go run ./cmd/interpres-decode`, reads TOML from stdin |
|
||||
| `just dev` | the same run, for iterating |
|
||||
| `just example` | `go run ./examples/basic`, the usage tour |
|
||||
| `just toml-test` | builds the adapter and runs the official toml-test compliance suite against it |
|
||||
| `just toml-test` | builds the adapter and runs the official toml-test compliance suite against it, decoder and encoder |
|
||||
| `just coverage-html` | `just test`, then `go tool cover -html` into `coverage.html` |
|
||||
| `just install` | builds, then copies the binary into `~/.local/bin` (`BINDIR` overrides) |
|
||||
| `just uninstall` | removes the installed binary |
|
||||
@@ -77,7 +77,9 @@ just bench
|
||||
```
|
||||
|
||||
Benchmark on an idle machine, and compare only runs made in one process against
|
||||
each other. The recipe sweeps `./...` five times with `-benchmem`.
|
||||
each other. The recipe sweeps `./...` five times with `-benchmem`. The binding
|
||||
measurement method, and what counts as a result, is in
|
||||
[docs/BENCHMARKING.md](BENCHMARKING.md).
|
||||
|
||||
## Debugging the build
|
||||
|
||||
@@ -97,12 +99,13 @@ pipeline.
|
||||
|---|---|---|
|
||||
| `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite |
|
||||
| `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally |
|
||||
| `release.yml` | a `v*` tag | the same gates plus the race detector, then the Gitea release from the CHANGELOG section |
|
||||
| `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut |
|
||||
|
||||
## Releases
|
||||
|
||||
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The
|
||||
tag drives the release workflow: it validates the tag, runs the full gate set
|
||||
including the race detector, extracts the matching `## [X.Y.Z]` section from
|
||||
`CHANGELOG.md`, and publishes the release with that section as its body. A
|
||||
library ships no binaries, so the release carries the notes and nothing else.
|
||||
tag drives the release workflow: it validates the tag, runs the static gates
|
||||
and the test suite with the coverage floor, extracts the matching `## [X.Y.Z]`
|
||||
section from `CHANGELOG.md`, and publishes the release with that section as its
|
||||
body. A library ships no binaries, so the release carries the notes and nothing
|
||||
else.
|
||||
|
||||
+196
@@ -0,0 +1,196 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package interpres
|
||||
|
||||
// A Document is a parsed TOML document: the values, plus what the map shape
|
||||
// cannot carry, which is the order the keys were written in, whether a table
|
||||
// was written inline or under a header, and the comments.
|
||||
//
|
||||
// The values are the tree ParseMap returns, shared rather than copied, so a
|
||||
// value read from a Document and from that map is the same value. A comment
|
||||
// belongs to the statement it precedes: the lines above a key belong to the
|
||||
// key, the lines above a header belong to the header's table, and a comment
|
||||
// block after the last statement belongs to the Document.
|
||||
//
|
||||
// Comments inside array and inline-table values are not carried yet; the
|
||||
// parser skips them as it always has.
|
||||
type Document struct {
|
||||
root *Table
|
||||
footer []string
|
||||
}
|
||||
|
||||
// Root returns the document's root table.
|
||||
func (d *Document) Root() *Table { return d.root }
|
||||
|
||||
// Map returns the value tree, the shape ParseMap gives. It is the tree the
|
||||
// document was parsed into, not a copy.
|
||||
func (d *Document) Map() map[string]any { return d.root.values }
|
||||
|
||||
// Footer returns the comment lines that follow the last statement, and every
|
||||
// line of a document that holds no statement at all.
|
||||
func (d *Document) Footer() []string { return d.footer }
|
||||
|
||||
// SetFooter replaces those lines.
|
||||
func (d *Document) SetFooter(lines []string) { d.footer = lines }
|
||||
|
||||
// A Table is one TOML table: its values, its keys in written order, and the
|
||||
// comments around the header or the key that introduced it.
|
||||
type Table struct {
|
||||
values map[string]any
|
||||
entries []*Entry
|
||||
index map[string]*Entry
|
||||
|
||||
// inline records that the table was written as an inline table, `{…}`,
|
||||
// rather than under a header or as a dotted key.
|
||||
inline bool
|
||||
|
||||
// comments are the lines above the table's header, trailing is the comment
|
||||
// on the header's own line. Both are empty for a table a dotted key
|
||||
// introduced, which has no line of its own.
|
||||
comments []string
|
||||
trailing string
|
||||
}
|
||||
|
||||
func newTable(values map[string]any) *Table {
|
||||
return &Table{values: values, index: map[string]*Entry{}}
|
||||
}
|
||||
|
||||
// Keys returns the table's keys in the order they were written.
|
||||
func (t *Table) Keys() []string {
|
||||
keys := make([]string, len(t.entries))
|
||||
for i, e := range t.entries {
|
||||
keys[i] = e.key
|
||||
}
|
||||
return keys
|
||||
}
|
||||
|
||||
// Values returns the table's values, which is the map the value tree holds for
|
||||
// it.
|
||||
func (t *Table) Values() map[string]any { return t.values }
|
||||
|
||||
// Entries returns the table's entries in written order.
|
||||
func (t *Table) Entries() []*Entry { return t.entries }
|
||||
|
||||
// Get returns the entry for key, and whether the table has one.
|
||||
func (t *Table) Get(key string) (*Entry, bool) {
|
||||
e, ok := t.index[key]
|
||||
return e, ok
|
||||
}
|
||||
|
||||
// Inline reports whether the table was written as an inline table, `{…}`,
|
||||
// rather than under a header or introduced by a dotted key.
|
||||
func (t *Table) Inline() bool { return t.inline }
|
||||
|
||||
// Comments returns the comment lines above the table's header, or above the
|
||||
// key that introduced it. Lines carry no leading '#' and no surrounding space.
|
||||
func (t *Table) Comments() []string { return t.comments }
|
||||
|
||||
// SetComments replaces those lines. Each line is written back with a "# " in
|
||||
// front of it, so a line should not carry one.
|
||||
func (t *Table) SetComments(lines []string) { t.comments = lines }
|
||||
|
||||
// Trailing returns the comment on the header's own line, without the '#'.
|
||||
func (t *Table) Trailing() string { return t.trailing }
|
||||
|
||||
// SetTrailing replaces that comment.
|
||||
func (t *Table) SetTrailing(line string) { t.trailing = line }
|
||||
|
||||
// addValue records a key of the table, in written order.
|
||||
func (t *Table) addValue(key string, val any, inline bool) *Entry {
|
||||
e := &Entry{table: t, key: key, inline: inline}
|
||||
t.entries = append(t.entries, e)
|
||||
t.index[key] = e
|
||||
if node, ok := val.(map[string]any); ok {
|
||||
e.child = newTable(node)
|
||||
}
|
||||
return e
|
||||
}
|
||||
|
||||
// addTable records a key whose value is a table introduced by a header or a
|
||||
// dotted key, and returns the table's node.
|
||||
func (t *Table) addTable(key string, values map[string]any) *Table {
|
||||
if e, ok := t.index[key]; ok {
|
||||
// The key was seen before, as the leaf of an earlier dotted key.
|
||||
if e.child == nil {
|
||||
e.child = newTable(values)
|
||||
}
|
||||
return e.child
|
||||
}
|
||||
e := t.addValue(key, values, false)
|
||||
e.child = newTable(values)
|
||||
return e.child
|
||||
}
|
||||
|
||||
// addElement records one element of an array of tables, and returns its node.
|
||||
func (t *Table) addElement(key string, values map[string]any) *Table {
|
||||
e, ok := t.index[key]
|
||||
if !ok {
|
||||
e = t.addValue(key, nil, false)
|
||||
e.elements = []*Table{}
|
||||
}
|
||||
el := newTable(values)
|
||||
e.elements = append(e.elements, el)
|
||||
return el
|
||||
}
|
||||
|
||||
// child returns the node of a table-valued key, or nil.
|
||||
func (t *Table) child(key string) *Table {
|
||||
if e, ok := t.index[key]; ok {
|
||||
return e.child
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// lastElement returns the node of the newest element of an array of tables.
|
||||
func (t *Table) lastElement(key string) *Table {
|
||||
if e, ok := t.index[key]; ok && len(e.elements) > 0 {
|
||||
return e.elements[len(e.elements)-1]
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// An Entry is one key of a table: the value and the comments around the key.
|
||||
type Entry struct {
|
||||
table *Table
|
||||
key string
|
||||
inline bool
|
||||
comments []string
|
||||
trailing string
|
||||
|
||||
// child is the table the value is, and elements are the tables of an array
|
||||
// of tables; one of them is set only when the value has that shape.
|
||||
child *Table
|
||||
elements []*Table
|
||||
}
|
||||
|
||||
// Key returns the key as it was written.
|
||||
func (e *Entry) Key() string { return e.key }
|
||||
|
||||
// Value returns the value the key holds. It is read from the table's map, so
|
||||
// it stays current if that map is changed.
|
||||
func (e *Entry) Value() any { return e.table.values[e.key] }
|
||||
|
||||
// Inline reports whether the value was written as an inline table, `{…}`.
|
||||
func (e *Entry) Inline() bool { return e.inline }
|
||||
|
||||
// Table returns the table the value is, or nil when it is not a table.
|
||||
func (e *Entry) Table() *Table { return e.child }
|
||||
|
||||
// Elements returns the tables of an array of tables, or nil when the value is
|
||||
// not one.
|
||||
func (e *Entry) Elements() []*Table { return e.elements }
|
||||
|
||||
// Comments returns the comment lines above the key. Lines carry no leading '#'
|
||||
// and no surrounding space.
|
||||
func (e *Entry) Comments() []string { return e.comments }
|
||||
|
||||
// SetComments replaces those lines. Each line is written back with a "# " in
|
||||
// front of it, so a line should not carry one.
|
||||
func (e *Entry) SetComments(lines []string) { e.comments = lines }
|
||||
|
||||
// Trailing returns the comment on the key's own line, without the '#'.
|
||||
func (e *Entry) Trailing() string { return e.trailing }
|
||||
|
||||
// SetTrailing replaces that comment.
|
||||
func (e *Entry) SetTrailing(line string) { e.trailing = line }
|
||||
@@ -0,0 +1,314 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package interpres
|
||||
|
||||
import (
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// mustEntry returns the entry a table must have, and fails the test when it
|
||||
// does not.
|
||||
func mustEntry(t *testing.T, tbl *Table, key string) *Entry {
|
||||
t.Helper()
|
||||
e, ok := tbl.Get(key)
|
||||
if !ok {
|
||||
t.Fatalf("%q is missing from the table", key)
|
||||
}
|
||||
return e
|
||||
}
|
||||
|
||||
func TestDocumentKeepsKeyOrder(t *testing.T) {
|
||||
doc, err := Parse([]byte(`
|
||||
b = 1
|
||||
a = 2
|
||||
inline = {y = 1, x = 2}
|
||||
|
||||
[table]
|
||||
z = 3
|
||||
m = 4
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
|
||||
// The root's keys come back in written order, not sorted.
|
||||
if got := doc.Root().Keys(); !slices.Equal(got, []string{"b", "a", "inline", "table"}) {
|
||||
t.Errorf("root keys = %v, want [b a inline table]", got)
|
||||
}
|
||||
|
||||
// So do the keys of an inline table, which the map shape loses.
|
||||
inline, ok := doc.Root().Get("inline")
|
||||
if !ok {
|
||||
t.Fatal("inline is missing from the root")
|
||||
}
|
||||
if !inline.Inline() {
|
||||
t.Error("inline is not marked inline")
|
||||
}
|
||||
if got := inline.Table().Keys(); !slices.Equal(got, []string{"y", "x"}) {
|
||||
t.Errorf("inline keys = %v, want [y x]", got)
|
||||
}
|
||||
|
||||
// And the keys of a table written under a header, which is not inline.
|
||||
tbl, ok := doc.Root().Get("table")
|
||||
if !ok {
|
||||
t.Fatal("table is missing from the root")
|
||||
}
|
||||
if tbl.Inline() {
|
||||
t.Error("table is marked inline")
|
||||
}
|
||||
if got := tbl.Table().Keys(); !slices.Equal(got, []string{"z", "m"}) {
|
||||
t.Errorf("table keys = %v, want [z m]", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentValuesAreTheTree(t *testing.T) {
|
||||
doc, err := Parse([]byte("n = 7\ns = \"x\"\n\n[t]\nk = true\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if got := mustEntry(t, doc.Root(), "n").Value(); got != int64(7) {
|
||||
t.Errorf("n = %#v, want int64(7)", got)
|
||||
}
|
||||
tbl := mustEntry(t, doc.Root(), "t").Table()
|
||||
if got := mustEntry(t, tbl, "k").Value(); got != true {
|
||||
t.Errorf("t.k = %#v, want true", got)
|
||||
}
|
||||
// Map gives the tree ParseMap would have returned, the same values.
|
||||
tree := doc.Map()
|
||||
if tree["n"] != int64(7) || tree["s"] != "x" {
|
||||
t.Errorf("Map = %#v", tree)
|
||||
}
|
||||
if tree["t"].(map[string]any)["k"] != true {
|
||||
t.Errorf("Map[t] = %#v", tree["t"])
|
||||
}
|
||||
if tbl.Values()["k"] != true {
|
||||
t.Errorf("t.Values() = %#v", tbl.Values())
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentComments(t *testing.T) {
|
||||
doc, err := Parse([]byte(`# above b
|
||||
b = 1 # trailing b
|
||||
|
||||
# above the table
|
||||
[table] # trailing table
|
||||
# above m
|
||||
m = 2
|
||||
|
||||
# footer
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
|
||||
b, ok := doc.Root().Get("b")
|
||||
if !ok {
|
||||
t.Fatal("b is missing")
|
||||
}
|
||||
if got := b.Comments(); !slices.Equal(got, []string{"above b"}) {
|
||||
t.Errorf("b comments = %q, want [above b]", got)
|
||||
}
|
||||
if got := b.Trailing(); got != "trailing b" {
|
||||
t.Errorf("b trailing = %q, want \"trailing b\"", got)
|
||||
}
|
||||
|
||||
tbl, ok := doc.Root().Get("table")
|
||||
if !ok {
|
||||
t.Fatal("table is missing")
|
||||
}
|
||||
// A [header] line introduces the table, so the comments around it belong
|
||||
// to the table node; the entry that names it stays bare.
|
||||
if got := tbl.Table().Comments(); !slices.Equal(got, []string{"above the table"}) {
|
||||
t.Errorf("table comments = %q, want [above the table]", got)
|
||||
}
|
||||
if got := tbl.Table().Trailing(); got != "trailing table" {
|
||||
t.Errorf("table trailing = %q, want \"trailing table\"", got)
|
||||
}
|
||||
if got := tbl.Comments(); got != nil {
|
||||
t.Errorf("entry comments = %q, want none", got)
|
||||
}
|
||||
m, ok := tbl.Table().Get("m")
|
||||
if !ok {
|
||||
t.Fatal("table.m is missing")
|
||||
}
|
||||
if got := m.Comments(); !slices.Equal(got, []string{"above m"}) {
|
||||
t.Errorf("m comments = %q, want [above m]", got)
|
||||
}
|
||||
|
||||
if got := doc.Footer(); !slices.Equal(got, []string{"footer"}) {
|
||||
t.Errorf("footer = %q, want [footer]", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentCommentsAreWritable(t *testing.T) {
|
||||
doc, err := Parse([]byte("# above\nk = 1\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
entry, ok := doc.Root().Get("k")
|
||||
if !ok {
|
||||
t.Fatal("k is missing")
|
||||
}
|
||||
entry.SetComments([]string{"first", "second"})
|
||||
entry.SetTrailing("beside")
|
||||
if got := entry.Comments(); !slices.Equal(got, []string{"first", "second"}) {
|
||||
t.Errorf("comments = %q", got)
|
||||
}
|
||||
if got := entry.Trailing(); got != "beside" {
|
||||
t.Errorf("trailing = %q", got)
|
||||
}
|
||||
|
||||
tbl := doc.Root()
|
||||
tbl.SetComments([]string{"above the root"})
|
||||
if got := tbl.Comments(); !slices.Equal(got, []string{"above the root"}) {
|
||||
t.Errorf("root comments = %q", got)
|
||||
}
|
||||
doc.SetFooter([]string{"end"})
|
||||
if got := doc.Footer(); !slices.Equal(got, []string{"end"}) {
|
||||
t.Errorf("footer = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentArrayOfTables(t *testing.T) {
|
||||
doc, err := Parse([]byte(`# first element
|
||||
[[item]]
|
||||
a = 1
|
||||
|
||||
[[item]]
|
||||
b = 2 # beside b
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
entry, ok := doc.Root().Get("item")
|
||||
if !ok {
|
||||
t.Fatal("item is missing")
|
||||
}
|
||||
elems := entry.Elements()
|
||||
if len(elems) != 2 {
|
||||
t.Fatalf("elements = %d, want 2", len(elems))
|
||||
}
|
||||
if got := elems[0].Keys(); !slices.Equal(got, []string{"a"}) {
|
||||
t.Errorf("first element keys = %v, want [a]", got)
|
||||
}
|
||||
if got := elems[0].Comments(); !slices.Equal(got, []string{"first element"}) {
|
||||
t.Errorf("first element comments = %q", got)
|
||||
}
|
||||
if got := elems[1].Keys(); !slices.Equal(got, []string{"b"}) {
|
||||
t.Errorf("second element keys = %v, want [b]", got)
|
||||
}
|
||||
if got := mustEntry(t, elems[1], "b").Trailing(); got != "beside b" {
|
||||
t.Errorf("b trailing = %q, want \"beside b\"", got)
|
||||
}
|
||||
// The value keeps the map shape the decoder reads.
|
||||
if _, ok := entry.Value().([]map[string]any); !ok {
|
||||
t.Errorf("item value = %#v, want []map[string]any", entry.Value())
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentDottedKeysAndValueArrays(t *testing.T) {
|
||||
doc, err := Parse([]byte("a.b.c = 1\narr = [1, {x = 1}]\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
|
||||
// A dotted key builds tables, and they are not inline ones.
|
||||
a, ok := doc.Root().Get("a")
|
||||
if !ok {
|
||||
t.Fatal("a is missing")
|
||||
}
|
||||
if a.Inline() {
|
||||
t.Error("a is marked inline")
|
||||
}
|
||||
b, ok := a.Table().Get("b")
|
||||
if !ok {
|
||||
t.Fatal("a.b is missing")
|
||||
}
|
||||
if b.Inline() {
|
||||
t.Error("a.b is marked inline")
|
||||
}
|
||||
if got := b.Table().Keys(); !slices.Equal(got, []string{"c"}) {
|
||||
t.Errorf("a.b keys = %v, want [c]", got)
|
||||
}
|
||||
|
||||
// An inline table inside a value array keeps its node in the elements
|
||||
// slice; the scalar before it has none.
|
||||
arr, ok := doc.Root().Get("arr")
|
||||
if !ok {
|
||||
t.Fatal("arr is missing")
|
||||
}
|
||||
elems := arr.Elements()
|
||||
if len(elems) != 2 {
|
||||
t.Fatalf("elements = %d, want 2", len(elems))
|
||||
}
|
||||
if elems[0] != nil {
|
||||
t.Errorf("elements[0] = %#v, want nil for a scalar", elems[0])
|
||||
}
|
||||
if got := elems[1].Keys(); !slices.Equal(got, []string{"x"}) {
|
||||
t.Errorf("elements[1] keys = %v, want [x]", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDocumentWithoutStatements(t *testing.T) {
|
||||
doc, err := Parse([]byte("# only a comment\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if got := doc.Root().Keys(); len(got) != 0 {
|
||||
t.Errorf("keys = %v, want none", got)
|
||||
}
|
||||
if got := doc.Footer(); !slices.Equal(got, []string{"only a comment"}) {
|
||||
t.Errorf("footer = %q, want [only a comment]", got)
|
||||
}
|
||||
|
||||
empty, err := Parse(nil)
|
||||
if err != nil {
|
||||
t.Fatalf("parse of nothing: %v", err)
|
||||
}
|
||||
if len(empty.Root().Keys()) != 0 || len(empty.Footer()) != 0 {
|
||||
t.Errorf("empty document = %v / %q", empty.Root().Keys(), empty.Footer())
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseMapIsTheValueTree(t *testing.T) {
|
||||
// ParseMap is the path that does not build a document, and it gives the
|
||||
// tree the decoder reads.
|
||||
tree, err := ParseMap([]byte("a = 1\n\n[t]\nb = \"x\"\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if tree["a"] != int64(1) {
|
||||
t.Errorf("a = %#v", tree["a"])
|
||||
}
|
||||
if tree["t"].(map[string]any)["b"] != "x" {
|
||||
t.Errorf("t = %#v", tree["t"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestMarshalRejectsDocument(t *testing.T) {
|
||||
// A Document is not a value to marshal: its order and comments would be
|
||||
// dropped, and a struct walk would silently write nothing at all.
|
||||
doc, err := Parse([]byte("a = 1\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if _, err := Marshal(doc); err == nil {
|
||||
t.Fatal("expected an error for a Document")
|
||||
} else if !strings.Contains(err.Error(), "Map()") {
|
||||
t.Errorf("err = %v, want it to point at Map()", err)
|
||||
}
|
||||
if _, err := Marshal(*doc); err == nil {
|
||||
t.Fatal("expected an error for a Document value")
|
||||
}
|
||||
// The tree marshals, which is the way through.
|
||||
out, err := Marshal(doc.Map())
|
||||
if err != nil {
|
||||
t.Fatalf("marshal of the tree: %v", err)
|
||||
}
|
||||
if want := "a = 1\n"; string(out) != want {
|
||||
t.Errorf("output = %q, want %q", out, want)
|
||||
}
|
||||
}
|
||||
+952
-14
File diff suppressed because it is too large
Load Diff
@@ -13,7 +13,7 @@ import (
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"sourcedock.dev/petrbalvin/interpres"
|
||||
"sourcedock.dev/petrbalvin/interpres/v2"
|
||||
)
|
||||
|
||||
// document is a small but realistic configuration: it has scalars, a
|
||||
|
||||
+123
@@ -0,0 +1,123 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package interpres
|
||||
|
||||
import (
|
||||
"math"
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// FuzzParse drives the parser with arbitrary input and holds it to the
|
||||
// round-trip invariant: every document Parse accepts must survive its own
|
||||
// re-emission. Marshal of the parsed tree must succeed, the emitted document
|
||||
// must parse again, and the re-parsed tree must equal the original one.
|
||||
func FuzzParse(f *testing.F) {
|
||||
seeds := []string{
|
||||
"",
|
||||
"title = \"interpres\"\n",
|
||||
"[server]\nhost = \"localhost\"\nport = 8080\n\n[server.tls]\nenabled = true\n",
|
||||
"[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n",
|
||||
"inline = { a = 1, b = [2, 3], c = { d = 4 } }\n",
|
||||
"arr = [1, 2.5, \"three\", true, 1979-05-27T07:32:00Z]\n",
|
||||
"mix = [1, {a = 2}, \"x\"]\n",
|
||||
"when = 1979-05-27T07:32:00Z\nlocal = 1979-05-27T07:32:00.999\nd = 1979-05-27\nt = 07:32:00\n",
|
||||
"multi = \"\"\"\nlines\n\"\"\"\nlit = 'literal'\n",
|
||||
"esc = \"\\u0000\\t\\n\\\"\\\\\"\n",
|
||||
"neg = -0.0\nnan = nan\ninf = -inf\nexp = 1e6\n",
|
||||
"\"quoted key\" = 'value'\n'1979-05-27' = 1\na.b.c = { d = \"dotted\" }\n",
|
||||
"hex = 0xFF\noct = 0o755\nbin = 0b1010\nsep = 1_000_000\n",
|
||||
"x = \"unterminated\n",
|
||||
"[a]\n[a]\n",
|
||||
"n = 0x1_0000_0000_0000_0000\n",
|
||||
// TOML 1.1 forms.
|
||||
"t = 13:37\ndt = 1979-05-27T07:32\nodt = 1979-05-27 07:32Z\n",
|
||||
"esc = \"\\e\\x41\\x7f\\x00\"\n",
|
||||
"m = {\n\ta = 1,\n\tb = [1, 2,],\n\tc = { d = 2 },\n} # close\n",
|
||||
}
|
||||
for _, s := range seeds {
|
||||
f.Add([]byte(s))
|
||||
}
|
||||
f.Fuzz(func(t *testing.T, data []byte) {
|
||||
tree, err := ParseMap(data)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
out, err := Marshal(tree)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal of a parsed tree failed: %v\ntree: %#v", err, tree)
|
||||
}
|
||||
re, err := ParseMap(out)
|
||||
if err != nil {
|
||||
t.Fatalf("re-parse of the emitted document failed: %v\ndoc:\n%s", err, out)
|
||||
}
|
||||
if !tomlEqual(tree, re) {
|
||||
t.Fatalf("round-trip changed the tree\ninput: %q\ndoc:\n%s\nwas: %#v\nnow: %#v", data, out, tree, re)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// tomlEqual reports whether two parsed trees are equal as TOML values. It
|
||||
// differs from reflect.DeepEqual where DeepEqual is wrong for this domain:
|
||||
// NaN compares equal to itself, date-times compare by their canonical TOML
|
||||
// rendering so two parses of one document stay equal, and the local variants
|
||||
// compare through their String form, which fully determines the value.
|
||||
func tomlEqual(a, b any) bool {
|
||||
switch av := a.(type) {
|
||||
case nil:
|
||||
return b == nil
|
||||
case float64:
|
||||
bv, ok := b.(float64)
|
||||
return ok && (av == bv || (math.IsNaN(av) && math.IsNaN(bv)))
|
||||
case time.Time:
|
||||
bv, ok := b.(time.Time)
|
||||
return ok && av.Format(time.RFC3339Nano) == bv.Format(time.RFC3339Nano)
|
||||
case LocalDateTime:
|
||||
bv, ok := b.(LocalDateTime)
|
||||
return ok && av.String() == bv.String()
|
||||
case LocalDate:
|
||||
bv, ok := b.(LocalDate)
|
||||
return ok && av.String() == bv.String()
|
||||
case LocalTime:
|
||||
bv, ok := b.(LocalTime)
|
||||
return ok && av.String() == bv.String()
|
||||
case []any:
|
||||
bv, ok := b.([]any)
|
||||
if !ok || len(av) != len(bv) {
|
||||
return false
|
||||
}
|
||||
for i := range av {
|
||||
if !tomlEqual(av[i], bv[i]) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
case []map[string]any:
|
||||
bv, ok := b.([]map[string]any)
|
||||
if !ok || len(av) != len(bv) {
|
||||
return false
|
||||
}
|
||||
for i := range av {
|
||||
if !tomlEqual(av[i], bv[i]) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
case map[string]any:
|
||||
bv, ok := b.(map[string]any)
|
||||
if !ok || len(av) != len(bv) {
|
||||
return false
|
||||
}
|
||||
for k, v := range av {
|
||||
other, ok := bv[k]
|
||||
if !ok || !tomlEqual(v, other) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
default:
|
||||
return reflect.DeepEqual(a, b)
|
||||
}
|
||||
}
|
||||
@@ -1,3 +1,3 @@
|
||||
module sourcedock.dev/petrbalvin/interpres
|
||||
module sourcedock.dev/petrbalvin/interpres/v2
|
||||
|
||||
go 1.27.0
|
||||
go 1.27.1
|
||||
|
||||
+219
-36
@@ -11,9 +11,10 @@
|
||||
//
|
||||
// out, err := interpres.Marshal(cfg)
|
||||
//
|
||||
// or, for an untyped tree:
|
||||
// or, for the document with its key order and comments:
|
||||
//
|
||||
// tree, err := interpres.Parse(data)
|
||||
// doc, err := interpres.Parse(data)
|
||||
// tree := doc.Map()
|
||||
//
|
||||
// A Decoder allows strict decoding that rejects keys without a matching
|
||||
// struct field, mirroring (*json.Decoder).DisallowUnknownFields.
|
||||
@@ -21,8 +22,9 @@ package interpres
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"unicode/utf8"
|
||||
"slices"
|
||||
)
|
||||
|
||||
// A SyntaxError describes a malformed TOML document, including the 1-based
|
||||
@@ -36,29 +38,129 @@ func (e *SyntaxError) Error() string {
|
||||
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
|
||||
}
|
||||
|
||||
// Parse decodes a TOML document into a nested map[string]any.
|
||||
// A DecodeError wraps a decoding failure with the key path at which it
|
||||
// happened. Path lists one segment per level from the document root, the
|
||||
// outermost key first: a key contributes its name and an array element its
|
||||
// bracketed index, so the path of the weight field in the first item reads
|
||||
// ["items", "[0]", "weight"]. The rendered message is unchanged by the type;
|
||||
// read it programmatically with errors.AsType:
|
||||
//
|
||||
// if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
|
||||
// fmt.Println(de.Path, de.Err)
|
||||
// }
|
||||
type DecodeError struct {
|
||||
// Path is the key path from the document root, outermost key first.
|
||||
Path []string
|
||||
// Err is the failure at that path.
|
||||
Err error
|
||||
}
|
||||
|
||||
func (e *DecodeError) Error() string { return e.Path[0] + ": " + e.Err.Error() }
|
||||
|
||||
// Unwrap returns the failure the path points at.
|
||||
func (e *DecodeError) Unwrap() error { return e.Err }
|
||||
|
||||
// newDecodeError wraps err with one path segment. The rest of the path comes
|
||||
// from the DecodeError err already carries, if any: the decoder wraps each
|
||||
// key and index on its way down, so the innermost wrap holds the deepest
|
||||
// segments and each outer wrap prepends one.
|
||||
func newDecodeError(key string, err error) *DecodeError {
|
||||
path := make([]string, 0, 4)
|
||||
path = append(path, key)
|
||||
if de, ok := errors.AsType[*DecodeError](err); ok {
|
||||
path = append(path, de.Path...)
|
||||
}
|
||||
return &DecodeError{Path: path, Err: err}
|
||||
}
|
||||
|
||||
// An EncodeError wraps an encoding failure with the key path of the value
|
||||
// that failed, in the notation of a TOML document: fields join with dots and
|
||||
// an array element carries its bracketed index, so the path of the third
|
||||
// port under server reads "server.ports[2]". The rendered message is
|
||||
// unchanged by the type; read it programmatically with errors.AsType.
|
||||
type EncodeError struct {
|
||||
// Path is the key path of the failing value.
|
||||
Path string
|
||||
// Err is the failure at that path.
|
||||
Err error
|
||||
}
|
||||
|
||||
func (e *EncodeError) Error() string { return "interpres: " + e.Path + ": " + e.Err.Error() }
|
||||
|
||||
// Unwrap returns the failure the path points at.
|
||||
func (e *EncodeError) Unwrap() error { return e.Err }
|
||||
|
||||
// Parse decodes a TOML document into a Document: the values, the order the
|
||||
// keys were written in, whether a table was written inline, and the comments.
|
||||
// ParseMap gives the plain value tree instead.
|
||||
//
|
||||
// Values are mapped to Go types as follows: strings to string, integers to
|
||||
// int64, floats to float64, booleans to bool, date-times to time.Time, arrays
|
||||
// to []any, and tables (including inline tables) to map[string]any.
|
||||
// int64, floats to float64, booleans to bool, offset date-times to
|
||||
// OffsetDateTime, the local date-time kinds to their wrappers, arrays to
|
||||
// []any, and tables (including inline tables) to map[string]any.
|
||||
//
|
||||
// Parse is equivalent to ParseContext with context.Background.
|
||||
func Parse(data []byte) (map[string]any, error) {
|
||||
func Parse(data []byte) (*Document, error) {
|
||||
return ParseContext(context.Background(), data)
|
||||
}
|
||||
|
||||
// ParseContext decodes a TOML document into a nested map[string]any, obeying
|
||||
// ctx. The context is checked between top-level statements so cancellation is
|
||||
// honoured before the parser has done substantial work.
|
||||
func ParseContext(ctx context.Context, data []byte) (map[string]any, error) {
|
||||
// ParseContext decodes a TOML document into a Document, obeying ctx. The
|
||||
// context is checked between top-level statements so cancellation is honoured
|
||||
// before the parser has done substantial work.
|
||||
func ParseContext(ctx context.Context, data []byte) (*Document, error) {
|
||||
_, doc, err := parseWithOptions(ctx, data, parseOptions{}, true)
|
||||
return doc, err
|
||||
}
|
||||
|
||||
// ParseMap decodes a TOML document into a nested map[string]any, the value
|
||||
// tree without the order and the comments a Document carries. It is the shape
|
||||
// this package parsed into before [Document] existed.
|
||||
//
|
||||
// ParseMap is equivalent to ParseMapContext with context.Background.
|
||||
func ParseMap(data []byte) (map[string]any, error) {
|
||||
return ParseMapContext(context.Background(), data)
|
||||
}
|
||||
|
||||
// ParseMapContext is the cancellable variant of ParseMap.
|
||||
func ParseMapContext(ctx context.Context, data []byte) (map[string]any, error) {
|
||||
tree, _, err := parseWithOptions(ctx, data, parseOptions{}, false)
|
||||
return tree, err
|
||||
}
|
||||
|
||||
// parseOptions bound the work one parse may do. A zero field takes the
|
||||
// default.
|
||||
type parseOptions struct {
|
||||
maxDepth int
|
||||
maxInputSize int
|
||||
}
|
||||
|
||||
// parseWithOptions parses data, building the node tree of a Document when
|
||||
// wantDoc asks for it, and returns both the value tree and that document.
|
||||
func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantDoc bool) (map[string]any, *Document, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return nil, err
|
||||
return nil, nil, err
|
||||
}
|
||||
if !utf8.Valid(data) {
|
||||
return nil, &SyntaxError{Line: 1, Msg: "input is not valid UTF-8"}
|
||||
if opts.maxInputSize > 0 && len(data) > opts.maxInputSize {
|
||||
return nil, nil, fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), opts.maxInputSize)
|
||||
}
|
||||
p := &parser{src: []rune(string(data)), line: 1, ctx: ctx}
|
||||
return p.parse()
|
||||
// UTF-8 validity is not checked in a pass of its own: the scanner
|
||||
// validates the multi-byte sequences where it meets them, so an invalid
|
||||
// byte is reported on its own line instead of always on line 1.
|
||||
maxDepth := opts.maxDepth
|
||||
if maxDepth <= 0 {
|
||||
maxDepth = maxNestingDepth
|
||||
}
|
||||
// The parser scans data in place; it only reads the buffer, and every
|
||||
// string it stores in the tree is copied out of it.
|
||||
p := &parser{src: data, line: 1, ctx: ctx, maxDepth: maxDepth, wantDoc: wantDoc}
|
||||
tree, err := p.parse()
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if !wantDoc {
|
||||
return tree, nil, nil
|
||||
}
|
||||
return tree, &Document{root: p.doc, footer: p.footer}, nil
|
||||
}
|
||||
|
||||
// Unmarshal parses a TOML document and stores the result in the value pointed
|
||||
@@ -68,6 +170,11 @@ func ParseContext(ctx context.Context, data []byte) (map[string]any, error) {
|
||||
// case-insensitive match on the field name when no tag is present. A tag of
|
||||
// "-" skips the field.
|
||||
//
|
||||
// A destination implementing Unmarshaler receives the parsed value as it is,
|
||||
// a TOML string fills a destination implementing encoding.TextUnmarshaler, and
|
||||
// a time.Duration destination takes a duration literal such as `1h30m` or a
|
||||
// bare integer as its nanosecond count.
|
||||
//
|
||||
// Unmarshal is equivalent to UnmarshalContext with context.Background.
|
||||
func Unmarshal(data []byte, v any) error {
|
||||
return UnmarshalContext(context.Background(), data, v)
|
||||
@@ -75,7 +182,7 @@ func Unmarshal(data []byte, v any) error {
|
||||
|
||||
// UnmarshalContext is the cancellable variant of Unmarshal.
|
||||
func UnmarshalContext(ctx context.Context, data []byte, v any) error {
|
||||
tree, err := ParseContext(ctx, data)
|
||||
tree, err := ParseMapContext(ctx, data)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -83,9 +190,11 @@ func UnmarshalContext(ctx context.Context, data []byte, v any) error {
|
||||
}
|
||||
|
||||
// A Decoder decodes a TOML document into a Go value with configurable
|
||||
// strictness.
|
||||
// strictness and configurable limits on the parse it performs.
|
||||
type Decoder struct {
|
||||
disallowUnknown bool
|
||||
maxDepth int
|
||||
maxInputSize int
|
||||
}
|
||||
|
||||
// NewDecoder returns a Decoder.
|
||||
@@ -98,6 +207,28 @@ func (d *Decoder) DisallowUnknownFields() *Decoder {
|
||||
return d
|
||||
}
|
||||
|
||||
// MaxDepth bounds how deeply arrays and inline tables may nest in a document
|
||||
// this decoder accepts. The parser is a recursive descent, so a document that
|
||||
// nests without bound would exhaust the stack; one that nests deeper than the
|
||||
// limit is rejected with a SyntaxError naming it instead. Use 0 or any
|
||||
// negative value for the default of 10000, which no hand-written document
|
||||
// approaches.
|
||||
func (d *Decoder) MaxDepth(depth int) *Decoder {
|
||||
d.maxDepth = depth
|
||||
return d
|
||||
}
|
||||
|
||||
// MaxInputSize bounds the size of a document this decoder accepts, in bytes; a
|
||||
// larger one is rejected before parsing starts. Use 0 or any negative value for
|
||||
// no limit, which is the default: the caller already holds the bytes, so the
|
||||
// size is a policy the caller sets rather than a protection the library
|
||||
// imposes on its own. Parse and ParseContext take no limit beyond the nesting
|
||||
// default.
|
||||
func (d *Decoder) MaxInputSize(size int) *Decoder {
|
||||
d.maxInputSize = size
|
||||
return d
|
||||
}
|
||||
|
||||
// Decode parses data and stores the result in the value pointed to by v,
|
||||
// honouring the decoder's strictness settings.
|
||||
//
|
||||
@@ -108,7 +239,10 @@ func (d *Decoder) Decode(data []byte, v any) error {
|
||||
|
||||
// DecodeContext is the cancellable variant of Decode.
|
||||
func (d *Decoder) DecodeContext(ctx context.Context, data []byte, v any) error {
|
||||
tree, err := ParseContext(ctx, data)
|
||||
tree, _, err := parseWithOptions(ctx, data, parseOptions{
|
||||
maxDepth: d.maxDepth,
|
||||
maxInputSize: d.maxInputSize,
|
||||
}, false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -122,6 +256,10 @@ func (d *Decoder) DecodeContext(ctx context.Context, data []byte, v any) error {
|
||||
// then encodes as if the returned value had been passed in its place, which
|
||||
// is useful for emitting a Go type as a different TOML shape (for example, a
|
||||
// struct as an inline table or a primitive alias as a richer value).
|
||||
//
|
||||
// MarshalTOML wins over encoding.TextMarshaler when a type implements both.
|
||||
// A type that implements only encoding.TextMarshaler is encoded as a TOML
|
||||
// string holding its text, and needs no method here.
|
||||
type Marshaler interface {
|
||||
MarshalTOML() (any, error)
|
||||
}
|
||||
@@ -129,39 +267,58 @@ type Marshaler interface {
|
||||
// Unmarshaler is the inverse of Marshaler: a type that wants control over
|
||||
// how it is decoded from a TOML value may implement UnmarshalTOML. The data
|
||||
// argument is whatever the parser produced for that key: one of string,
|
||||
// bool, int64, float64, time.Time, LocalDateTime, LocalDate, LocalTime,
|
||||
// []any, or map[string]any. UnmarshalTOML may parse, inspect, or transform
|
||||
// the value however it likes, then store the result by mutating its
|
||||
// receiver through the standard pointer-indirection rules of the reflect
|
||||
// package (i.e. via reflect.Value.Set or by reassigning fields through a
|
||||
// pointer the receiver holds).
|
||||
// bool, int64, float64, OffsetDateTime, LocalDateTime, LocalDate, LocalTime,
|
||||
// []any, or map[string]any. A tree built by hand may carry a plain time.Time
|
||||
// where the parser would put an OffsetDateTime.
|
||||
//
|
||||
// UnmarshalTOML may parse, inspect, or transform the value however it likes,
|
||||
// then store the result by mutating its receiver through the standard
|
||||
// pointer-indirection rules of the reflect package (i.e. via
|
||||
// reflect.Value.Set or by reassigning fields through a pointer the receiver
|
||||
// holds).
|
||||
//
|
||||
// UnmarshalTOML is invoked from (*Decoder).Decode / Unmarshal when the
|
||||
// destination type implements the interface. The decoder does not need to
|
||||
// consult the concrete return value; whatever the receiver stores is kept.
|
||||
//
|
||||
// UnmarshalTOML wins over encoding.TextUnmarshaler when a type implements
|
||||
// both. A type that implements only encoding.TextUnmarshaler is filled from a
|
||||
// TOML string holding its text, and needs no method here.
|
||||
type Unmarshaler interface {
|
||||
UnmarshalTOML(data any) error
|
||||
}
|
||||
|
||||
// Marshal returns the TOML 1.0 encoding of v.
|
||||
// Marshal returns the TOML encoding of v. The output is valid TOML 1.1.
|
||||
//
|
||||
// Marshal traverses v using reflection and applies the following rules:
|
||||
//
|
||||
// - The top-level value must be a struct or a map[string]V. Pointers are
|
||||
// followed; a nil top-level pointer is an error.
|
||||
// - Struct fields are matched by `toml:"name"` tag (case-insensitive
|
||||
// fallback to field name; `-` skips). Anonymous (embedded) fields without
|
||||
// a tag are inlined.
|
||||
// fallback to field name; `-` skips). The tag options `omitzero` (skip
|
||||
// the zero value of the field's type) and `omitempty` (skip an empty
|
||||
// slice, array, or map) drop a field from the output on encode; the
|
||||
// decoder ignores them. Anonymous (embedded) fields without a tag are
|
||||
// inlined.
|
||||
// - Maps use sorted keys for deterministic output.
|
||||
// - Slices and arrays of structs or maps become TOML arrays of tables; a
|
||||
// nil or empty array of tables is omitted (TOML forbids an empty `[[a]]`),
|
||||
// while other empty arrays emit as `key = []`.
|
||||
// - Other slices and arrays become TOML arrays.
|
||||
// - Other slices and arrays become TOML arrays; a table element inside a
|
||||
// value array (for example an inline table in a mixed array) emits as an
|
||||
// inline table.
|
||||
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
|
||||
// (offset date-time), and LocalDateTime/LocalDate/LocalTime (local
|
||||
// variants).
|
||||
// and OffsetDateTime (offset date-time), and LocalDateTime/LocalDate/
|
||||
// LocalTime (local variants). A date-time writes its seconds only when the value carries
|
||||
// them, and drops the trailing zeros of a fractional second.
|
||||
// - A table element of a value array, and a sub-table inlined by
|
||||
// Encoder.InlineTables, is written as an inline table, across lines when it
|
||||
// does not fit one.
|
||||
// - Values implementing Marshaler are encoded by calling MarshalTOML and
|
||||
// using its result.
|
||||
// - Values implementing encoding.TextMarshaler, and not one of the
|
||||
// date-time types, encode as a TOML string holding the text the method
|
||||
// returns. time.Duration is written in its canonical Go form, `1h30m0s`.
|
||||
// - nil pointer fields are omitted.
|
||||
//
|
||||
// Marshal cannot encode cyclic data structures; passing one will loop until
|
||||
@@ -185,14 +342,16 @@ func MarshalContext(ctx context.Context, v any) ([]byte, error) {
|
||||
|
||||
// An Encoder encodes Go values into TOML.
|
||||
//
|
||||
// All options default to behaviour that preserves byte-for-byte compatibility
|
||||
// with previous releases and passes the toml-test compliance suite:
|
||||
// All options default to the behaviour that passes the toml-test compliance
|
||||
// suite in both directions:
|
||||
//
|
||||
// GroupByKind: true (scalars first, then tables, then arrays of tables)
|
||||
// OmitEmptyArrays: false (a nil/empty []string slice emits [] as a value;
|
||||
// a nil/empty []Item struct slice is still skipped)
|
||||
// LiteralMultilineAt: 0 (always emit basic multi-line strings with
|
||||
// escape sequences, never literal ones)
|
||||
// LiteralMultilineAt: 0 (always emit the escaped basic form, never a
|
||||
// literal one)
|
||||
// InlineTablesAt: 0 (always emit a table header, never an inline
|
||||
// table)
|
||||
//
|
||||
// Use the chainable option methods to opt out. The option state is private;
|
||||
// callers that need the underlying knobs reach for the methods rather than
|
||||
@@ -201,6 +360,7 @@ type Encoder struct {
|
||||
groupByKind bool // default true; set via (*Encoder).GroupByKind
|
||||
omitEmptyArrays bool // default false; set via (*Encoder).OmitEmptyArrays
|
||||
literalMultilineAt int // default 0; set via (*Encoder).UseLiteralMultiline
|
||||
inlineTablesAt int // default 0; set via (*Encoder).InlineTables
|
||||
}
|
||||
|
||||
// NewEncoder returns an Encoder with default options.
|
||||
@@ -233,6 +393,24 @@ func (e *Encoder) UseLiteralMultiline(threshold int) *Encoder {
|
||||
return e
|
||||
}
|
||||
|
||||
// InlineTables sets the size limit, in bytes of the single-line rendering, at
|
||||
// which a sub-table is written as an inline table instead of a table header,
|
||||
// which makes a document of small tables shorter. Use 0 or any negative value
|
||||
// to disable (always emit a header).
|
||||
//
|
||||
// A sub-table is inlined only when doing so keeps every value's type: an array
|
||||
// of tables keeps its header form, because its inline form would re-parse as a
|
||||
// value array. An inlined table that does not fit the line is written across
|
||||
// lines, which TOML 1.1 allows.
|
||||
//
|
||||
// With GroupByKind(false) the layout is already for presentation only, and an
|
||||
// inlined table follows the same rule as any other value line: it lands in the
|
||||
// section of the header that precedes it.
|
||||
func (e *Encoder) InlineTables(threshold int) *Encoder {
|
||||
e.inlineTablesAt = threshold
|
||||
return e
|
||||
}
|
||||
|
||||
// Marshal encodes v to TOML bytes. It is equivalent to calling Marshal with v.
|
||||
//
|
||||
// Marshal is equivalent to MarshalContext with context.Background.
|
||||
@@ -246,7 +424,12 @@ func (e *Encoder) MarshalContext(ctx context.Context, v any) ([]byte, error) {
|
||||
enc.ctx = ctx
|
||||
enc.opts = *e
|
||||
if err := enc.encode(v); err != nil {
|
||||
enc.release()
|
||||
return nil, err
|
||||
}
|
||||
return enc.bytes(), nil
|
||||
// The output leaves the pooled buffer as a copy, so the next Marshal
|
||||
// reuses the buffer without touching what the caller holds.
|
||||
out := slices.Clone(enc.buf.Bytes())
|
||||
enc.release()
|
||||
return out, nil
|
||||
}
|
||||
|
||||
+244
-29
@@ -4,13 +4,15 @@
|
||||
package interpres
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"math"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestParseScalars(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
title = "interpres"
|
||||
count = 42
|
||||
ratio = 3.14
|
||||
@@ -48,7 +50,7 @@ expv = 1e3
|
||||
}
|
||||
|
||||
func TestParseInfNan(t *testing.T) {
|
||||
tree, err := Parse([]byte("pos = inf\nneg = -inf\nbad = nan\n"))
|
||||
tree, err := ParseMap([]byte("pos = inf\nneg = -inf\nbad = nan\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
@@ -64,7 +66,7 @@ func TestParseInfNan(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestParseStrings(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
basic = "a\tb\nc"
|
||||
literal = 'C:\path\no\escape'
|
||||
quote = "say \"hi\""
|
||||
@@ -88,7 +90,7 @@ unicode = "\u00e9"
|
||||
}
|
||||
|
||||
func TestParseMultilineString(t *testing.T) {
|
||||
tree, err := Parse([]byte("text = \"\"\"\nfirst\nsecond\"\"\"\n"))
|
||||
tree, err := ParseMap([]byte("text = \"\"\"\nfirst\nsecond\"\"\"\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
@@ -98,7 +100,7 @@ func TestParseMultilineString(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestParseMultilineLineEndingBackslash(t *testing.T) {
|
||||
tree, err := Parse([]byte("text = \"\"\"\\\n one \\\n two\"\"\"\n"))
|
||||
tree, err := ParseMap([]byte("text = \"\"\"\\\n one \\\n two\"\"\"\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
@@ -108,7 +110,7 @@ func TestParseMultilineLineEndingBackslash(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestParseTablesAndDottedKeys(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
owner.name = "Petr"
|
||||
|
||||
[server]
|
||||
@@ -136,7 +138,7 @@ enabled = true
|
||||
}
|
||||
|
||||
func TestParseArrayOfTables(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
[[forms]]
|
||||
name = "contact"
|
||||
|
||||
@@ -156,7 +158,7 @@ name = "feedback"
|
||||
}
|
||||
|
||||
func TestParseArraysAndInlineTables(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
ports = [80, 443]
|
||||
mixed = [
|
||||
"a",
|
||||
@@ -182,7 +184,7 @@ point = { x = 1, y = 2 }
|
||||
}
|
||||
|
||||
func TestParseDateTime(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
offset = 1979-05-27T07:32:00Z
|
||||
local = 1979-05-27T07:32:00
|
||||
day = 1979-05-27
|
||||
@@ -191,7 +193,7 @@ clock = 07:32:00
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if off, ok := tree["offset"].(time.Time); !ok || off.Year() != 1979 || off.Hour() != 7 {
|
||||
if off, ok := tree["offset"].(OffsetDateTime); !ok || off.Year() != 1979 || off.Hour() != 7 {
|
||||
t.Errorf("offset = %#v (%T)", tree["offset"], tree["offset"])
|
||||
}
|
||||
if ldt, ok := tree["local"].(LocalDateTime); !ok || ldt.Year() != 1979 || ldt.Hour() != 7 {
|
||||
@@ -206,15 +208,15 @@ clock = 07:32:00
|
||||
}
|
||||
|
||||
func TestDateTimeFormats(t *testing.T) {
|
||||
tree, err := Parse([]byte("a = 1987-07-05 17:45:00Z\nb = 1987-07-05t17:45:00z\nc = 1977-12-21T10:32:00.555\n"))
|
||||
tree, err := ParseMap([]byte("a = 1987-07-05 17:45:00Z\nb = 1987-07-05t17:45:00z\nc = 1977-12-21T10:32:00.555\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if _, ok := tree["a"].(time.Time); !ok {
|
||||
t.Errorf("a is %T, want time.Time", tree["a"])
|
||||
if _, ok := tree["a"].(OffsetDateTime); !ok {
|
||||
t.Errorf("a is %T, want OffsetDateTime", tree["a"])
|
||||
}
|
||||
if _, ok := tree["b"].(time.Time); !ok {
|
||||
t.Errorf("b is %T, want time.Time", tree["b"])
|
||||
if _, ok := tree["b"].(OffsetDateTime); !ok {
|
||||
t.Errorf("b is %T, want OffsetDateTime", tree["b"])
|
||||
}
|
||||
if _, ok := tree["c"].(LocalDateTime); !ok {
|
||||
t.Errorf("c is %T, want LocalDateTime", tree["c"])
|
||||
@@ -316,6 +318,26 @@ func TestDisallowUnknownFields(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestDisallowUnknownFieldsReportsSmallestKey(t *testing.T) {
|
||||
// Map iteration order is random, so the reported key must be chosen
|
||||
// deterministically: the smallest unknown key, whichever order the map
|
||||
// iterates in.
|
||||
type C struct {
|
||||
Known string `toml:"known"`
|
||||
}
|
||||
data := []byte("known = \"x\"\nzeta = 1\nalpha = 2\nmu = 3\n")
|
||||
for range 20 {
|
||||
var c C
|
||||
err := NewDecoder().DisallowUnknownFields().Decode(data, &c)
|
||||
if err == nil {
|
||||
t.Fatal("expected error for unknown fields")
|
||||
}
|
||||
if !strings.Contains(err.Error(), `unknown field "alpha"`) {
|
||||
t.Fatalf("err = %v, want the smallest unknown key alpha", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSkippedFieldTag(t *testing.T) {
|
||||
type C struct {
|
||||
Keep string `toml:"keep"`
|
||||
@@ -331,7 +353,7 @@ func TestSkippedFieldTag(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestSyntaxErrorReportsLine(t *testing.T) {
|
||||
_, err := Parse([]byte("a = 1\nb = \nc = 3\n"))
|
||||
_, err := ParseMap([]byte("a = 1\nb = \nc = 3\n"))
|
||||
if err == nil {
|
||||
t.Fatal("expected a syntax error")
|
||||
}
|
||||
@@ -345,7 +367,7 @@ func TestSyntaxErrorReportsLine(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestComments(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
# a leading comment
|
||||
key = "value" # trailing comment
|
||||
# another
|
||||
@@ -359,7 +381,7 @@ key = "value" # trailing comment
|
||||
}
|
||||
|
||||
func TestDuplicateKeyRejected(t *testing.T) {
|
||||
_, err := Parse([]byte("a = 1\na = 2\n"))
|
||||
_, err := ParseMap([]byte("a = 1\na = 2\n"))
|
||||
if err == nil {
|
||||
t.Fatal("expected duplicate key error")
|
||||
}
|
||||
@@ -370,15 +392,44 @@ func TestRejectsInvalidNumbers(t *testing.T) {
|
||||
"01", "-01", "00",
|
||||
"1__0", "_1", "1_", "0x_1", "1_.0",
|
||||
"1.", ".5", "1.2.3", "1.e2",
|
||||
"1e", "1e+", "1e-", "0.0E", "0.0e", "1.5e+",
|
||||
"0x", "0o", "0b", "0b2", "0o8", "0xG",
|
||||
"+0x1",
|
||||
} {
|
||||
if _, err := Parse([]byte("v = " + tok + "\n")); err == nil {
|
||||
if _, err := ParseMap([]byte("v = " + tok + "\n")); err == nil {
|
||||
t.Errorf("%q: expected an error, got none", tok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseRejectsOffsetOutOfRange(t *testing.T) {
|
||||
for _, tok := range []string{
|
||||
"1979-05-27T07:32:00+00:60",
|
||||
"1979-05-27T07:32:00-00:99",
|
||||
"1979-05-27T07:32:00+24:00",
|
||||
"1979-05-27T07:32:00+99:99",
|
||||
} {
|
||||
if _, err := ParseMap([]byte("v = " + tok + "\n")); err == nil {
|
||||
t.Errorf("%q: expected an error, got none", tok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAcceptsOffsetBounds(t *testing.T) {
|
||||
tree, err := ParseMap([]byte("a = 1979-05-27T07:32:00+23:59\nb = 1979-05-27T07:32:00-23:59\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
a := tree["a"].(OffsetDateTime)
|
||||
if _, offset := a.Zone(); offset != 23*3600+59*60 {
|
||||
t.Fatalf("a offset = %d, want %d", offset, 23*3600+59*60)
|
||||
}
|
||||
b := tree["b"].(OffsetDateTime)
|
||||
if _, offset := b.Zone(); offset != -(23*3600 + 59*60) {
|
||||
t.Fatalf("b offset = %d", offset)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAcceptsNumberEdgeCases(t *testing.T) {
|
||||
cases := map[string]any{
|
||||
"0": int64(0),
|
||||
@@ -392,10 +443,14 @@ func TestAcceptsNumberEdgeCases(t *testing.T) {
|
||||
"3.14": 3.14,
|
||||
"6.022e23": 6.022e23,
|
||||
"1e10": 1e10,
|
||||
"1e0": 1.0,
|
||||
"1e06": 1e6,
|
||||
"0e00": 0.0,
|
||||
"2E-3": 2e-3,
|
||||
"-2.5E-3": -2.5e-3,
|
||||
}
|
||||
for tok, want := range cases {
|
||||
tree, err := Parse([]byte("v = " + tok + "\n"))
|
||||
tree, err := ParseMap([]byte("v = " + tok + "\n"))
|
||||
if err != nil {
|
||||
t.Errorf("%q: %v", tok, err)
|
||||
continue
|
||||
@@ -407,14 +462,14 @@ func TestAcceptsNumberEdgeCases(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestRejectsTableRedefinition(t *testing.T) {
|
||||
_, err := Parse([]byte("[a]\nx = 1\n\n[a]\ny = 2\n"))
|
||||
_, err := ParseMap([]byte("[a]\nx = 1\n\n[a]\ny = 2\n"))
|
||||
if err == nil {
|
||||
t.Fatal("expected a table-redefinition error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAllowsImplicitThenExplicitTable(t *testing.T) {
|
||||
tree, err := Parse([]byte("[a.b]\nx = 1\n\n[a]\ny = 2\n"))
|
||||
tree, err := ParseMap([]byte("[a.b]\nx = 1\n\n[a]\ny = 2\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
@@ -428,13 +483,13 @@ func TestAllowsImplicitThenExplicitTable(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestRejectsControlCharInString(t *testing.T) {
|
||||
if _, err := Parse([]byte("v = \"a\x01b\"\n")); err == nil {
|
||||
if _, err := ParseMap([]byte("v = \"a\x01b\"\n")); err == nil {
|
||||
t.Fatal("expected a control-character error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAllowsEscapedControlChar(t *testing.T) {
|
||||
tree, err := Parse([]byte(`v = "\u0000"`))
|
||||
tree, err := ParseMap([]byte(`v = "\u0000"`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
@@ -444,7 +499,7 @@ func TestAllowsEscapedControlChar(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestMultilineQuotesAtDelimiter(t *testing.T) {
|
||||
tree, err := Parse([]byte("a = '''''two quotes'''''\n"))
|
||||
tree, err := ParseMap([]byte("a = '''''two quotes'''''\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
@@ -458,14 +513,47 @@ func TestRejectsInlineTableExtension(t *testing.T) {
|
||||
"by header": "a = { b = 1 }\n[a.c]\nx = 2\n",
|
||||
"by dotted key": "a = { b = 1 }\na.c = 2\n",
|
||||
"header over it": "a = { b = 1 }\n[a]\nx = 2\n",
|
||||
// The frozen check must cover the intermediate steps of an array-of-tables
|
||||
// header, not only the leaf: [[a.b.c]] walks through a and a.b.
|
||||
"by nested array header": "a = { b = {} }\n[[a.b.c]]\nx = 2\n",
|
||||
}
|
||||
for name, doc := range cases {
|
||||
if _, err := Parse([]byte(doc)); err == nil {
|
||||
if _, err := ParseMap([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an inline-table extension error", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A new element of an array of tables starts a fresh scope: sub-table headers,
|
||||
// nested arrays of tables, and dotted-key paths recorded for the previous
|
||||
// element must not block the same paths in the next one.
|
||||
func TestArrayOfTablesFreshScopePerElement(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"nested array of tables": "[[a]]\n[[a.b]]\nx = 1\n[[a]]\n[a.b]\ny = 2\n",
|
||||
"dotted key": "[[a]]\nb.c = 1\n[[a]]\n[a.b]\nd = 2\n",
|
||||
}
|
||||
for name, doc := range cases {
|
||||
tree, err := ParseMap([]byte(doc))
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", name, err)
|
||||
continue
|
||||
}
|
||||
elements := tree["a"].([]map[string]any)
|
||||
if len(elements) != 2 {
|
||||
t.Errorf("%s: len(a) = %d, want 2", name, len(elements))
|
||||
}
|
||||
}
|
||||
// Within one element the redefinition rules keep applying.
|
||||
for name, doc := range map[string]string{
|
||||
"header over dotted in one element": "[[a]]\nb.c = 1\n[a.b]\nd = 2\n",
|
||||
"table over nested array": "[[a]]\n[[a.b]]\n[a.b]\nx = 1\n",
|
||||
} {
|
||||
if _, err := ParseMap([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRejectsSpecInvalid(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"single-digit hour": "a = 2023-10-01T1:32:00Z\n",
|
||||
@@ -474,17 +562,18 @@ func TestRejectsSpecInvalid(t *testing.T) {
|
||||
"dotted over header": "[a.b]\nx = 1\n[a]\nb.y = 2\n",
|
||||
"table over array": "[[t]]\n[t]\n",
|
||||
"truncated datetime": "a = 2026-01-02T\n",
|
||||
"datetime no seconds": "a = 2026-01-02T07:32\n",
|
||||
// "datetime no seconds" moved to the acceptance tests: TOML 1.1
|
||||
// makes the seconds optional.
|
||||
}
|
||||
for name, doc := range cases {
|
||||
if _, err := Parse([]byte(doc)); err == nil {
|
||||
if _, err := ParseMap([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an error", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestArrayOfTablesPerElementSubtable(t *testing.T) {
|
||||
tree, err := Parse([]byte(`
|
||||
tree, err := ParseMap([]byte(`
|
||||
[[forms]]
|
||||
name = "a"
|
||||
|
||||
@@ -511,3 +600,129 @@ host = "h2"
|
||||
t.Errorf("forms[1].smtp.host = %v", h)
|
||||
}
|
||||
}
|
||||
|
||||
// --- TOML 1.1 --------------------------------------------------------------
|
||||
|
||||
func TestParseAcceptsNoSecondsDatetimes(t *testing.T) {
|
||||
tree, err := ParseMap([]byte(`t = 13:37
|
||||
dt = 1979-05-27T07:32
|
||||
odt1 = 1979-05-27 07:32Z
|
||||
odt2 = 1979-05-27 07:32-07:00
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
// A value written without seconds comes back without them: the seconds are
|
||||
// only written when the value carries them.
|
||||
if got := tree["t"].(LocalTime).String(); got != "13:37" {
|
||||
t.Errorf("t = %q, want %q", got, "13:37")
|
||||
}
|
||||
if got := tree["dt"].(LocalDateTime).String(); got != "1979-05-27T07:32" {
|
||||
t.Errorf("dt = %q, want %q", got, "1979-05-27T07:32")
|
||||
}
|
||||
if got := tree["odt1"].(OffsetDateTime).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00Z" {
|
||||
t.Errorf("odt1 = %q", got)
|
||||
}
|
||||
if got := tree["odt2"].(OffsetDateTime).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00-07:00" {
|
||||
t.Errorf("odt2 = %q", got)
|
||||
}
|
||||
// The fraction still requires the seconds it belongs to.
|
||||
if _, err := ParseMap([]byte("a = 07:32.5\n")); err == nil {
|
||||
t.Error("07:32.5: expected an error, got none")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAcceptsEscapeAndHexEscapes(t *testing.T) {
|
||||
tree, err := ParseMap([]byte(`esc = "\e"
|
||||
hex = "\x20\x7f\xf8"
|
||||
nul = "\x00"
|
||||
multi = """\x68\x65"""
|
||||
lit = '\x20'
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if got := tree["esc"].(string); got != "\x1b" {
|
||||
t.Errorf("esc = %q, want the escape character", got)
|
||||
}
|
||||
if got := tree["hex"].(string); got != " \x7f\u00f8" {
|
||||
t.Errorf("hex = %q", got)
|
||||
}
|
||||
if got := tree["nul"].(string); got != "\x00" {
|
||||
t.Errorf("nul = %q", got)
|
||||
}
|
||||
if got := tree["multi"].(string); got != "he" {
|
||||
t.Errorf("multi = %q", got)
|
||||
}
|
||||
// A literal string carries the sequence verbatim.
|
||||
if got := tree["lit"].(string); got != `\x20` {
|
||||
t.Errorf("lit = %q, want the verbatim sequence", got)
|
||||
}
|
||||
// Two digits exactly; a short or non-hex escape is an error.
|
||||
for _, doc := range []string{`a = "\x4"`, `a = "\x"`, `a = "\xgg"`} {
|
||||
if _, err := ParseMap([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", doc)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAcceptsMultilineInlineTables(t *testing.T) {
|
||||
tree, err := ParseMap([]byte("tbl = {\n\thello = \"world\",\n\tarr = [1,\n\t\t2,\n\t],\n\tsub = {\n\t\tk = 1,\n\t},\n\tbare = 2}\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
tbl := tree["tbl"].(map[string]any)
|
||||
if tbl["hello"] != "world" || tbl["bare"] != int64(2) {
|
||||
t.Fatalf("tbl = %#v", tbl)
|
||||
}
|
||||
if arr := tbl["arr"].([]any); len(arr) != 2 {
|
||||
t.Errorf("arr = %#v", tbl["arr"])
|
||||
}
|
||||
if sub := tbl["sub"].(map[string]any); sub["k"] != int64(1) {
|
||||
t.Errorf("sub = %#v", tbl["sub"])
|
||||
}
|
||||
// Comments inside the table, and a trailing comma at both depths.
|
||||
tree, err = ParseMap([]byte("m = { # one\n\t# two\n\ta = 1, # three\n\t# four\n}\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse with comments: %v", err)
|
||||
}
|
||||
if m := tree["m"].(map[string]any); m["a"] != int64(1) {
|
||||
t.Errorf("m = %#v", m)
|
||||
}
|
||||
// The old single-line shapes keep working, with and without the comma.
|
||||
if _, err := ParseMap([]byte("a = { b = 1, c = 2 }\n")); err != nil {
|
||||
t.Errorf("single line: %v", err)
|
||||
}
|
||||
// Still rejected: two commas, a missing value, and an unclosed table.
|
||||
for name, doc := range map[string]string{
|
||||
"double comma": "a = { b = 1,, c = 2 }\n",
|
||||
"missing value": "a = {\n\tb =\n}\n",
|
||||
"unterminated": "a = { b = 1,\n",
|
||||
} {
|
||||
if _, err := ParseMap([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseNestingLimit(t *testing.T) {
|
||||
// The parser is a recursive descent, so a document that nests without bound
|
||||
// is rejected instead of exhausting the stack.
|
||||
deep := func(n int) []byte {
|
||||
return []byte("v = " + strings.Repeat("[", n) + strings.Repeat("]", n) + "\n")
|
||||
}
|
||||
if _, err := ParseMap(deep(100)); err != nil {
|
||||
t.Fatalf("a document well inside the limit: %v", err)
|
||||
}
|
||||
_, err := ParseMap(deep(maxNestingDepth + 1))
|
||||
if err == nil {
|
||||
t.Fatal("expected a nesting error")
|
||||
}
|
||||
var se *SyntaxError
|
||||
if !errors.As(err, &se) {
|
||||
t.Fatalf("expected a *SyntaxError, got %T: %v", err, err)
|
||||
}
|
||||
if !strings.Contains(se.Msg, "nesting") {
|
||||
t.Errorf("Msg = %q, want it to name the nesting limit", se.Msg)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -92,9 +92,9 @@ run:
|
||||
dev:
|
||||
go run -buildvcs=true {{package}}
|
||||
|
||||
# Runs the official toml-test compliance suite against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0); not standard because no canonical recipe covers a domain compliance suite.
|
||||
# Runs the official toml-test compliance suite in both directions, decoder and encoder, against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
|
||||
toml-test: build
|
||||
toml-test bin/interpres-decode
|
||||
toml-test test -decoder=bin/interpres-decode -encoder='bin/interpres-decode -encode' -toml=1.1
|
||||
|
||||
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
|
||||
coverage-html: test
|
||||
|
||||
@@ -41,6 +41,15 @@ func decodeDecimalInt(tok string) (any, error) {
|
||||
if err := checkNoLeadingZero(digits); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// An unsigned token parses in place; only a sign needs the concatenated
|
||||
// copy, and concatenating an empty sign still allocated.
|
||||
if sign == "" {
|
||||
i, err := strconv.ParseInt(digits, 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("integer %q out of range", tok)
|
||||
}
|
||||
return i, nil
|
||||
}
|
||||
i, err := strconv.ParseInt(sign+digits, 10, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("integer %q out of range", tok)
|
||||
@@ -74,15 +83,16 @@ func decodeFloat(tok string) (any, error) {
|
||||
sign, s := splitSign(tok)
|
||||
|
||||
mantissa, exp := s, ""
|
||||
hasExp := false
|
||||
if i := strings.IndexAny(s, "eE"); i >= 0 {
|
||||
mantissa, exp = s[:i], s[i+1:]
|
||||
mantissa, exp, hasExp = s[:i], s[i+1:], true
|
||||
}
|
||||
|
||||
intPart, frac, hasDot := mantissa, "", false
|
||||
if i := strings.IndexByte(mantissa, '.'); i >= 0 {
|
||||
intPart, frac, hasDot = mantissa[:i], mantissa[i+1:], true
|
||||
}
|
||||
if !hasDot && exp == "" {
|
||||
if !hasDot && !hasExp {
|
||||
return nil, fmt.Errorf("invalid float %q", tok)
|
||||
}
|
||||
|
||||
@@ -93,24 +103,43 @@ func decodeFloat(tok string) (any, error) {
|
||||
if err := checkNoLeadingZero(ip); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
build := sign + ip
|
||||
|
||||
fp := ""
|
||||
if hasDot {
|
||||
fp, err := joinDigits(frac, isDecDigit)
|
||||
if err != nil {
|
||||
if fp, err = joinDigits(frac, isDecDigit); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
// The ABNF requires at least one digit after the exponent marker, so a
|
||||
// trailing e or E is an error even though strconv would accept it. The
|
||||
// digits are a zero-prefixable integer, so leading zeros are fine here
|
||||
// (the corpus holds valid cases such as 1e06 and 0e00).
|
||||
esign, ed := "", ""
|
||||
if hasExp {
|
||||
var digits string
|
||||
esign, digits = splitSign(exp)
|
||||
if ed, err = joinDigits(digits, isDecDigit); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
|
||||
// The checks above validated the token's shape, and every character a
|
||||
// valid token may carry is one strconv.ParseFloat accepts in place, so
|
||||
// only a token with underscores needs the stripped rebuild.
|
||||
if !strings.ContainsRune(tok, '_') {
|
||||
f, err := strconv.ParseFloat(tok, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("invalid float %q", tok)
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
build := sign + ip
|
||||
if hasDot {
|
||||
build += "." + fp
|
||||
}
|
||||
if exp != "" {
|
||||
esign, edigits := splitSign(exp)
|
||||
ed, err := joinDigits(edigits, isDecDigit)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if hasExp {
|
||||
build += "e" + esign + ed
|
||||
}
|
||||
|
||||
f, err := strconv.ParseFloat(build, 64)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("invalid float %q", tok)
|
||||
@@ -120,11 +149,20 @@ func decodeFloat(tok string) (any, error) {
|
||||
|
||||
// joinDigits validates that every rune is a digit (per isDigit) and that each
|
||||
// underscore sits between two digits, returning the digits with underscores
|
||||
// removed.
|
||||
// removed. A token without underscores, the common case, is validated in
|
||||
// place and returned without a copy.
|
||||
func joinDigits(s string, isDigit func(byte) bool) (string, error) {
|
||||
if s == "" {
|
||||
return "", fmt.Errorf("number is missing digits")
|
||||
}
|
||||
if !strings.ContainsRune(s, '_') {
|
||||
for i := range len(s) {
|
||||
if !isDigit(s[i]) {
|
||||
return "", fmt.Errorf("invalid character %q in number", string(s[i]))
|
||||
}
|
||||
}
|
||||
return s, nil
|
||||
}
|
||||
var b strings.Builder
|
||||
for i := range len(s) {
|
||||
c := s[i]
|
||||
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
go test fuzz v1
|
||||
[]byte("0=[{}]")
|
||||
Reference in New Issue
Block a user