8 Commits
Author SHA1 Message Date
petrbalvin 59937749fa chore: prepare release v1.0.0
Test / test (push) Successful in 2m5s
Release / gates (push) Successful in 1m50s
Release / release (push) Successful in 43s
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:48 +02:00
petrbalvin ca3b6c6cba docs: document set and changelog
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
petrbalvin d6258cbbc5 ci: hand-written test, race and release pipelines
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
petrbalvin 3a83dbd432 build: canonical justfile
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
petrbalvin bcc8ce2078 feat: usage example
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
petrbalvin add5ab588b feat: toml-test adapter command
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
petrbalvin 4db19b0ac6 feat: TOML 1.0 parser and encoder
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
petrbalvin 387f5e80da chore: initialise repository
Assisted-by: GLM 5.3 Flash
2026-09-16 22:02:04 +02:00
26 changed files with 297 additions and 2259 deletions
+1
View File
@@ -11,6 +11,7 @@ on:
workflow_dispatch: workflow_dispatch:
env: env:
GOAMD64: v3
# One core: parallelism buys no speed here and costs memory the box does not have. # One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1 GOFLAGS: -p=1
GOMAXPROCS: "2" GOMAXPROCS: "2"
+13 -15
View File
@@ -2,10 +2,9 @@
# #
# A library ships no binaries, so there is no build matrix and no smoke test: the release # A library ships no binaries, so there is no build matrix and no smoke test: the release
# carries the CHANGELOG section as its body and nothing else. The gates still run first, # carries the CHANGELOG section as its body and nothing else. The gates still run first,
# in their own job and once, minus the race detector: race never runs on a push path or a # in their own job and once, including the race detector, which the push pipeline cannot
# tag, and the local gate raced this tree before the tag was cut. The write permission # afford and the release can. The write permission sits on the release job alone, and the
# sits on the release job alone, and the version contract these steps implement is in the # version contract these steps implement is in the `release` skill.
# `release` skill.
# #
# Every step is one command, so the step that fails is the gate that failed, and no shell # Every step is one command, so the step that fails is the gate that failed, and no shell
# option has to be trusted for the run to stop. The scripted steps are Perl, not shell and # option has to be trusted for the run to stop. The scripted steps are Perl, not shell and
@@ -21,15 +20,16 @@ on:
tags: ["v*"] tags: ["v*"]
env: env:
GOAMD64: v3
# The box is shared with the forge, so parallelism is bounded on purpose. The gates job # The box is shared with the forge, so parallelism is bounded on purpose. The gates job
# needs it most, since it runs the suite. # needs it most, since it runs the suite and the race detector.
GOFLAGS: -p=1 GOFLAGS: -p=1
GOMAXPROCS: "2" GOMAXPROCS: "2"
jobs: jobs:
gates: gates:
runs-on: fedora runs-on: fedora
timeout-minutes: 10 timeout-minutes: 25
steps: steps:
- uses: actions/checkout@v7 - uses: actions/checkout@v7
@@ -39,10 +39,10 @@ jobs:
go-version-file: go.mod go-version-file: go.mod
cache: true cache: true
- name: Install Perl - name: Install Perl and gcc
# Perl for the steps below. The install is a no-op where the package # Perl for the steps below, gcc for the race detector. Both are no-ops where the
# is already present. # package is already present.
run: dnf install -y perl run: dnf install -y perl gcc
- name: Validate the tag - name: Validate the tag
env: env:
@@ -91,16 +91,14 @@ jobs:
exit($total < 80 ? 1 : 0); exit($total < 80 ? 1 : 0);
' '
- name: Race
run: go test -race -count=1 -timeout 30m ./...
release: release:
runs-on: fedora runs-on: fedora
timeout-minutes: 15 timeout-minutes: 15
needs: gates needs: gates
permissions: permissions:
# contents: read is required for the checkout: a job that declares any
# permissions gets a token scoped to exactly those, and releases: write
# alone leaves the fetch with no read access, which Gitea answers with
# a 404 "Repository not found". Verified on the instance 2026-09-16.
contents: read
releases: write releases: write
steps: steps:
- uses: actions/checkout@v7 - uses: actions/checkout@v7
+5 -13
View File
@@ -26,22 +26,16 @@ on:
branches: [development] branches: [development]
env: env:
# Portable baseline, x86-64-v3 minimum. Other GOARCH values ignore it.
GOAMD64: v3
# One core: parallelism buys no speed here and costs memory the box does not have. # One core: parallelism buys no speed here and costs memory the box does not have.
GOFLAGS: -p=1 GOFLAGS: -p=1
GOMAXPROCS: "2" GOMAXPROCS: "2"
# A superseded run of the same ref is cancelled instead of queueing behind one
# that no longer matters. Verified on this Gitea on 2026-09-17: a queued run
# whose ref moved on is cancelled before it ever reaches the runner, while a
# run already dispatched there runs to completion.
concurrency:
group: ${{ gitea.workflow }}-${{ gitea.ref }}
cancel-in-progress: true
jobs: jobs:
test: test:
runs-on: fedora runs-on: fedora
timeout-minutes: 10 timeout-minutes: 20
steps: steps:
- uses: actions/checkout@v7 - uses: actions/checkout@v7
@@ -99,12 +93,10 @@ jobs:
# output has to be captured into a variable. # output has to be captured into a variable.
env: env:
GOBIN: ${{ gitea.workspace }}/bin GOBIN: ${{ gitea.workspace }}/bin
run: go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0 run: go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0
- name: Build the decoder - name: Build the decoder
run: go build -o bin/interpres-decode ./cmd/interpres-decode run: go build -o bin/interpres-decode ./cmd/interpres-decode
- name: Compliance suite - name: Compliance suite
# interpres implements TOML 1.0 and 1.1; the mode is pinned so an upstream run: bin/toml-test bin/interpres-decode
# default change cannot silently move the corpus.
run: bin/toml-test test -decoder=bin/interpres-decode -toml=1.1
+1 -108
View File
@@ -11,114 +11,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- -
## [1.1.0] - 2026-09-18 ## [1.0.0] - 2026-09-16
### Added
- TOML 1.1 support, on by default: date-times and times without seconds
(`07:32`, `1979-05-27T07:32`, normalised to full seconds on output), the
`\e` and `\xHH` escape sequences, and multi-line inline tables with
comments and trailing commas. The compliance suite runs in TOML 1.1 mode:
214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document
parses exactly as before.
- `interpres-decode -validate [file ...]`: a validate mode beside the
toml-test adapter. It parses each named file, or stdin when none are named,
prints one line per invalid document to stderr, and exits 0 when all are
valid, 1 when one is not, and 2 on a usage or read failure. Install it with
`go install .../cmd/interpres-decode@latest`; releases still ship no
binaries.
- `DecodeError` and `EncodeError`: decode and encode failures are wrapped in
typed errors carrying the key path, read with `errors.AsType` instead of
parsing the message text. The rendered messages keep their shape; the only
visible change is that an encode failure on a top-level field no longer
gains a meaningless leading dot in its path.
- `omitzero` and `omitempty` tag options on encode: `toml:"name,omitzero"`
skips a field whose value is the zero value of its type (a type with an
`IsZero() bool` method decides through the method), and
`toml:"name,omitempty"` skips a nil or empty slice, array, or map. The
decoder ignores both options.
### Changed
- The compliance suite is [toml-test](https://github.com/toml-lang/toml-test)
v2.2.0, up from v1.6.0. Its TOML 1.0 corpus holds 205 valid and 474 invalid
cases (185 and 371 before), and it caught the two documents the parser
still accepted, fixed below.
- The flattened struct layout the decoder consults is cached per struct type
and shared with the encoder, which now resolves duplicate field keys with
it. Strict decoding of an array of tables of structs runs about a quarter
faster; marshalling structs gained the same layout without measurable cost.
- The parser scans the input bytes in place instead of building a `[]rune`
copy of the document: every character that drives the grammar is ASCII and
the input is validated UTF-8 up front, so the conversion pass and its four
bytes per rune were pure overhead. Parsing a large array-of-tables document
runs about a fifth faster and allocates about half the memory.
- Numeric tokens without underscores skip the normalising rebuild: digits are
validated in place in `joinDigits`, and a float whose token is already
clean goes to `strconv.ParseFloat` directly. One allocation per integer
atom and two per float atom disappear.
### Fixed
- A `MarshalTOML` result of `nil` with a nil error fails the marshal with
`MarshalTOML returned a nil value`. The field silently vanished before, and
inside a value array the nil result reached reflection as a zero value and
panicked.
- Strict decoding reports the smallest unknown key. Several unknown keys in
one table made the message depend on Go's random map iteration order, so
the same document reported different keys across runs.
- Decoding into a struct that embeds a pointer to itself terminates. The
schema walk recursed through the embedded type forever, so such a
`Unmarshal` call hung the process; the walk now tracks the struct types on
the current path and stops when one repeats.
- An array-of-tables header whose path runs through an inline table
(`a = {b = {}}` followed by `[[a.b.c]]`) is rejected. The frozen-inline-table
check covered `[table]` headers and dotted keys but not the intermediate
steps of an array-of-tables header, so such a document silently extended the
inline table.
- A new element of an array of tables starts a fresh scope for dotted-key paths
and nested arrays of tables: `[[a]]`, `b.c = 1`, `[[a]]`, `[a.b]` parses, as
the TOML examples in the spec shape it. The records of the previous element
falsely rejected the same paths in the next one.
- `Marshal` emits exactly one key when two struct fields resolve to the same
TOML name, picking the field the decoder would fill (the shallower one, the
later declaration at equal depth). Such a struct previously marshalled into
a duplicate key, and the output never re-parsed, breaking the round-trip
guarantee.
- `Marshal` returns an error for a table header key or an inline-table key that
is not valid UTF-8, the way scalar keys already did, instead of silently
emitting corrupt TOML (a header that lost its key, an inline table with a
missing key).
- `UseLiteralMultiline` falls back to the escaped basic string when the value
cannot be carried verbatim by the literal form: a run of three single quotes,
a control character, or a lone carriage return. Such values previously
produced output that did not re-parse.
- A `[]any` holding only tables marshals in the value-array form with inline
tables, keeping the type `Parse` produces for such an array. It previously
took the `[[header]]` form, so a round-trip changed the value's type from
`[]any` to `[]map[string]any`.
- Decoding into a `uint` destination checks the type's platform width instead
of only the fixed widths, so a 32-bit `uint` no longer truncates silently;
decoding a finite float beyond the `float32` range is an overflow error
instead of a silent infinity.
- Struct fields that resolve to one key at equal depth decode through the
field declared later, matching the documented rule; the first one won before.
- A float with an exponent marker but no digits (`1e`, `0.0E`) is rejected;
the exponent requires at least one digit.
- A date-time offset outside 00:00 through 23:59 is rejected; such offsets
were accepted and silently rolled over (`+00:60` decoded as `+01:00`).
- Untagged embedded fields now decode symmetrically with encode: an embedded
struct receives its keys inline (a nil embedded pointer struct is
allocated), an embedded map catches the keys no field claims, and a name
clash resolves in favour of the shallower field. A struct with an untagged
embedded field previously decoded with all inline keys dropped and did not
round-trip.
- `Marshal` re-emits arrays that mix tables with scalars: the table elements
render as inline tables inside the value array. A tree that `Parse` accepts
from such a document previously failed with
`cannot encode map[string]interface {}`.
## [1.0.0] - 2026-08-20
First stable release: a dependency-free TOML 1.0 parser and encoder for Go that First stable release: a dependency-free TOML 1.0 parser and encoder for Go that
uses only the standard library and passes the entire uses only the standard library and passes the entire
+6 -27
View File
@@ -1,25 +1,6 @@
# Contributing # Contributing
Contributions to **interpres** are governed by the Contributor terms Thanks for contributing to **interpres**.
below; submitting one means you accept them.
## Contributor terms
1. This project belongs to its owner alone. The owner decides what is
accepted, in what form and when; the decision is final and needs no
justification.
2. By submitting a contribution you assign to Petr Balvín
<opensource@petrbalvin.org> all present and future copyright and
related rights in it, worldwide, for the full term of the rights,
with the right to relicense and sublicense without restriction,
including under proprietary terms.
3. Where that assignment is not effective, it counts as a perpetual,
irrevocable, royalty-free licence with the same scope.
4. To the fullest extent permitted by law, you waive any right of
attribution and integrity in the contribution. The project names no
contributors and keeps no credits list.
5. By submitting you represent that the work is yours and that you
hold the rights to assign it as above.
## Development setup ## Development setup
@@ -53,11 +34,9 @@ just test
7. Open a pull request against `development`. 7. Open a pull request against `development`.
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The
release workflow validates the tag, runs the static gates and the test suite release workflow runs the full gate set including the race detector and
with the coverage floor, and publishes the Gitea release with the matching publishes the Gitea release with the matching `CHANGELOG.md` section as its
`CHANGELOG.md` section as its notes. The race detector is not in that set: race notes.
never runs on a push path, and the local `just gates` raced the tree before the
tag was cut.
## Code style ## Code style
@@ -90,7 +69,7 @@ understandable, reviewable and genuinely useful.
``` ```
Name the model that did the work, spelled the way its maker spells it, for Name the model that did the work, spelled the way its maker spells it, for
example `GLM 5.3`, `DeepSeek V4.1 Flash` or `Qwen 3.8 Flash`. No example `GLM 5.3`, `DeepSeek V4 Flash` or `Qwen 3.8 Flash`. No
`Co-Authored-By`, no `Signed-off-by`, no other trailers, and no prose: the `Co-Authored-By`, no `Signed-off-by`, no other trailers, and no prose: the
trailer is the disclosure. trailer is the disclosure.
- **Issues and pull requests** attribute the assistance in a comment, for - **Issues and pull requests** attribute the assistance in a comment, for
@@ -116,7 +95,7 @@ Workflows live in `.gitea/workflows/` and run on the project's own runners:
|---|---|---| |---|---|---|
| Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite | | Test | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the coverage floor, the toml-test compliance suite |
| Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs | | Race | `workflow_dispatch`, by hand | the suite under the race detector, the same race gate the local `just gates` runs |
| Release | a `v*` tag | tag validation, format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release created from the `CHANGELOG.md` section; no race detector | | Release | a `v*` tag | the same gates plus the race detector, then the Gitea release created from the `CHANGELOG.md` section |
The local equivalent is `just gates`, which is the same set plus the race The local equivalent is `just gates`, which is the same set plus the race
detector. detector.
+10 -11
View File
@@ -1,18 +1,17 @@
# interpres # interpres
A TOML 1.0 and 1.1 parser and encoder for Go, written with the standard A TOML 1.0 parser and encoder for Go, written with the standard library alone.
library alone. `interpres` (Latin for *interpreter*) gives zero-dependency `interpres` (Latin for *interpreter*) gives zero-dependency programs an
programs an `encoding/json`-style API for reading and writing TOML, and passes `encoding/json`-style API for reading and writing TOML, and passes the entire
the entire official [toml-test](https://github.com/toml-lang/toml-test) suite: official [toml-test](https://github.com/toml-lang/toml-test) suite: 185 valid
214 valid and 467 invalid cases, zero failures. and 371 invalid cases, zero failures.
## Features ## Features
- **Full TOML 1.0 and 1.1**: bare, quoted and dotted keys; tables and arrays of - **Full TOML 1.0**: bare, quoted and dotted keys; tables and arrays of tables;
tables; basic and literal strings including multiline, with the 1.1 `\e` and basic and literal strings including multiline; integers in the four radixes
`\xHH` escapes; integers in the four radixes with `_` separators; floats with with `_` separators; floats with exponents, `inf` and `nan`; booleans; the
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds four date-time kinds; arrays and inline tables.
optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and - **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and
`Marshal` for structs and maps, mirroring `encoding/json`. `Marshal` for structs and maps, mirroring `encoding/json`.
- **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that - **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that
@@ -161,7 +160,7 @@ See [docs/DEVELOPMENT.md](docs/DEVELOPMENT.md) for the full workflow, and
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow - [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md): components and data flow
- [docs/API.md](docs/API.md): the API reference, decoding and encoding rules - [docs/API.md](docs/API.md): the API reference, decoding and encoding rules
- [docs/CLI.md](docs/CLI.md): the interpres-decode toml-test adapter and validator - [docs/CLI.md](docs/CLI.md): the interpres-decode toml-test adapter
## Licence ## Licence
-129
View File
@@ -1,129 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"fmt"
"strings"
"testing"
"time"
)
// benchDoc is a representative configuration document: every scalar kind, an
// inline table, sub-tables, and an array of tables.
var benchDoc = []byte(`title = "benchmark configuration"
replicas = 3
ratio = 0.75
enabled = true
when = 2026-09-17T12:00:00Z
local = 2026-09-17T12:00:00
tags = ["alpha", "beta", "gamma"]
limits = { cpu = 4, memory = 1024 }
[server]
host = "localhost"
port = 8080
hosts = ["a.example", "b.example"]
[server.tls]
enabled = true
cert = "/etc/cert.pem"
[[items]]
name = "first"
weight = 10
flags = ["x", "y"]
[[items]]
name = "second"
weight = 20
flags = ["z"]
`)
// longDoc is generated once so the large-input benchmarks measure parsing,
// not document construction. Roughly 2000 array-of-tables entries.
var longDoc = func() []byte {
var b strings.Builder
b.WriteString("title = \"long\"\n")
for i := range 2000 {
fmt.Fprintf(&b, "[[entry]]\nname = \"entry-%d\"\nweight = %d\nwhen = 2026-09-17T12:00:00Z\nratio = 0.5\ntags = [\"a\", \"b\", \"c\"]\n\n", i, i)
}
return []byte(b.String())
}()
type benchTLS struct {
Enabled bool `toml:"enabled"`
Cert string `toml:"cert"`
}
type benchServer struct {
Host string `toml:"host"`
Port int `toml:"port"`
Hosts []string `toml:"hosts"`
TLS benchTLS `toml:"tls"`
}
type benchItem struct {
Name string `toml:"name"`
Weight int `toml:"weight"`
Flags []string `toml:"flags"`
}
type benchConfig struct {
Title string `toml:"title"`
Replicas int `toml:"replicas"`
Ratio float64 `toml:"ratio"`
Enabled bool `toml:"enabled"`
When time.Time `toml:"when"`
Local LocalDateTime `toml:"local"`
Tags []string `toml:"tags"`
Limits map[string]any `toml:"limits"`
Server benchServer `toml:"server"`
Items []benchItem `toml:"items"`
}
func BenchmarkParse(b *testing.B) {
b.ReportAllocs()
b.SetBytes(int64(len(benchDoc)))
for b.Loop() {
if _, err := Parse(benchDoc); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkMarshal(b *testing.B) {
tree, err := Parse(benchDoc)
if err != nil {
b.Fatal(err)
}
b.ReportAllocs()
b.SetBytes(int64(len(benchDoc)))
for b.Loop() {
if _, err := Marshal(tree); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkStrictDecode(b *testing.B) {
dec := NewDecoder().DisallowUnknownFields()
b.ReportAllocs()
for b.Loop() {
var cfg benchConfig
if err := dec.Decode(benchDoc, &cfg); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkParseLong(b *testing.B) {
b.ReportAllocs()
b.SetBytes(int64(len(longDoc)))
for b.Loop() {
if _, err := Parse(longDoc); err != nil {
b.Fatal(err)
}
}
}
+9 -64
View File
@@ -1,23 +1,17 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org) // Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT // SPDX-License-Identifier: MIT
// Command interpres-decode is the toml-test harness adapter and a TOML // Command interpres-decode reads a TOML document from standard input and writes
// validator. Without flags it reads a TOML document from standard input and // the toml-test "tagged JSON" representation to standard output.
// writes the toml-test "tagged JSON" representation to standard output. With
// -validate it checks the named documents, or standard input when none are
// named, and exits non-zero on the first invalid one:
// //
// interpres-decode -validate config.toml // It exits non-zero on a parse error, which is how the toml-test harness checks
// // that invalid documents are rejected. Run the official suite against it with:
// Run the official suite against the adapter with:
// //
// toml-test ./interpres-decode // toml-test ./interpres-decode
package main package main
import ( import (
"encoding/json" "encoding/json"
"errors"
"flag"
"fmt" "fmt"
"io" "io"
"math" "math"
@@ -29,29 +23,13 @@ import (
) )
func main() { func main() {
os.Exit(Run(os.Args[1:], os.Stdin, os.Stdout, os.Stderr)) os.Exit(Run(os.Stdin, os.Stdout, os.Stderr))
} }
// Run runs the command line and returns the process exit code: 0 success, // Run reads a TOML document from stdin, emits the toml-test tagged-JSON form
// 1 an invalid document, 2 a usage, reading, encoding, or // on stdout, and returns the process exit code (0 success, 1 parse error,
// unsupported-value error. // 2 I/O, encoding, or unsupported-value error).
func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int { func Run(stdin io.Reader, stdout, stderr io.Writer) int {
fs := flag.NewFlagSet("interpres-decode", flag.ContinueOnError)
fs.SetOutput(stderr)
validate := fs.Bool("validate", false, "validate the documents instead of emitting tagged JSON")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
return 0
}
return 2
}
if *validate {
return validatePaths(fs.Args(), stdin, stderr)
}
if fs.NArg() > 0 {
fmt.Fprintln(stderr, "interpres-decode: the adapter mode takes no arguments; name files with -validate")
return 2
}
data, err := io.ReadAll(stdin) data, err := io.ReadAll(stdin)
if err != nil { if err != nil {
fmt.Fprintln(stderr, "read stdin:", err) fmt.Fprintln(stderr, "read stdin:", err)
@@ -76,39 +54,6 @@ func Run(args []string, stdin io.Reader, stdout, stderr io.Writer) int {
return 0 return 0
} }
// validatePaths parses every named file, or standard input when none are
// named, and reports each invalid document on stderr. It returns 0 when all
// documents parse, 1 when one does not, and 2 on a usage or read failure.
func validatePaths(paths []string, stdin io.Reader, stderr io.Writer) int {
if len(paths) == 0 {
paths = []string{"-"}
}
valid := true
for _, p := range paths {
name := p
var data []byte
var err error
if p == "-" {
data, err = io.ReadAll(stdin)
name = "<stdin>"
} else {
data, err = os.ReadFile(p)
}
if err != nil {
fmt.Fprintf(stderr, "interpres-decode: %s: %v\n", name, err)
return 2
}
if _, err := interpres.Parse(data); err != nil {
fmt.Fprintf(stderr, "%s: %v\n", name, err)
valid = false
}
}
if !valid {
return 1
}
return 0
}
// tag converts an interpres value into its toml-test tagged-JSON form. Tables // tag converts an interpres value into its toml-test tagged-JSON form. Tables
// become JSON objects and arrays become JSON arrays; scalars are wrapped in a // become JSON objects and arrays become JSON arrays; scalars are wrapped in a
// {"type", "value"} object. An error is returned for value types the encoder // {"type", "value"} object. An error is returned for value types the encoder
+4 -77
View File
@@ -7,7 +7,6 @@ import (
"bytes" "bytes"
"encoding/json" "encoding/json"
"errors" "errors"
"os"
"strings" "strings"
"testing" "testing"
"time" "time"
@@ -21,7 +20,7 @@ func TestRunParsesValidTOML(t *testing.T) {
port = 8080 port = 8080
enabled = true enabled = true
`)) `))
if code := Run(nil, in, &stdout, &stderr); code != 0 { if code := Run(in, &stdout, &stderr); code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String()) t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
} }
var got map[string]any var got map[string]any
@@ -42,7 +41,7 @@ enabled = true
func TestRunRejectsInvalidInput(t *testing.T) { func TestRunRejectsInvalidInput(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("v = \n")) in := bytes.NewReader([]byte("v = \n"))
code := Run(nil, in, &stdout, &stderr) code := Run(in, &stdout, &stderr)
if code != 1 { if code != 1 {
t.Errorf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String()) t.Errorf("Run returned %d, want 1 (parse error); stderr = %q", code, stderr.String())
} }
@@ -53,7 +52,7 @@ func TestRunRejectsInvalidInput(t *testing.T) {
func TestRunReadErrorReturnsTwo(t *testing.T) { func TestRunReadErrorReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
code := Run(nil, errorReader{}, &stdout, &stderr) code := Run(errorReader{}, &stdout, &stderr)
if code != 2 { if code != 2 {
t.Errorf("Run returned %d, want 2 (read error); stderr = %q", code, stderr.String()) t.Errorf("Run returned %d, want 2 (read error); stderr = %q", code, stderr.String())
} }
@@ -71,7 +70,7 @@ func TestRunEncodeErrorReturnsTwo(t *testing.T) {
var stderr bytes.Buffer var stderr bytes.Buffer
w := errorWriter{} w := errorWriter{}
in := bytes.NewReader([]byte(`k = "v"` + "\n")) in := bytes.NewReader([]byte(`k = "v"` + "\n"))
code := Run(nil, in, w, &stderr) code := Run(in, w, &stderr)
if code != 2 { if code != 2 {
t.Errorf("Run returned %d, want 2 (encode error); stderr = %q", code, stderr.String()) t.Errorf("Run returned %d, want 2 (encode error); stderr = %q", code, stderr.String())
} }
@@ -211,75 +210,3 @@ func TestTaggedHelper(t *testing.T) {
t.Errorf("tagged = %#v", got) t.Errorf("tagged = %#v", got)
} }
} }
func TestValidateStdinAcceptsValidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"ok\"\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 0 {
t.Fatalf("Run returned %d, stderr = %q", code, stderr.String())
}
if stdout.Len() != 0 || stderr.Len() != 0 {
t.Fatalf("validate should be quiet on success, stdout %q stderr %q", stdout.String(), stderr.String())
}
}
func TestValidateStdinRejectsInvalidDocument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("title = \"unterminated\n"))
if code := Run([]string{"-validate"}, in, &stdout, &stderr); code != 1 {
t.Fatalf("Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "<stdin>") || !strings.Contains(stderr.String(), "line 1") {
t.Fatalf("stderr = %q, want the name and the line", stderr.String())
}
if stdout.Len() != 0 {
t.Fatalf("stdout should stay empty, got %q", stdout.String())
}
}
func TestValidateFiles(t *testing.T) {
dir := t.TempDir()
good := dir + "/good.toml"
bad := dir + "/bad.toml"
if err := os.WriteFile(good, []byte("a = 1\n"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(bad, []byte("a =\n"), 0o644); err != nil {
t.Fatal(err)
}
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", good}, nil, &stdout, &stderr); code != 0 {
t.Fatalf("one valid file: Run returned %d, stderr = %q", code, stderr.String())
}
if code := Run([]string{"-validate", good, bad}, nil, &stdout, &stderr); code != 1 {
t.Fatalf("valid plus invalid: Run returned %d, want 1; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), bad) || !strings.Contains(stderr.String(), "line 1") {
t.Fatalf("stderr = %q, want the file name and the line", stderr.String())
}
}
func TestValidateMissingFileReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-validate", "no-such-file.toml"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
func TestAdapterModeRejectsPositionalArgument(t *testing.T) {
var stdout, stderr bytes.Buffer
in := bytes.NewReader([]byte("a = 1\n"))
if code := Run([]string{"file.toml"}, in, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
if !strings.Contains(stderr.String(), "-validate") {
t.Fatalf("stderr = %q, want it to point at -validate", stderr.String())
}
}
func TestUnknownFlagReturnsTwo(t *testing.T) {
var stdout, stderr bytes.Buffer
if code := Run([]string{"-nope"}, nil, &stdout, &stderr); code != 2 {
t.Fatalf("Run returned %d, want 2; stderr = %q", code, stderr.String())
}
}
+4 -25
View File
@@ -6,7 +6,6 @@ package interpres
import ( import (
"fmt" "fmt"
"regexp" "regexp"
"strconv"
"strings" "strings"
"time" "time"
) )
@@ -59,39 +58,26 @@ var (
"2006-01-02T15:04:05Z07:00", "2006-01-02T15:04:05Z07:00",
"2006-01-02 15:04:05.999999999Z07:00", "2006-01-02 15:04:05.999999999Z07:00",
"2006-01-02 15:04:05Z07:00", "2006-01-02 15:04:05Z07:00",
// TOML 1.1 makes the seconds optional.
"2006-01-02T15:04Z07:00",
"2006-01-02 15:04Z07:00",
} }
localDateTimeLayouts = []string{ localDateTimeLayouts = []string{
"2006-01-02T15:04:05.999999999", "2006-01-02T15:04:05.999999999",
"2006-01-02T15:04:05", "2006-01-02T15:04:05",
"2006-01-02 15:04:05.999999999", "2006-01-02 15:04:05.999999999",
"2006-01-02 15:04:05", "2006-01-02 15:04:05",
"2006-01-02T15:04",
"2006-01-02 15:04",
} }
localTimeLayouts = []string{ localTimeLayouts = []string{
"15:04:05.999999999", "15:04:05.999999999",
"15:04:05", "15:04:05",
"15:04",
} }
) )
// dateTimeShape enforces the strict TOML grammar (two-digit components, // dateTimeShape enforces the strict TOML grammar (two-digit components) that
// seconds optional since 1.1, a fraction only after seconds) that time.Parse // time.Parse would otherwise accept loosely (e.g. a single-digit hour).
// would otherwise accept loosely (e.g. a single-digit hour).
var dateTimeShape = regexp.MustCompile( var dateTimeShape = regexp.MustCompile(
`^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}(:\d{2}(\.\d+)?)?([Zz]|[+-]\d{2}:\d{2})?)?$` + `^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}:\d{2}(\.\d+)?([Zz]|[+-]\d{2}:\d{2})?)?$` +
`|^\d{2}:\d{2}(:\d{2}(\.\d+)?)?$`, `|^\d{2}:\d{2}:\d{2}(\.\d+)?$`,
) )
// offsetBounds extracts the numeric offset of a date-time. The ABNF bounds it
// to 00:00 through 23:59, but time.Parse accepts values outside that range
// and rolls them over (for example "+00:60" becomes "+01:00"), so the bounds
// are enforced here.
var offsetBounds = regexp.MustCompile(`([+-])(\d{2}):(\d{2})$`)
// parseDateTime classifies and parses a bare token as a TOML date-time value. // parseDateTime classifies and parses a bare token as a TOML date-time value.
// It returns the decoded value (time.Time, LocalDateTime, LocalDate, or // It returns the decoded value (time.Time, LocalDateTime, LocalDate, or
// LocalTime) and whether the token was a date-time at all. // LocalTime) and whether the token was a date-time at all.
@@ -105,13 +91,6 @@ func parseDateTime(tok string) (any, bool) {
if !dateTimeShape.MatchString(tok) { if !dateTimeShape.MatchString(tok) {
return nil, false return nil, false
} }
if m := offsetBounds.FindStringSubmatch(tok); m != nil {
hour, _ := strconv.Atoi(m[2])
minute, _ := strconv.Atoi(m[3])
if hour > 23 || minute > 59 {
return nil, false
}
}
// The ABNF accepts lowercase "t"/"z"; time.Parse only matches uppercase. // The ABNF accepts lowercase "t"/"z"; time.Parse only matches uppercase.
norm := strings.ToUpper(tok) norm := strings.ToUpper(tok)
for _, layout := range offsetDateTimeLayouts { for _, layout := range offsetDateTimeLayouts {
+38 -162
View File
@@ -5,10 +5,9 @@ package interpres
import ( import (
"fmt" "fmt"
"math"
"reflect" "reflect"
"slices"
"strings" "strings"
"sync"
"time" "time"
) )
@@ -106,45 +105,17 @@ func (d *decoder) assignTable(tbl map[string]any, dst reflect.Value) error {
} }
func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error { func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
schema := cachedStructSchema(dst.Type()) fields := structFields(dst.Type())
if d.disallowUnknown {
// Map iteration order is random, so pick the unknown key to report
// deterministically: the smallest one.
unknown := ""
for key := range tbl {
if _, ok := schema.byName[strings.ToLower(key)]; ok {
continue
}
if unknown == "" || key < unknown {
unknown = key
}
}
if unknown != "" {
return fmt.Errorf("interpres: unknown field %q for %s", unknown, dst.Type())
}
}
for key, val := range tbl { for key, val := range tbl {
field, ok := schema.byName[strings.ToLower(key)] field, ok := fields[strings.ToLower(key)]
if !ok { if !ok {
if schema.embedMaps != nil { if d.disallowUnknown {
// Leftover keys land in an untagged embedded map, the inverse return fmt.Errorf("interpres: unknown field %q for %s", key, dst.Type())
// of the encoder inlining that map's entries.
mv, err := fieldByIndex(dst, schema.embedMaps[0])
if err != nil {
return newDecodeError(key, err)
}
if err := d.assignMap(map[string]any{key: val}, mv); err != nil {
return newDecodeError(key, err)
}
} }
continue continue
} }
fv, err := fieldByIndex(dst, field.index) if err := d.assign(val, dst.Field(field)); err != nil {
if err != nil { return fmt.Errorf("%s: %w", key, err)
return newDecodeError(key, err)
}
if err := d.assign(val, fv); err != nil {
return newDecodeError(key, err)
} }
} }
return nil return nil
@@ -161,7 +132,7 @@ func (d *decoder) assignMap(tbl map[string]any, dst reflect.Value) error {
for key, val := range tbl { for key, val := range tbl {
elem := reflect.New(elemType).Elem() elem := reflect.New(elemType).Elem()
if err := d.assign(val, elem); err != nil { if err := d.assign(val, elem); err != nil {
return newDecodeError(key, err) return fmt.Errorf("%s: %w", key, err)
} }
dst.SetMapIndex(reflect.ValueOf(key), elem) dst.SetMapIndex(reflect.ValueOf(key), elem)
} }
@@ -175,7 +146,7 @@ func (d *decoder) assignSlice(items []any, dst reflect.Value) error {
out := reflect.MakeSlice(dst.Type(), len(items), len(items)) out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items { for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil { if err := d.assign(item, out.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err) return fmt.Errorf("[%d]: %w", i, err)
} }
} }
dst.Set(out) dst.Set(out)
@@ -189,7 +160,7 @@ func (d *decoder) assignTableSlice(items []map[string]any, dst reflect.Value) er
out := reflect.MakeSlice(dst.Type(), len(items), len(items)) out := reflect.MakeSlice(dst.Type(), len(items), len(items))
for i, item := range items { for i, item := range items {
if err := d.assign(item, out.Index(i)); err != nil { if err := d.assign(item, out.Index(i)); err != nil {
return newDecodeError(fmt.Sprintf("[%d]", i), err) return fmt.Errorf("[%d]: %w", i, err)
} }
} }
dst.Set(out) dst.Set(out)
@@ -217,21 +188,21 @@ func setInt(dst reflect.Value, v int64) error {
if v < 0 { if v < 0 {
return fmt.Errorf("interpres: cannot assign negative %d to %s", v, dst.Type()) return fmt.Errorf("interpres: cannot assign negative %d to %s", v, dst.Type())
} }
// OverflowUint knows every width, uint included on platforms where it var max uint64
// is narrower than uint64; SetUint would silently truncate instead. switch dst.Kind() {
if dst.OverflowUint(uint64(v)) { case reflect.Uint8:
max = math.MaxUint8
case reflect.Uint16:
max = math.MaxUint16
case reflect.Uint32:
max = math.MaxUint32
}
if max != 0 && uint64(v) > max {
return fmt.Errorf("interpres: integer %d overflows %s", v, dst.Type()) return fmt.Errorf("interpres: integer %d overflows %s", v, dst.Type())
} }
dst.SetUint(uint64(v)) dst.SetUint(uint64(v))
case reflect.Float32, reflect.Float64: case reflect.Float32, reflect.Float64:
// A finite value beyond the float32 range would silently become ±Inf; dst.SetFloat(float64(v))
// infinities and NaN themselves pass through. An int64 never
// overflows either float width.
f := float64(v)
if dst.OverflowFloat(f) {
return fmt.Errorf("interpres: integer %d overflows %s", v, dst.Type())
}
dst.SetFloat(f)
default: default:
return fmt.Errorf("interpres: cannot assign integer to %s", dst.Type()) return fmt.Errorf("interpres: cannot assign integer to %s", dst.Type())
} }
@@ -241,9 +212,6 @@ func setInt(dst reflect.Value, v int64) error {
func setFloat(dst reflect.Value, v float64) error { func setFloat(dst reflect.Value, v float64) error {
switch dst.Kind() { switch dst.Kind() {
case reflect.Float32, reflect.Float64: case reflect.Float32, reflect.Float64:
if dst.OverflowFloat(v) {
return fmt.Errorf("interpres: float %g overflows %s", v, dst.Type())
}
dst.SetFloat(v) dst.SetFloat(v)
return nil return nil
default: default:
@@ -251,118 +219,26 @@ func setFloat(dst reflect.Value, v float64) error {
} }
} }
// structFieldLoc locates one destination field by its index path from the // structFields builds a lower-cased lookup of field name → field index for the
// struct root and by the depth the field sits at, which breaks name clashes // exported fields of t, honouring `toml:"name"` tags.
// in favour of the shallower field. func structFields(t reflect.Type) map[string]int {
type structFieldLoc struct { fields := make(map[string]int, t.NumField())
index []int for i := range t.NumField() {
depth int f := t.Field(i)
} if f.PkgPath != "" { // unexported
continue
// structSchema flattens the exported fields of t for decode, mirroring the }
// encoder: an untagged embedded struct is inlined, so its own fields match name := f.Name
// keys of the same table, and an untagged embedded map is recorded in if tag, ok := f.Tag.Lookup("toml"); ok {
// embedMaps (first declaration first) as the destination for leftover keys. tag = strings.Split(tag, ",")[0]
// When two fields resolve to one name, the shallower wins, then the later if tag == "-" {
// declaration.
type structSchema struct {
byName map[string]structFieldLoc
embedMaps [][]int
}
// structSchemaCache holds one schema per struct type. A schema is immutable
// once published, so concurrent callers only race to build an identical value,
// the same trade-off encoding/json's field cache makes. The cache grows with
// the number of distinct types decoded or encoded, never per document.
var structSchemaCache sync.Map // reflect.Type -> structSchema
func cachedStructSchema(t reflect.Type) structSchema {
if s, ok := structSchemaCache.Load(t); ok {
return s.(structSchema)
}
s := newStructSchema(t)
actual, _ := structSchemaCache.LoadOrStore(t, s)
return actual.(structSchema)
}
func newStructSchema(t reflect.Type) structSchema {
s := structSchema{byName: make(map[string]structFieldLoc, t.NumField())}
// A struct may embed a pointer to itself, which is legal Go, so the walk
// tracks the struct types on the current path and stops when one repeats;
// without the guard the recursion never terminates. A self-promoted key
// always loses to the shallower original, so skipping it changes nothing.
visiting := map[reflect.Type]bool{}
var walk func(t reflect.Type, prefix []int, depth int)
walk = func(t reflect.Type, prefix []int, depth int) {
visiting[t] = true
defer delete(visiting, t)
for i := range t.NumField() {
f := t.Field(i)
if f.PkgPath != "" { // unexported
continue continue
} }
path := append(append([]int{}, prefix...), i) if tag != "" {
name := "" name = tag
if tag, ok := f.Tag.Lookup("toml"); ok {
name, _, _ = strings.Cut(tag, ",")
if name == "-" {
continue
}
}
if f.Anonymous && name == "" {
ft := f.Type
for ft.Kind() == reflect.Pointer {
ft = ft.Elem()
}
switch {
case ft.Kind() == reflect.Struct && !isScalarStruct(ft):
if !visiting[ft] {
walk(ft, path, depth+1)
}
continue
case ft.Kind() == reflect.Map && ft.Key().Kind() == reflect.String:
s.embedMaps = append(s.embedMaps, path)
continue
}
name = f.Name
}
if name == "" {
name = f.Name
}
key := strings.ToLower(name)
if existing, ok := s.byName[key]; !ok || depth <= existing.depth {
s.byName[key] = structFieldLoc{index: path, depth: depth}
} }
} }
fields[strings.ToLower(name)] = i
} }
walk(t, nil, 0) return fields
return s
}
// ownsKey reports whether the field at path is the one that resolves key.
// The encoder consults it to emit exactly the field the decoder would fill,
// so a struct with two fields mapping to one key does not marshal into a
// duplicate TOML key.
func (s structSchema) ownsKey(key string, path []int) bool {
loc, ok := s.byName[key]
return ok && slices.Equal(loc.index, path)
}
// fieldByIndex walks an index path from a struct value, allocating nil
// pointers along the way so a key can reach through an embedded pointer
// struct. Every field on the path is exported, so each step is settable.
func fieldByIndex(v reflect.Value, path []int) (reflect.Value, error) {
for i, x := range path {
v = v.Field(x)
if i < len(path)-1 && v.Kind() == reflect.Pointer {
if v.IsNil() {
if !v.CanSet() {
return reflect.Value{}, fmt.Errorf("cannot allocate nil embedded pointer")
}
v.Set(reflect.New(v.Type().Elem()))
}
v = v.Elem()
}
}
return v, nil
} }
-231
View File
@@ -8,7 +8,6 @@ import (
"errors" "errors"
"fmt" "fmt"
"math" "math"
"slices"
"strings" "strings"
"testing" "testing"
) )
@@ -223,33 +222,6 @@ func TestUnmarshalIntToUint64FitsMaxInt64(t *testing.T) {
} }
} }
func TestUnmarshalFloat32Overflow(t *testing.T) {
// A finite float64 beyond the float32 range must not decode silently as
// an infinity.
type C struct {
X float32 `toml:"x"`
}
var c C
err := Unmarshal([]byte("x = 1e300\n"), &c)
if err == nil {
t.Fatal("expected overflow error for float32")
}
if !strings.Contains(err.Error(), "overflow") {
t.Errorf("err = %v, want substring 'overflow'", err.Error())
}
// Infinities themselves pass through, and in-range values are untouched.
var ok C
if err := Unmarshal([]byte("x = inf\n"), &ok); err != nil {
t.Fatalf("inf should decode into float32, got %v", err)
}
if !math.IsInf(float64(ok.X), 1) {
t.Errorf("X = %v, want +Inf", ok.X)
}
if err := Unmarshal([]byte("x = 1.5\n"), &ok); err != nil || ok.X != 1.5 {
t.Fatalf("1.5 should decode into float32, got %v (X=%v)", err, ok.X)
}
}
func TestUnmarshalNegativeIntToUint(t *testing.T) { func TestUnmarshalNegativeIntToUint(t *testing.T) {
type C struct { type C struct {
X uint8 `toml:"x"` X uint8 `toml:"x"`
@@ -522,206 +494,3 @@ field = "y"
t.Errorf("Field = %q, want \"y\"", cfg.R.Field) t.Errorf("Field = %q, want \"y\"", cfg.R.Field)
} }
} }
// --- embedded field symmetry -----------------------------------------------
type RoundTripBase struct {
ID int `toml:"id"`
Name string `toml:"name"`
}
type RoundTripDerived struct {
RoundTripBase
X string `toml:"x"`
}
func TestUnmarshalEmbeddedStructRoundTrip(t *testing.T) {
orig := RoundTripDerived{ID: 1, Name: "b", X: "x"}
out, err := Marshal(orig)
if err != nil {
t.Fatalf("marshal: %v", err)
}
var back RoundTripDerived
if err := Unmarshal(out, &back); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if back != orig {
t.Fatalf("round-trip mismatch:\nwas: %+v\nnow: %+v", orig, back)
}
}
type RoundTripPtrCfg struct {
*RoundTripBase
X string `toml:"x"`
}
func TestUnmarshalEmbeddedPointerStruct(t *testing.T) {
var cfg RoundTripPtrCfg
if err := Unmarshal([]byte("id = 7\nname = \"n\"\nx = \"x\"\n"), &cfg); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.RoundTripBase == nil || cfg.ID != 7 || cfg.Name != "n" || cfg.X != "x" {
t.Fatalf("decoded: %+v", cfg)
}
}
// A struct embedding a pointer to itself is legal Go; decoding into it must
// terminate. The schema walk used to recurse through the embedded type
// forever.
func TestUnmarshalSelfEmbeddedPointerStructTerminates(t *testing.T) {
type SelfLink struct {
*SelfLink
X int `toml:"x"`
Y string `toml:"y"`
}
var n SelfLink
if err := Unmarshal([]byte("x = 1\ny = \"s\"\n"), &n); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if n.X != 1 || n.Y != "s" {
t.Fatalf("decoded: %+v", n)
}
// A nil self pointer on the encode side stays skippable, as any nil
// embedded pointer is.
out, err := Marshal(SelfLink{X: 2})
if err != nil {
t.Fatalf("marshal: %v", err)
}
if want := "x = 2\ny = \"\"\n"; string(out) != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
type RoundTripExtra map[string]int
type RoundTripMapCfg struct {
RoundTripExtra
X string `toml:"x"`
}
func TestUnmarshalEmbeddedMap(t *testing.T) {
var cfg RoundTripMapCfg
if err := Unmarshal([]byte("alpha = 1\nx = \"x\"\n"), &cfg); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.RoundTripExtra["alpha"] != 1 || cfg.X != "x" {
t.Fatalf("decoded: %+v", cfg)
}
orig := RoundTripMapCfg{RoundTripExtra: RoundTripExtra{"a": 1}, X: "x"}
out, err := Marshal(orig)
if err != nil {
t.Fatalf("marshal: %v", err)
}
var back RoundTripMapCfg
if err := Unmarshal(out, &back); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if back.X != "x" || back.RoundTripExtra["a"] != 1 {
t.Fatalf("round-trip mismatch: %+v", back)
}
}
func TestUnmarshalEmbeddedNameClashShallowerWins(t *testing.T) {
type Inner struct {
Name string `toml:"name"`
Deep string `toml:"deep"`
}
type Outer struct {
Inner
Name string `toml:"name"`
}
var v Outer
if err := Unmarshal([]byte("name = \"outer\"\ndeep = \"d\"\n"), &v); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if v.Name != "outer" || v.Deep != "d" {
t.Fatalf("decoded: %+v", v)
}
}
func TestUnmarshalNameClashEqualDepthLaterWins(t *testing.T) {
// At equal depth the field declared later resolves the name, matching the
// documented rule.
type C struct {
First string `toml:"v"`
Second int `toml:"v"`
}
var c C
if err := Unmarshal([]byte("v = 1\n"), &c); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if c.Second != 1 {
t.Fatalf("decoded: %+v, want the later field to take the value", c)
}
}
func TestUnmarshalUnknownKeyWithoutEmbeddedMap(t *testing.T) {
var cfg RoundTripDerived
if err := Unmarshal([]byte("rogue = 1\n"), &cfg); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if cfg.ID != 0 || cfg.X != "" {
t.Fatalf("decoded: %+v", cfg)
}
}
func TestUnmarshalStrictEmbeddedMapStaysStrict(t *testing.T) {
type Cfg struct {
RoundTripExtra
Name string `toml:"name"`
}
dec := NewDecoder().DisallowUnknownFields()
err := dec.Decode([]byte("name = \"n\"\nrogue = 1\n"), &Cfg{})
if err == nil || !strings.Contains(err.Error(), "unknown field") {
t.Fatalf("expected unknown field error, got: %v", err)
}
}
func TestDecodeErrorCarriesPath(t *testing.T) {
type Item struct {
Name string `toml:"name"`
Weight uint8 `toml:"weight"`
}
type Cfg struct {
Tags []string `toml:"tags"`
Items []Item `toml:"items"`
}
var cfg Cfg
err := Unmarshal([]byte("[[items]]\nname = \"a\"\nweight = 300\n"), &cfg)
if err == nil {
t.Fatal("expected an overflow error")
}
de, ok := errors.AsType[*DecodeError](err)
if !ok {
t.Fatalf("expected a *DecodeError, got %T: %v", err, err)
}
want := []string{"items", "[0]", "weight"}
if !slices.Equal(de.Path, want) {
t.Fatalf("Path = %v, want %v", de.Path, want)
}
if de.Err == nil || !strings.Contains(de.Err.Error(), "overflows uint8") {
t.Fatalf("Err = %v", de.Err)
}
// The rendered message keeps its shape: segments joined with ": ".
wantMsg := "items: [0]: weight: interpres: integer 300 overflows uint8"
if err.Error() != wantMsg {
t.Fatalf("message = %q, want %q", err.Error(), wantMsg)
}
}
func TestDecodeErrorOnMapDestination(t *testing.T) {
var m map[string]uint8
err := Unmarshal([]byte("count = -1\n"), &m)
if err == nil {
t.Fatal("expected an error")
}
de, ok := errors.AsType[*DecodeError](err)
if !ok {
t.Fatalf("expected a *DecodeError, got %T: %v", err, err)
}
if !slices.Equal(de.Path, []string{"count"}) {
t.Fatalf("Path = %v", de.Path)
}
}
+20 -105
View File
@@ -7,11 +7,6 @@ package. The snippets assume:
import "sourcedock.dev/petrbalvin/interpres" import "sourcedock.dev/petrbalvin/interpres"
``` ```
The parser accepts TOML 1.0 documents plus the TOML 1.1 extensions: date-times
and times without seconds, the `\e` and `\xHH` escape sequences, and
multi-line inline tables with comments and trailing commas. The encoder emits
TOML 1.0, which is valid under both versions.
## Functions ## Functions
### `func Parse(data []byte) (map[string]any, error)` ### `func Parse(data []byte) (map[string]any, error)`
@@ -51,7 +46,7 @@ The cancellable variant of `Unmarshal`.
### `func Marshal(v any) ([]byte, error)` ### `func Marshal(v any) ([]byte, error)`
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
TOML document. The emission rules are in the [Encoding](#encoding) section TOML 1.0 document. The emission rules are in the [Encoding](#encoding) section
below. Equivalent to `MarshalContext(context.Background(), v)`. below. Equivalent to `MarshalContext(context.Background(), v)`.
```go ```go
@@ -105,23 +100,16 @@ For a struct destination, a TOML key matches a field as follows:
1. The `toml:"name"` tag, using the part before any comma. The literal `-` 1. The `toml:"name"` tag, using the part before any comma. The literal `-`
excludes the field. excludes the field.
2. Without a tag, the lower-cased field name. 2. Without a tag, the lower-cased field name.
3. An anonymous (embedded) field without a tag is inlined: the decoder walks 3. The key itself is lower-cased before lookup, so the match is
into the embedded struct and matches its own fields against the same keys,
mirroring how the encoder flattens it. A nil embedded pointer struct is
allocated on demand. An untagged embedded map receives the keys no field
claims.
4. The key itself is lower-cased before lookup, so the match is
case-insensitive on both sides: `DATABASEURL` matches a field named case-insensitive on both sides: `DATABASEURL` matches a field named
`DatabaseUrl`. `DatabaseUrl`.
The match is exact after lower-casing. No separator is inserted, so a TOML key The match is exact after lower-casing. No separator is inserted, so a TOML key
`database_url` does not match a field named `DatabaseUrl`; tag such a field `database_url` does not match a field named `DatabaseUrl`; tag such a field
(`toml:"database_url"`) or use the lower-cased name as the key. When two (`toml:"database_url"`) or use the lower-cased name as the key. When two fields
fields resolve to the same name, the shallower one wins; at equal depth, the resolve to the same name, the one declared later wins.
one declared later wins.
Unknown keys are ignored by default, landing in an untagged embedded map when Unknown keys are ignored by default; see [Strict decoding](#strict-decoding).
the struct has one; [Strict decoding](#strict-decoding) rejects them instead.
### Numeric conversion ### Numeric conversion
@@ -131,8 +119,8 @@ The decoder converts to the destination type with explicit overflow checks:
| Destination kind | Rule | | Destination kind | Rule |
|---|---| |---|---|
| `int`, `int8`, `int16`, `int32`, `int64` | the `int64` value must not overflow the destination | | `int`, `int8`, `int16`, `int32`, `int64` | the `int64` value must not overflow the destination |
| `uint`, `uint8`, `uint16`, `uint32`, `uint64` | the value must be non-negative and must not overflow the destination's own width, `uint` on a 32-bit platform included; `uint64` accepts any non-negative `int64` | | `uint`, `uint8`, `uint16`, `uint32`, `uint64` | the value must be non-negative; `uint8`, `uint16` and `uint32` enforce their own maxima; `uint64` accepts any non-negative `int64` |
| `float32`, `float64` | copied verbatim, except that a finite value beyond the `float32` range is an overflow error rather than a silent infinity; an integer also coerces, so TOML `5` decodes into `5.0` | | `float32`, `float64` | copied verbatim; an integer also coerces, so TOML `5` decodes into `5.0` |
| `bool`, `string` | exact kind match only, no coercion across kinds | | `bool`, `string` | exact kind match only, no coercion across kinds |
| `time.Time` | offset date-times only; no implicit conversion to or from the local variants | | `time.Time` | offset date-times only; no implicit conversion to or from the local variants |
@@ -144,10 +132,8 @@ offending key or index, for example `p: interpres: integer 300 overflows uint8`.
Offset date-times decode into `time.Time` and keep their offset. The local Offset date-times decode into `time.Time` and keep their offset. The local
variants decode into `LocalDateTime`, `LocalDate` and `LocalTime`, whose variants decode into `LocalDateTime`, `LocalDate` and `LocalTime`, whose
embedded `time.Time` is normalised to UTC (midnight UTC for a local date, the embedded `time.Time` is normalised to UTC (midnight UTC for a local date, the
zero date for a local time). Every kind may omit the seconds as of TOML 1.1 zero date for a local time). There is no implicit conversion between the offset
(`07:32`, `1979-05-27T07:32`); such a value carries a zero second, and the and local kinds; assigning one to the other is an error.
canonical rendering writes full seconds. There is no implicit conversion
between the offset and local kinds; assigning one to the other is an error.
### Arrays of tables ### Arrays of tables
@@ -193,8 +179,7 @@ A typo such as `database_urls` then fails with
`interpres: unknown field "database_urls" for main.Config` instead of a silent `interpres: unknown field "database_urls" for main.Config` instead of a silent
default-zero run. Strictness applies to every struct the decode reaches, at any default-zero run. Strictness applies to every struct the decode reaches, at any
depth, including struct elements inside slices; map destinations accept every depth, including struct elements inside slices; map destinations accept every
key by nature. When several keys are unknown, the message names the smallest key by nature.
one, so it does not depend on map iteration order.
### Cancellation ### Cancellation
@@ -248,33 +233,10 @@ Keys that match `[A-Za-z0-9_-]+` are emitted bare, all others quoted. A
`map[string]V` emits its keys in sorted order for deterministic output, and a `map[string]V` emits its keys in sorted order for deterministic output, and a
nil map emits nothing. nil map emits nothing.
### Tag options Note the asymmetry: the encoder inlines untagged embedded structs, while the
decoder expects them under their lower-cased type name. A struct with an
The part of a `toml` tag after the first comma carries options. Both options untagged embedded struct therefore does not round-trip through `Unmarshal` into
shape emission only; the decoder ignores them. the same type.
- `omitzero` skips the field when its value is the zero value of its type. A
type with an `IsZero() bool` method (time.Time among them) decides through
that method, so a zero `time.Time` or an all-zero struct disappears from
the output.
- `omitempty` skips the field when it holds an empty collection: a nil or
empty slice or array, or a nil or empty map. Strings and other scalars are
not covered by `omitempty`; use `omitzero` for those.
```go
type Config struct {
Host string `toml:"host,omitzero"`
Started time.Time `toml:"started,omitzero"`
Tags []string `toml:"tags,omitempty"`
}
```
Options combine after the name: `toml:"name,omitempty,omitzero"` is valid, and
an unknown option is ignored.
Untagged embedded fields round-trip: the decoder inlines embedded structs and
routes unclaimed keys into an embedded map exactly where the encoder flattened
them.
### Group-by-kind layout ### Group-by-kind layout
@@ -317,10 +279,7 @@ The returned value is encoded as if it had been passed in place of the
receiver, so it may be a scalar, a slice, an array of tables, or another receiver, so it may be a scalar, a slice, an array of tables, or another
struct or map, including the `Marshaler` result of another type; the encoder struct or map, including the `Marshaler` result of another type; the encoder
recurses. An error returned from `MarshalTOML` fails the marshal wrapped with recurses. An error returned from `MarshalTOML` fails the marshal wrapped with
the key path, for example `interpres: server.port: bad timestamp`. A result the key path, for example `interpres: server.port: bad timestamp`.
of `nil` with a nil error fails the same way with
`MarshalTOML returned a nil value`: nil has no TOML representation, so
dropping the field silently is not an option.
```go ```go
type Port int type Port int
@@ -330,23 +289,6 @@ func (p Port) MarshalTOML() (any, error) {
} }
``` ```
### Arrays
An array whose every element is a table (`[]struct`, `[]map[string]V`, after
pointer dereference) emits as an array of tables. TOML also lets one array mix
tables with scalars; such an array emits as a plain value array, with the
table elements rendered as inline tables:
```go
tree, _ := interpres.Parse([]byte(`arr = [1, {a = 2}, "x"]`))
out, _ := interpres.Marshal(tree) // arr = [1, {a = 2}, "x"]
```
A `[]any` holding only tables keeps the value-array form as well, because that
is the shape `Parse` gives a value array of inline tables; emitting it as
`[[headers]]` would re-parse as `[]map[string]any` and change the value's type
across a round-trip.
### Empty arrays ### Empty arrays
A nil slice is always omitted. An empty (length 0) array of tables is always A nil slice is always omitted. An empty (length 0) array of tables is always
@@ -367,10 +309,7 @@ out, err := interpres.NewEncoder().UseLiteralMultiline(80).Marshal(cfg)
``` ```
Single-line strings keep the basic form regardless of the threshold, and a Single-line strings keep the basic form regardless of the threshold, and a
threshold of `0` or less disables the option. A string the literal form cannot threshold of `0` or less disables the option.
carry verbatim (an embedded run of three single quotes, a control character
other than tab or newline, or a carriage return outside a CRLF pair) also keeps
the basic form, so the output always re-parses to the same value.
### Cancellation ### Cancellation
@@ -423,27 +362,6 @@ if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
} }
``` ```
### `type DecodeError struct{ Path []string; Err error }`
Wraps a decoding failure with the key path at which it happened. `Path` lists
one segment per level from the document root, the outermost key first: a key
contributes its name, an array element its bracketed index, so the path of the
`weight` field in the first item reads `["items", "[0]", "weight"]`. The
rendered message is unchanged by the type; read the fields instead of parsing
the message:
```go
if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
fmt.Println(de.Path, de.Err)
}
```
### `type EncodeError struct{ Path string; Err error }`
Wraps an encoding failure with the key path of the value that failed, in the
document's own notation: `server.ports[2]`. Read it with `errors.AsType` the
same way.
### `type Decoder` ### `type Decoder`
Configurable strictness for decoding, constructed with `NewDecoder`. Set up Configurable strictness for decoding, constructed with `NewDecoder`. Set up
@@ -500,14 +418,11 @@ types are produced by `Parse` and accepted by `Marshal`.
The entry points return: The entry points return:
- `*SyntaxError` for a malformed document, with the 1-based line - `*SyntaxError` for a malformed document, with the 1-based line
- `*DecodeError` for a decoding failure, with the key path in `Path` - a plain error for everything else: a non-pointer decode target, a type
- `*EncodeError` for an encoding failure, with the key path in `Path` mismatch, an overflow, a marshal policy violation, a cancelled context
- a plain error for the rest: a non-pointer decode target, a cancelled
context, a key that is not valid UTF-8
Decode and encode failures carry the key path or element index in the typed Decode and encode failures are wrapped with the key path or element index using
wrappers above, so `errors.Is` and `errors.AsType` see through them and the `fmt.Errorf`, so `errors.Is` and `errors.AsType` see through them.
path reads from a field instead of the message text.
## Notes ## Notes
+7 -13
View File
@@ -6,9 +6,9 @@ source tree; nothing is aspirational.
## Overview ## Overview
interpres is one public library package, one command, and one example. The interpres is one public library package, one command, and one example. The
library implements the whole of TOML 1.0 and 1.1, decoding and encoding, in the library implements the whole of TOML 1.0, decoding and encoding, in the
standard library alone; the command wraps the parser for the toml-test standard library alone; the command wraps the parser for the toml-test
compliance harness, against which it stands at 214 valid and 467 invalid cases compliance harness, against which it stands at 185 valid and 371 invalid cases
with zero failures; the example demonstrates the API. with zero failures; the example demonstrates the API.
```mermaid ```mermaid
@@ -43,7 +43,7 @@ Inside the library package, one file owns one concern:
| File | Responsibility | | File | Responsibility |
|---|---| |---|---|
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 and 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. | | `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 (table redefinitions, dotted keys, arrays of tables). Reports a 1-based line on failure. |
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. | | `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. | | `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. | | `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
@@ -99,17 +99,11 @@ sequenceDiagram
`Decode`, `DecodeContext`, `Marshal` and `MarshalContext` call allocates its `Decode`, `DecodeContext`, `Marshal` and `MarshalContext` call allocates its
own unexported worker, so a configured type is safe for concurrent use; the own unexported worker, so a configured type is safe for concurrent use; the
setter methods are not, and must finish before the value is shared. setter methods are not, and must finish before the value is shared.
- The parser is allocated per `ParseContext` call; the parser itself caches - The parser is allocated per `ParseContext` call; nothing is cached between
nothing between documents. documents.
- The one piece of shared state is the struct-schema cache in `decode.go`: a
`sync.Map` keyed by `reflect.Type`, holding the flattened field layout the
decoder and the encoder both consult. A schema is immutable once published,
so concurrent callers only race to build an identical value, the same
trade-off `encoding/json`'s field cache makes. The cache grows with the
number of distinct struct types, never with document size.
- The date-time wrappers are values, not pointers, and are immutable in use. - The date-time wrappers are values, not pointers, and are immutable in use.
- Nothing in the library starts goroutines; apart from the schema cache above, - Nothing in the library starts goroutines or holds locks; concurrency safety
which never mutates a published entry, there is no shared mutable state. comes from having no shared mutable state.
## Dependencies ## Dependencies
-44
View File
@@ -1,44 +0,0 @@
# Benchmarking
How the performance numbers attached to this project are measured, so that a
number in a changelog entry or a release note can be reproduced and trusted.
## The suite
The benchmarks live in `bench_test.go`, next to the code they measure:
| Benchmark | What it measures |
|---|---|
| `BenchmarkParse` | `Parse` over a representative configuration document |
| `BenchmarkMarshal` | `Marshal` of the tree `Parse` produced from the same document |
| `BenchmarkStrictDecode` | `Decode` into a struct under `DisallowUnknownFields` |
| `BenchmarkParseLong` | `Parse` over a generated document with about 2000 array-of-tables entries |
## Running
```sh
just bench
```
The recipe runs the suite with `-benchmem -count=5`. Every benchmark uses
`b.Loop`, so setup runs outside the timed region, and `ReportAllocs` records
allocations per operation. The parse and marshal benchmarks set `SetBytes`, so
their results read as input bytes per second.
## Method
- An idle machine only: a loaded box times whatever else is running, and the
fastest sample can land on the wrong function.
- An A/B comparison runs both variants inside one process, in one binary;
separate processes of identical binaries differ by more than the effect
being measured.
- The five counts are compared through their medians, allocations and bytes
per operation alongside the times. Differences within 1 to 2 percent are
noise; only a difference beyond that is a result.
- When timing is hopeless, the allocation and byte counts are the result.
## Reports
The repository stores no benchmark reports. A performance claim in
`CHANGELOG.md` is measured with the method above on the change that makes it,
and the number travels with the claim.
+13 -47
View File
@@ -1,45 +1,26 @@
# Command line # Command line
The reference below is taken from the program itself. `interpres-decode` is The reference below is taken from the program itself. `interpres-decode` is the
the toml-test harness adapter, and it also validates documents. Install it toml-test harness adapter, not a general-purpose tool: it takes no flags and no
with Go itself, no release assets involved: arguments, reads one TOML document from stdin, and writes the toml-test
tagged-JSON form to stdout.
```sh
go install sourcedock.dev/petrbalvin/interpres/cmd/interpres-decode@latest
```
## Synopsis ## Synopsis
```sh ```sh
interpres-decode [flags] interpres-decode < document.toml
interpres-decode -validate [file ...]
``` ```
Without `-validate` the program is the toml-test adapter: it takes no Build it with `just build`, which compiles it into `bin/interpres-decode`, or
arguments, reads one TOML document from stdin, and writes the toml-test run it straight from the module directory with `just run`.
tagged-JSON form to stdout. Build it locally with `just build`, which
compiles it into `bin/interpres-decode`, or run it straight from the module
directory with `just run`.
With `-validate` the program parses each named file instead, or stdin when no
file is named, and prints one line per invalid document to stderr. It is
quiet on valid documents, which is the shape a CI step wants. The `-` name
means stdin.
## Flags
| Flag | Effect |
|---|---|
| `-validate` | validate the documents instead of emitting tagged JSON |
| `-h` | print the usage |
## Exit codes ## Exit codes
| Code | Meaning | | Code | Meaning |
|---|---| |---|---|
| `0` | adapter: the document parsed and the tagged JSON was written; validate: every document parsed | | `0` | the document parsed, tagged JSON written to stdout |
| `1` | adapter: parse error; validate: at least one document is invalid | | `1` | parse error, the document is malformed; the message goes to stderr |
| `2` | a usage error, a read failure, or a value with no tagged representation | | `2` | reading stdin failed, or a value has no tagged representation |
## Wire format ## Wire format
@@ -78,28 +59,13 @@ port = 9090
' | ./bin/interpres-decode ' | ./bin/interpres-decode
``` ```
The output is the equivalent value tree as one JSON object. Validate the The output is the equivalent value tree as one JSON object. Run the official
TOML files of another repository in CI: compliance suite against the binary:
```sh
interpres-decode -validate config.toml deploy/example.toml
```
An invalid document reports the file and the library's line number:
```sh
$ interpres-decode -validate bad.toml
bad.toml: interpres: line 1: expected a value
$ echo $?
1
```
Run the official compliance suite against the adapter:
```sh ```sh
just toml-test just toml-test
``` ```
That recipe needs the `toml-test` binary on `PATH`, installed with That recipe needs the `toml-test` binary on `PATH`, installed with
`go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0`. The full `go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0`. The full
reference for the library itself is [API.md](API.md). reference for the library itself is [API.md](API.md).
+7 -10
View File
@@ -7,7 +7,7 @@ How to work on interpres.
- Go 1.27.1, the version the `go` directive in `go.mod` declares. - Go 1.27.1, the version the `go` directive in `go.mod` declares.
- [just](https://github.com/casey/just) for the recipes. - [just](https://github.com/casey/just) for the recipes.
- The `toml-test` binary on `PATH` for the compliance recipe, installed with - The `toml-test` binary on `PATH` for the compliance recipe, installed with
`go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0`. `go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0`.
The module has no third-party dependencies, so there is nothing else to fetch. The module has no third-party dependencies, so there is nothing else to fetch.
@@ -77,9 +77,7 @@ just bench
``` ```
Benchmark on an idle machine, and compare only runs made in one process against Benchmark on an idle machine, and compare only runs made in one process against
each other. The recipe sweeps `./...` five times with `-benchmem`. The binding each other. The recipe sweeps `./...` five times with `-benchmem`.
measurement method, and what counts as a result, is in
[docs/BENCHMARKING.md](BENCHMARKING.md).
## Debugging the build ## Debugging the build
@@ -99,13 +97,12 @@ pipeline.
|---|---|---| |---|---|---|
| `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite | | `test.yml` | push or pull request to `development` | format check, vet, modernisation, build, the test suite with the 80 percent coverage floor, then the toml-test compliance suite |
| `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally | | `race.yml` | `workflow_dispatch`, by hand | the suite under the race detector; the same race gate `just gates` runs locally |
| `release.yml` | a `v*` tag | tag validation, then format, vet, modernisation, build and the test suite with the coverage floor, then the Gitea release from the CHANGELOG section. No race detector: race never runs on a push path, and the local `just gates` raced the tree before the tag was cut | | `release.yml` | a `v*` tag | the same gates plus the race detector, then the Gitea release from the CHANGELOG section |
## Releases ## Releases
Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The Releases are cut by merging `development` into `main` and tagging `vX.Y.Z`. The
tag drives the release workflow: it validates the tag, runs the static gates tag drives the release workflow: it validates the tag, runs the full gate set
and the test suite with the coverage floor, extracts the matching `## [X.Y.Z]` including the race detector, extracts the matching `## [X.Y.Z]` section from
section from `CHANGELOG.md`, and publishes the release with that section as its `CHANGELOG.md`, and publishes the release with that section as its body. A
body. A library ships no binaries, so the release carries the notes and nothing library ships no binaries, so the release carries the notes and nothing else.
else.
+29 -222
View File
@@ -6,9 +6,7 @@ package interpres
import ( import (
"bytes" "bytes"
"context" "context"
"errors"
"fmt" "fmt"
"maps"
"math" "math"
"reflect" "reflect"
"slices" "slices"
@@ -144,16 +142,6 @@ func (d *tomlDoc) partitionedEntries() (scalars []entry, tables []entry, arrays
// --- reflection walk: struct --------------------------------------------- // --- reflection walk: struct ---------------------------------------------
func buildStructDoc(v reflect.Value, doc *tomlDoc, ctx string) error { func buildStructDoc(v reflect.Value, doc *tomlDoc, ctx string) error {
return walkStructDoc(v, doc, ctx, nil, cachedStructSchema(v.Type()))
}
// walkStructDoc emits the fields of v into doc. prefix is v's index path from
// the struct whose schema resolves key conflicts; an embedded struct is walked
// with the outer schema and a longer prefix, so every leaf competes under the
// decoder's rule: the shallower field wins, the later declaration at equal
// depth. A field another field shadows is skipped, because emitting both
// would duplicate the key and the output would not re-parse.
func walkStructDoc(v reflect.Value, doc *tomlDoc, ctx string, prefix []int, schema structSchema) error {
t := v.Type() t := v.Type()
for i := range t.NumField() { for i := range t.NumField() {
if i%ctxCheckInterval == 0 { if i%ctxCheckInterval == 0 {
@@ -165,7 +153,6 @@ func walkStructDoc(v reflect.Value, doc *tomlDoc, ctx string, prefix []int, sche
if f.PkgPath != "" { if f.PkgPath != "" {
continue continue
} }
path := append(append([]int{}, prefix...), i)
if f.Anonymous { if f.Anonymous {
tag, _ := f.Tag.Lookup("toml") tag, _ := f.Tag.Lookup("toml")
if tag == "-" { if tag == "-" {
@@ -180,15 +167,12 @@ func walkStructDoc(v reflect.Value, doc *tomlDoc, ctx string, prefix []int, sche
case reflect.Struct: case reflect.Struct:
if isScalarStruct(fv.Type()) { if isScalarStruct(fv.Type()) {
name := strings.ToLower(f.Name) name := strings.ToLower(f.Name)
if !schema.ownsKey(name, path) {
continue
}
if err := doc.appendScalar(name, fv.Interface(), ctx); err != nil { if err := doc.appendScalar(name, fv.Interface(), ctx); err != nil {
return err return err
} }
continue continue
} }
if err := walkStructDoc(fv, doc, ctx, path, schema); err != nil { if err := buildStructDoc(fv, doc, ctx); err != nil {
return err return err
} }
continue continue
@@ -204,12 +188,6 @@ func walkStructDoc(v reflect.Value, doc *tomlDoc, ctx string, prefix []int, sche
if name == "-" { if name == "-" {
continue continue
} }
if !schema.ownsKey(strings.ToLower(name), path) {
continue
}
if fieldOmitted(f, v.Field(i)) {
continue
}
if err := addField(doc, name, v.Field(i), ctx); err != nil { if err := addField(doc, name, v.Field(i), ctx); err != nil {
return err return err
} }
@@ -217,49 +195,6 @@ func walkStructDoc(v reflect.Value, doc *tomlDoc, ctx string, prefix []int, sche
return nil return nil
} }
// isZeroer mirrors encoding/json's omitzero: a type that knows its own zero
// state decides through that method before reflection is consulted.
type isZeroer interface{ IsZero() bool }
// fieldOmitted reports whether the field's tag options drop it from the
// output: omitzero skips the zero value of the field's type, omitempty skips
// an empty collection (slice, array, or map). The decoder ignores both
// options; they shape emission only.
func fieldOmitted(f reflect.StructField, v reflect.Value) bool {
tag, ok := f.Tag.Lookup("toml")
if !ok {
return false
}
_, opts, _ := strings.Cut(tag, ",")
for opts != "" {
var opt string
opt, opts, _ = strings.Cut(opts, ",")
switch opt {
case "omitzero":
if isZeroValue(v) {
return true
}
case "omitempty":
switch v.Kind() {
case reflect.Slice, reflect.Array, reflect.Map:
if v.Len() == 0 {
return true
}
}
}
}
return false
}
func isZeroValue(v reflect.Value) bool {
if v.CanInterface() {
if z, ok := v.Interface().(isZeroer); ok {
return z.IsZero()
}
}
return v.IsZero()
}
// fieldName returns the TOML key for a struct field, honouring the `toml` // fieldName returns the TOML key for a struct field, honouring the `toml`
// tag (name or `-`) and falling back to a lower-cased field name. // tag (name or `-`) and falling back to a lower-cased field name.
func fieldName(f reflect.StructField) string { func fieldName(f reflect.StructField) string {
@@ -300,21 +235,12 @@ func buildMapDoc(v reflect.Value, doc *tomlDoc, ctx string) error {
// --- reflection walk: field dispatch ------------------------------------- // --- reflection walk: field dispatch -------------------------------------
// errNilMarshalTOML reports a Marshaler whose method returned a nil value
// with no error. nil has no TOML representation, so dropping the field
// silently or panicking on the invalid reflect.Value would both hide the
// contract violation.
var errNilMarshalTOML = errors.New("MarshalTOML returned a nil value")
func addField(doc *tomlDoc, name string, v reflect.Value, ctx string) error { func addField(doc *tomlDoc, name string, v reflect.Value, ctx string) error {
if v.CanInterface() { if v.CanInterface() {
if m, ok := v.Interface().(Marshaler); ok { if m, ok := v.Interface().(Marshaler); ok {
mv, err := m.MarshalTOML() mv, err := m.MarshalTOML()
if err != nil { if err != nil {
return &EncodeError{Path: joinKey(ctx, name), Err: err} return fmt.Errorf("interpres: %s.%s: %w", ctx, name, err)
}
if mv == nil {
return &EncodeError{Path: joinKey(ctx, name), Err: errNilMarshalTOML}
} }
v = reflect.ValueOf(mv) v = reflect.ValueOf(mv)
} }
@@ -387,24 +313,7 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
return doc.appendScalar(name, []any{}, ctx) return doc.appendScalar(name, []any{}, ctx)
} }
// An array keeps the [[header]] form only when every element is a table. if isTableElementValue(v.Index(0)) {
// TOML lets one array mix tables with scalars, and that mix renders as a
// value array with the table elements written inline.
allTables := true
for i := range n {
if !isTableElementValue(v.Index(i)) {
allTables = false
break
}
}
// A []any of tables is what Parse produces for a value array of inline
// tables; the [[header]] form would re-parse as []map[string]any and so
// change the value's Go type across a round-trip. The header form is
// reserved for typed table slices.
if v.Type().Elem().Kind() == reflect.Interface {
allTables = false
}
if allTables {
subs := make([]*tomlDoc, n) subs := make([]*tomlDoc, n)
for i := range n { for i := range n {
if i%ctxCheckInterval == 0 { if i%ctxCheckInterval == 0 {
@@ -414,13 +323,13 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
} }
ev := followPtr(v.Index(i)) ev := followPtr(v.Index(i))
if !ev.IsValid() { if !ev.IsValid() {
return &EncodeError{Path: fmt.Sprintf("%s[%d]", joinKey(ctx, name), i), Err: errors.New("nil element")} return fmt.Errorf("interpres: %s.%s[%d]: nil element", ctx, name, i)
} }
sub := &tomlDoc{ctx: doc.ctx, opts: doc.opts} sub := &tomlDoc{ctx: doc.ctx, opts: doc.opts}
switch ev.Kind() { switch ev.Kind() {
case reflect.Struct: case reflect.Struct:
if isScalarStruct(ev.Type()) { if isScalarStruct(ev.Type()) {
return &EncodeError{Path: fmt.Sprintf("%s[%d]", joinKey(ctx, name), i), Err: errors.New("heterogeneous array contains scalar")} return fmt.Errorf("interpres: %s.%s[%d]: heterogeneous array contains scalar", ctx, name, i)
} }
if err := buildStructDoc(ev, sub, joinKey(ctx, fmt.Sprintf("%s[%d]", name, i))); err != nil { if err := buildStructDoc(ev, sub, joinKey(ctx, fmt.Sprintf("%s[%d]", name, i))); err != nil {
return err return err
@@ -430,7 +339,7 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
return err return err
} }
default: default:
return &EncodeError{Path: joinKey(ctx, name), Err: errors.New("heterogeneous array, expected table")} return fmt.Errorf("interpres: %s.%s: heterogeneous array, expected table", ctx, name)
} }
subs[i] = sub subs[i] = sub
} }
@@ -438,8 +347,7 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
return nil return nil
} }
// Value array. Table elements normalise to map[string]any and the emitter // Regular array of scalars.
// writes them as inline tables.
items := make([]any, n) items := make([]any, n)
for i := range n { for i := range n {
if i%ctxCheckInterval == 0 { if i%ctxCheckInterval == 0 {
@@ -449,16 +357,13 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
} }
ev := followPtr(v.Index(i)) ev := followPtr(v.Index(i))
if !ev.IsValid() { if !ev.IsValid() {
return &EncodeError{Path: fmt.Sprintf("%s[%d]", joinKey(ctx, name), i), Err: errors.New("nil element")} return fmt.Errorf("interpres: %s.%s[%d]: nil element", ctx, name, i)
} }
if ev.CanInterface() { if ev.CanInterface() {
if m, ok := ev.Interface().(Marshaler); ok { if m, ok := ev.Interface().(Marshaler); ok {
mv, err := m.MarshalTOML() mv, err := m.MarshalTOML()
if err != nil { if err != nil {
return &EncodeError{Path: fmt.Sprintf("%s[%d]", joinKey(ctx, name), i), Err: err} return fmt.Errorf("interpres: %s.%s[%d]: %w", ctx, name, i, err)
}
if mv == nil {
return &EncodeError{Path: fmt.Sprintf("%s[%d]", joinKey(ctx, name), i), Err: errNilMarshalTOML}
} }
ev = reflect.ValueOf(mv) ev = reflect.ValueOf(mv)
ev = followPtr(ev) ev = followPtr(ev)
@@ -466,7 +371,7 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
} }
val, err := normaliseValue(ev) val, err := normaliseValue(ev)
if err != nil { if err != nil {
return &EncodeError{Path: fmt.Sprintf("%s[%d]", joinKey(ctx, name), i), Err: err} return fmt.Errorf("interpres: %s.%s[%d]: %w", ctx, name, i, err)
} }
items[i] = val items[i] = val
} }
@@ -477,29 +382,11 @@ func addArrayValue(doc *tomlDoc, name string, v reflect.Value, ctx string) error
// nested-array representations the emitter understands. Slices and arrays are // nested-array representations the emitter understands. Slices and arrays are
// recursively normalised so that nested arrays (e.g. [][]int) work. // recursively normalised so that nested arrays (e.g. [][]int) work.
func normaliseValue(v reflect.Value) (any, error) { func normaliseValue(v reflect.Value) (any, error) {
// Map and slice elements arrive wrapped in interface{}; look through them.
for v.Kind() == reflect.Interface && !v.IsNil() {
v = v.Elem()
}
if v.Kind() == reflect.Interface {
return nil, fmt.Errorf("cannot encode nil value")
}
if v.CanInterface() { if v.CanInterface() {
if m, ok := v.Interface().(Marshaler); ok { if m, ok := v.Interface().(Marshaler); ok {
mv, err := m.MarshalTOML() return m.MarshalTOML()
if err != nil {
return nil, err
}
if mv == nil {
return nil, errNilMarshalTOML
}
return mv, nil
} }
} }
// The datetime structs are TOML scalars; the emitter renders each of them.
if t := v.Type(); t == timeGoType || isLocalDateType(t) {
return v.Interface(), nil
}
switch v.Kind() { switch v.Kind() {
case reflect.String: case reflect.String:
return v.String(), nil return v.String(), nil
@@ -515,21 +402,6 @@ func normaliseValue(v reflect.Value) (any, error) {
return int64(u), nil return int64(u), nil
case reflect.Float32, reflect.Float64: case reflect.Float32, reflect.Float64:
return v.Float(), nil return v.Float(), nil
case reflect.Map:
// A table nested in a value array has no header form, so it renders
// inline; the keys normalise to strings for the emitter.
if v.Type().Key().Kind() != reflect.String {
return nil, fmt.Errorf("map key must be string, got %s", v.Type().Key())
}
out := make(map[string]any, v.Len())
for _, k := range v.MapKeys() {
val, err := normaliseValue(v.MapIndex(k))
if err != nil {
return nil, fmt.Errorf("[%s]: %w", k.String(), err)
}
out[k.String()] = val
}
return out, nil
case reflect.Slice, reflect.Array: case reflect.Slice, reflect.Array:
items := make([]any, v.Len()) items := make([]any, v.Len())
for i := range v.Len() { for i := range v.Len() {
@@ -622,9 +494,7 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
path := append(append([]string{}, prefix...), t.key) path := append(append([]string{}, prefix...), t.key)
e.writeBlankLine() e.writeBlankLine()
e.buf.WriteByte('[') e.buf.WriteByte('[')
if err := e.writeKeyPath(path); err != nil { writeKeyPath(&e.buf, path)
return err
}
e.buf.WriteString("]\n") e.buf.WriteString("]\n")
if err := e.emitDoc(t.doc, path); err != nil { if err := e.emitDoc(t.doc, path); err != nil {
return err return err
@@ -635,9 +505,7 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
for _, sub := range a.docs { for _, sub := range a.docs {
e.writeBlankLine() e.writeBlankLine()
e.buf.WriteString("[[") e.buf.WriteString("[[")
if err := e.writeKeyPath(path); err != nil { writeKeyPath(&e.buf, path)
return err
}
e.buf.WriteString("]]\n") e.buf.WriteString("]]\n")
if err := e.emitDoc(sub, path); err != nil { if err := e.emitDoc(sub, path); err != nil {
return err return err
@@ -662,9 +530,7 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
path := append(append([]string{}, prefix...), ent.key) path := append(append([]string{}, prefix...), ent.key)
e.writeBlankLine() e.writeBlankLine()
e.buf.WriteByte('[') e.buf.WriteByte('[')
if err := e.writeKeyPath(path); err != nil { writeKeyPath(&e.buf, path)
return err
}
e.buf.WriteString("]\n") e.buf.WriteString("]\n")
if err := e.emitDoc(ent.doc, path); err != nil { if err := e.emitDoc(ent.doc, path); err != nil {
return err return err
@@ -674,9 +540,7 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
for _, sub := range ent.docs { for _, sub := range ent.docs {
e.writeBlankLine() e.writeBlankLine()
e.buf.WriteString("[[") e.buf.WriteString("[[")
if err := e.writeKeyPath(path); err != nil { writeKeyPath(&e.buf, path)
return err
}
e.buf.WriteString("]]\n") e.buf.WriteString("]]\n")
if err := e.emitDoc(sub, path); err != nil { if err := e.emitDoc(sub, path); err != nil {
return err return err
@@ -688,9 +552,10 @@ func (e *encoder) emitDoc(doc *tomlDoc, prefix []string) error {
} }
func (e *encoder) writeKV(key string, val any) error { func (e *encoder) writeKV(key string, val any) error {
if err := e.writeKey(key); err != nil { if !utf8.ValidString(key) {
return err return fmt.Errorf("interpres: key %q is not valid UTF-8", key)
} }
e.writeKey(key)
e.buf.WriteString(" = ") e.buf.WriteString(" = ")
if err := e.writeValue(val); err != nil { if err := e.writeValue(val); err != nil {
return err return err
@@ -699,30 +564,25 @@ func (e *encoder) writeKV(key string, val any) error {
return nil return nil
} }
func (e *encoder) writeKeyPath(path []string) error { func writeKeyPath(buf *bytes.Buffer, path []string) {
for i, p := range path { for i, p := range path {
if i > 0 { if i > 0 {
e.buf.WriteByte('.') buf.WriteByte('.')
} }
if err := e.writeKey(p); err != nil { if isBareKey(p) {
return err buf.WriteString(p)
continue
} }
writeQuotedString(buf, p)
} }
return nil
} }
// writeKey writes one key, bare when it qualifies and quoted otherwise. A key func (e *encoder) writeKey(key string) {
// that is not valid UTF-8 is an error; writing it anyway would emit corrupt
// TOML, because the quoted form has no representation for it.
func (e *encoder) writeKey(key string) error {
if isBareKey(key) { if isBareKey(key) {
e.buf.WriteString(key) e.buf.WriteString(key)
return nil return
} }
if !utf8.ValidString(key) { writeQuotedString(&e.buf, key)
return fmt.Errorf("interpres: key %q is not valid UTF-8", key)
}
return writeQuotedString(&e.buf, key)
} }
// writeQuotedString writes s as a TOML basic string (double-quoted) to buf. // writeQuotedString writes s as a TOML basic string (double-quoted) to buf.
@@ -823,8 +683,6 @@ func (e *encoder) writeValue(val any) error {
} }
e.buf.WriteByte(']') e.buf.WriteByte(']')
return nil return nil
case map[string]any:
return e.writeInlineTable(v)
case nil: case nil:
return fmt.Errorf("interpres: cannot encode nil value") return fmt.Errorf("interpres: cannot encode nil value")
default: default:
@@ -832,63 +690,13 @@ func (e *encoder) writeValue(val any) error {
} }
} }
// writeInlineTable renders m as a TOML inline table with sorted keys, the
// order buildMapDoc uses for header tables. It backs the table elements of a
// value array, where the [[header]] form is not available.
func (e *encoder) writeInlineTable(m map[string]any) error {
keys := slices.Sorted(maps.Keys(m))
e.buf.WriteByte('{')
for i, k := range keys {
if i > 0 {
e.buf.WriteString(", ")
}
if err := e.writeKey(k); err != nil {
return err
}
e.buf.WriteString(" = ")
if err := e.writeValue(m[k]); err != nil {
return err
}
}
e.buf.WriteByte('}')
return nil
}
func (e *encoder) writeStringVal(s string) error { func (e *encoder) writeStringVal(s string) error {
if e.opts.literalMultilineAt > 0 && strings.ContainsRune(s, '\n') && if e.opts.literalMultilineAt > 0 && strings.ContainsRune(s, '\n') && len(s) >= e.opts.literalMultilineAt {
len(s) >= e.opts.literalMultilineAt && canBeLiteralMultiline(s) {
return writeLiteralMultilineString(&e.buf, s) return writeLiteralMultilineString(&e.buf, s)
} }
return writeQuotedString(&e.buf, s) return writeQuotedString(&e.buf, s)
} }
// canBeLiteralMultiline reports whether s can be carried verbatim by the
// literal ”'...”' form: the form has no escapes, so a run of three single
// quotes would close the delimiter early, and control characters beyond tab,
// and a carriage return outside a CRLF pair, have no representation at all.
// Anything else falls back to the escaped basic string.
func canBeLiteralMultiline(s string) bool {
if strings.Contains(s, "'''") {
return false
}
for i := 0; i < len(s); {
r, size := utf8.DecodeRuneInString(s[i:])
switch {
case r == '\t' || r == '\n':
case r == '\r':
if !strings.HasPrefix(s[i+size:], "\n") {
return false
}
default:
if r < 0x20 || r == 0x7f {
return false
}
}
i += size
}
return true
}
// writeLiteralMultilineString writes s as a TOML literal multi-line string, // writeLiteralMultilineString writes s as a TOML literal multi-line string,
// surrounded by triple single quotes. The opening delimiter is followed by a // surrounded by triple single quotes. The opening delimiter is followed by a
// newline that the reader trims, so we always include one. The closing // newline that the reader trims, so we always include one. The closing
@@ -916,8 +724,7 @@ func (e *encoder) writeFloat(v float64) error {
case math.IsInf(v, -1): case math.IsInf(v, -1):
e.buf.WriteString("-inf") e.buf.WriteString("-inf")
case v == 0: case v == 0:
// Normalise negative zero to positive zero, the contract the output // Normalise negative zero to positive zero (TOML has no -0).
// rules in the documentation state.
e.buf.WriteString("0.0") e.buf.WriteString("0.0")
default: default:
s := strconv.FormatFloat(v, 'g', -1, 64) s := strconv.FormatFloat(v, 'g', -1, 64)
+1 -365
View File
@@ -67,7 +67,7 @@ func TestMarshalFloatSpecials(t *testing.T) {
} }
func TestMarshalFloatNormalizesNegativeZero(t *testing.T) { func TestMarshalFloatNormalizesNegativeZero(t *testing.T) {
// The output contract normalises negative zero to "0.0". // TOML has no -0; the emitter must normalise negative zero to "0.0".
type Cfg struct { type Cfg struct {
Z float64 `toml:"z"` Z float64 `toml:"z"`
} }
@@ -267,38 +267,6 @@ func TestEncoderUseLiteralMultilineThresholdZero(t *testing.T) {
} }
} }
func TestEncoderLiteralMultilineFallsBackWhenUnsafe(t *testing.T) {
// The literal form carries the value verbatim, so content it cannot
// represent must fall back to the escaped basic string instead of
// producing output that does not re-parse.
cases := []struct {
name string
in string
}{
{"embedded delimiter", "before ''' after\nsecond line"},
{"control character", "a\x01b\nsecond"},
{"delete character", "a\x7fb\nsecond"},
{"lone carriage return", "first\rsecond\nthird"},
}
for _, c := range cases {
out, err := NewEncoder().UseLiteralMultiline(5).Marshal(map[string]any{"s": c.in})
if err != nil {
t.Fatalf("%s: marshal: %v", c.name, err)
}
if !bytes.HasPrefix(out, []byte("s = \"")) {
t.Errorf("%s: expected the basic quoted form, got:\n%s", c.name, out)
}
re, err := Parse(out)
if err != nil {
t.Errorf("%s: re-parse: %v\ndoc:\n%s", c.name, err, out)
continue
}
if re["s"] != c.in {
t.Errorf("%s: round-trip changed the value: %q", c.name, re["s"])
}
}
}
// marshalerFunc adapts a plain function value to the Marshaler interface. // marshalerFunc adapts a plain function value to the Marshaler interface.
// Tests use it to express "this field produces this TOML value" without a // Tests use it to express "this field produces this TOML value" without a
// dedicated struct definition. // dedicated struct definition.
@@ -383,75 +351,6 @@ func TestMarshalerErrorPropagates(t *testing.T) {
} }
} }
// nilMarshalerFunc is a Marshaler whose method returns nil with no error.
type nilMarshalerFunc struct{}
func (nilMarshalerFunc) MarshalTOML() (any, error) { return nil, nil }
func TestMarshalRejectsNilMarshalerResult(t *testing.T) {
// nil has no TOML representation, so a MarshalTOML result of nil is an
// error, not a silently dropped field.
_, err := Marshal(struct {
F nilMarshalerFunc `toml:"f"`
}{})
if err == nil {
t.Fatal("expected an error for a nil MarshalTOML result")
}
ee, ok := errors.AsType[*EncodeError](err)
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "f" {
t.Fatalf("Path = %q, want %q", ee.Path, "f")
}
// Inside a value array the nil result used to reach reflection as a zero
// Value and panic.
_, err = Marshal(map[string]any{"arr": []any{1, nilMarshalerFunc{}}})
if err == nil {
t.Fatal("expected an error for a nil MarshalTOML result in an array")
}
if !strings.Contains(err.Error(), "MarshalTOML returned a nil value") {
t.Errorf("err = %v, want the nil-result message", err)
}
}
// Two fields that resolve to one TOML key must marshal as one key, resolved
// the way the decoder resolves it, or the output would carry a duplicate key
// and never re-parse.
func TestMarshalDuplicateKeyResolvesToOneField(t *testing.T) {
type SameLevel struct {
First int `toml:"v"`
Second string `toml:"v"`
}
out, err := Marshal(SameLevel{First: 1, Second: "s"})
if err != nil {
t.Fatalf("marshal: %v", err)
}
if want := "v = \"s\"\n"; string(out) != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
type Base struct {
Name string `toml:"name"`
}
type Embedded struct {
Base
Name string `toml:"name"`
}
out, err = Marshal(Embedded{Base: Base{Name: "inner"}, Name: "outer"})
if err != nil {
t.Fatalf("marshal: %v", err)
}
// The shallower field wins, matching the decoder.
if want := "name = \"outer\"\n"; string(out) != want {
t.Errorf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
if _, err := Parse(out); err != nil {
t.Errorf("re-parse: %v\ndoc:\n%s", err, out)
}
}
func TestMarshalEmbeddedScalarStruct(t *testing.T) { func TestMarshalEmbeddedScalarStruct(t *testing.T) {
// A field declared directly as a scalar-struct type (here LocalDateTime) // A field declared directly as a scalar-struct type (here LocalDateTime)
// must be encoded as a TOML scalar at the parent level, not rendered as // must be encoded as a TOML scalar at the parent level, not rendered as
@@ -605,109 +504,6 @@ func TestMarshalNestedArrays(t *testing.T) {
} }
} }
func TestMarshalMixedArrayWithInlineTable(t *testing.T) {
// Parse accepts a mixed array (TOML allows any value kinds in one array),
// so Marshal of the parsed tree must re-emit it. The table element has no
// header form inside a value array and renders inline.
tree, err := Parse([]byte("arr = [1, {a = 2}, \"x\"]\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
out, err := Marshal(tree)
if err != nil {
t.Fatalf("marshal: %v", err)
}
want := "arr = [1, {a = 2}, \"x\"]\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
re, err := Parse(out)
if err != nil {
t.Fatalf("re-parse: %v", err)
}
if !reflect.DeepEqual(tree, re) {
t.Fatalf("round-trip changed the tree:\nwas: %#v\nnow: %#v", tree, re)
}
}
// A []any of tables is what Parse produces for a value array of inline
// tables; it must stay in the value-array form, or the output would re-parse
// as []map[string]any and the round-trip would change the value's type.
func TestMarshalValueArrayOfTablesStaysInline(t *testing.T) {
for _, doc := range []string{
"0=[{}]",
"a = [{x = 1}, {x = 2}]\n",
"b = [{x = 1}, 2, \"three\"]\n",
} {
tree, err := Parse([]byte(doc))
if err != nil {
t.Fatalf("%s: parse: %v", doc, err)
}
out, err := Marshal(tree)
if err != nil {
t.Fatalf("%s: marshal: %v", doc, err)
}
if bytes.HasPrefix(out, []byte("[[")) {
t.Errorf("%s: emitted the [[header]] form for a value array:\n%s", doc, out)
}
re, err := Parse(out)
if err != nil {
t.Fatalf("%s: re-parse: %v\ndoc:\n%s", doc, err, out)
}
if !tomlEqual(tree, re) {
t.Errorf("%s: round-trip changed the tree:\nwas: %#v\nnow: %#v\ndoc:\n%s", doc, tree, re, out)
}
}
}
func TestMarshalNestedInlineTables(t *testing.T) {
tree := map[string]any{
"mix": []any{
int64(1),
map[string]any{"deep": map[string]any{"n": int64(0)}, "list": []any{"a", true}},
map[string]any{},
},
}
out, err := Marshal(tree)
if err != nil {
t.Fatalf("marshal: %v", err)
}
want := "mix = [1, {deep = {n = 0}, list = [\"a\", true]}, {}]\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalInlineTableWithDatetime(t *testing.T) {
when := time.Date(1979, 5, 27, 7, 32, 0, 0, time.UTC)
tree := map[string]any{
"mix": []any{when, map[string]any{"t": LocalDateTime{when}}},
}
out, err := Marshal(tree)
if err != nil {
t.Fatalf("marshal: %v", err)
}
want := "mix = [1979-05-27T07:32:00Z, {t = 1979-05-27T07:32:00}]\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalArrayOfTablesStaysHeaderForm(t *testing.T) {
tree, err := Parse([]byte("[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
out, err := Marshal(tree)
if err != nil {
t.Fatalf("marshal: %v", err)
}
want := "[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalFloatExponentNoLeadingZero(t *testing.T) { func TestMarshalFloatExponentNoLeadingZero(t *testing.T) {
// strconv.FormatFloat with 'g' would produce "1e+06" (leading zero in // strconv.FormatFloat with 'g' would produce "1e+06" (leading zero in
// exponent). The encoder must strip it so the output is "1e+6". // exponent). The encoder must strip it so the output is "1e+6".
@@ -879,85 +675,6 @@ func TestMarshalEmbeddedStructAsTable(t *testing.T) {
} }
} }
func TestMarshalTagOptionOmitZero(t *testing.T) {
type Server struct {
Host string `toml:"host"`
}
type Cfg struct {
Name string `toml:"name,omitzero"`
Count int `toml:"count,omitzero"`
Ratio float64 `toml:"ratio,omitzero"`
When time.Time `toml:"when,omitzero"`
Server Server `toml:"server,omitzero"`
Always string `toml:"always"`
}
out, err := Marshal(Cfg{Always: "kept"})
if err != nil {
t.Fatalf("marshal: %v", err)
}
// Every omitzero field sits at its zero value, so only always is emitted.
want := "always = \"kept\"\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
when := time.Date(2026, 9, 17, 12, 0, 0, 0, time.UTC)
out, err = Marshal(Cfg{Name: "x", Count: 1, Ratio: 0.5, When: when, Server: Server{Host: "h"}, Always: "kept"})
if err != nil {
t.Fatalf("marshal: %v", err)
}
want = "name = \"x\"\ncount = 1\nratio = 0.5\nwhen = 2026-09-17T12:00:00Z\nalways = \"kept\"\n\n[server]\nhost = \"h\"\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalTagOptionOmitEmpty(t *testing.T) {
type Cfg struct {
Tags []string `toml:"tags,omitempty"`
Ports []int `toml:"ports,omitempty"`
Matrix [][]int `toml:"matrix,omitempty"`
Extra map[string]any `toml:"extra,omitempty"`
Name string `toml:"name,omitempty"`
Keep []string `toml:"keep"`
}
out, err := Marshal(Cfg{
Ports: []int{},
Matrix: [][]int{{1}},
Extra: map[string]any{},
Name: "set",
Keep: []string{},
})
if err != nil {
t.Fatalf("marshal: %v", err)
}
// tags is nil (omitted anyway), ports and extra are empty collections
// dropped by omitempty, matrix is populated, name is a string the option
// does not cover, keep is empty but carries no option so it emits [].
want := "matrix = [[1]]\nname = \"set\"\nkeep = []\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalTagOptionOnTaggedEmbeddedStruct(t *testing.T) {
type Inner struct {
N int `toml:"n"`
}
type Cfg struct {
Inner Inner `toml:"inner,omitzero"`
Name string `toml:"name"`
}
out, err := Marshal(Cfg{Name: "x"})
if err != nil {
t.Fatalf("marshal: %v", err)
}
want := "name = \"x\"\n"
if string(out) != want {
t.Fatalf("output mismatch:\ngot: %q\nwant: %q", out, want)
}
}
func TestMarshalMapKeysSorted(t *testing.T) { func TestMarshalMapKeysSorted(t *testing.T) {
m := map[string]any{ m := map[string]any{
"zeta": 1, "zeta": 1,
@@ -1227,17 +944,6 @@ func TestMarshalKeyRequiresUTF8(t *testing.T) {
if _, err := Marshal(m); err == nil { if _, err := Marshal(m); err == nil {
t.Errorf("expected error for invalid UTF-8 key") t.Errorf("expected error for invalid UTF-8 key")
} }
// The check must reach the keys of table headers and of inline tables
// nested inside value arrays, not only scalar keys: both write keys
// through the same path.
nested := map[string]any{"\xff": map[string]any{"k": "v"}}
if _, err := Marshal(nested); err == nil {
t.Errorf("expected error for invalid UTF-8 table header key")
}
inline := map[string]any{"mix": []any{1, map[string]any{"\xff": 1}}}
if _, err := Marshal(inline); err == nil {
t.Errorf("expected error for invalid UTF-8 inline table key")
}
} }
func TestMarshalStringRequiresUTF8(t *testing.T) { func TestMarshalStringRequiresUTF8(t *testing.T) {
@@ -1302,73 +1008,3 @@ type Custom struct {
} }
func (c Custom) MarshalTOML() (any, error) { return c.tag, nil } func (c Custom) MarshalTOML() (any, error) { return c.tag, nil }
// encodeErrBad is a Marshaler whose MarshalTOML always fails.
type encodeErrBad struct {
msg string
}
func (encodeErrBad) MarshalTOML() (any, error) { return nil, errors.New("bad timestamp") }
func TestEncodeErrorCarriesPath(t *testing.T) {
type Inner struct {
Port encodeErrBad `toml:"port"`
}
type Cfg struct {
Server Inner `toml:"server"`
}
_, err := Marshal(Cfg{Server: Inner{Port: encodeErrBad{}}})
if err == nil {
t.Fatal("expected a marshal error")
}
ee, ok := errors.AsType[*EncodeError](err)
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "server.port" {
t.Fatalf("Path = %q, want %q", ee.Path, "server.port")
}
if ee.Err == nil || ee.Err.Error() != "bad timestamp" {
t.Fatalf("Err = %v", ee.Err)
}
if err.Error() != "interpres: server.port: bad timestamp" {
t.Fatalf("message = %q", err.Error())
}
}
func TestEncodeErrorTopLevelPathHasNoLeadingDot(t *testing.T) {
type Cfg struct {
Port encodeErrBad `toml:"port"`
}
_, err := Marshal(Cfg{})
ee, ok := errors.AsType[*EncodeError](err)
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "port" {
t.Fatalf("Path = %q, want %q", ee.Path, "port")
}
if err.Error() != "interpres: port: bad timestamp" {
t.Fatalf("message = %q", err.Error())
}
}
func TestEncodeErrorHeterogeneousArrayPath(t *testing.T) {
type Item struct {
N int `toml:"n"`
}
cfg := map[string]any{
"items": []any{Item{}, 3},
}
_, err := Marshal(cfg)
if err == nil {
t.Fatal("expected a heterogeneous array error")
}
ee, ok := errors.AsType[*EncodeError](err)
if !ok {
t.Fatalf("expected an *EncodeError, got %T: %v", err, err)
}
if ee.Path != "items[0]" {
t.Fatalf("Path = %q, want %q", ee.Path, "items[0]")
}
}
-123
View File
@@ -1,123 +0,0 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package interpres
import (
"math"
"reflect"
"testing"
"time"
)
// FuzzParse drives the parser with arbitrary input and holds it to the
// round-trip invariant: every document Parse accepts must survive its own
// re-emission. Marshal of the parsed tree must succeed, the emitted document
// must parse again, and the re-parsed tree must equal the original one.
func FuzzParse(f *testing.F) {
seeds := []string{
"",
"title = \"interpres\"\n",
"[server]\nhost = \"localhost\"\nport = 8080\n\n[server.tls]\nenabled = true\n",
"[[items]]\nname = \"a\"\n\n[[items]]\nname = \"b\"\n",
"inline = { a = 1, b = [2, 3], c = { d = 4 } }\n",
"arr = [1, 2.5, \"three\", true, 1979-05-27T07:32:00Z]\n",
"mix = [1, {a = 2}, \"x\"]\n",
"when = 1979-05-27T07:32:00Z\nlocal = 1979-05-27T07:32:00.999\nd = 1979-05-27\nt = 07:32:00\n",
"multi = \"\"\"\nlines\n\"\"\"\nlit = 'literal'\n",
"esc = \"\\u0000\\t\\n\\\"\\\\\"\n",
"neg = -0.0\nnan = nan\ninf = -inf\nexp = 1e6\n",
"\"quoted key\" = 'value'\n'1979-05-27' = 1\na.b.c = { d = \"dotted\" }\n",
"hex = 0xFF\noct = 0o755\nbin = 0b1010\nsep = 1_000_000\n",
"x = \"unterminated\n",
"[a]\n[a]\n",
"n = 0x1_0000_0000_0000_0000\n",
// TOML 1.1 forms.
"t = 13:37\ndt = 1979-05-27T07:32\nodt = 1979-05-27 07:32Z\n",
"esc = \"\\e\\x41\\x7f\\x00\"\n",
"m = {\n\ta = 1,\n\tb = [1, 2,],\n\tc = { d = 2 },\n} # close\n",
}
for _, s := range seeds {
f.Add([]byte(s))
}
f.Fuzz(func(t *testing.T, data []byte) {
tree, err := Parse(data)
if err != nil {
return
}
out, err := Marshal(tree)
if err != nil {
t.Fatalf("marshal of a parsed tree failed: %v\ntree: %#v", err, tree)
}
re, err := Parse(out)
if err != nil {
t.Fatalf("re-parse of the emitted document failed: %v\ndoc:\n%s", err, out)
}
if !tomlEqual(tree, re) {
t.Fatalf("round-trip changed the tree\ninput: %q\ndoc:\n%s\nwas: %#v\nnow: %#v", data, out, tree, re)
}
})
}
// tomlEqual reports whether two parsed trees are equal as TOML values. It
// differs from reflect.DeepEqual where DeepEqual is wrong for this domain:
// NaN compares equal to itself, date-times compare by their canonical TOML
// rendering so two parses of one document stay equal, and the local variants
// compare through their String form, which fully determines the value.
func tomlEqual(a, b any) bool {
switch av := a.(type) {
case nil:
return b == nil
case float64:
bv, ok := b.(float64)
return ok && (av == bv || (math.IsNaN(av) && math.IsNaN(bv)))
case time.Time:
bv, ok := b.(time.Time)
return ok && av.Format(time.RFC3339Nano) == bv.Format(time.RFC3339Nano)
case LocalDateTime:
bv, ok := b.(LocalDateTime)
return ok && av.String() == bv.String()
case LocalDate:
bv, ok := b.(LocalDate)
return ok && av.String() == bv.String()
case LocalTime:
bv, ok := b.(LocalTime)
return ok && av.String() == bv.String()
case []any:
bv, ok := b.([]any)
if !ok || len(av) != len(bv) {
return false
}
for i := range av {
if !tomlEqual(av[i], bv[i]) {
return false
}
}
return true
case []map[string]any:
bv, ok := b.([]map[string]any)
if !ok || len(av) != len(bv) {
return false
}
for i := range av {
if !tomlEqual(av[i], bv[i]) {
return false
}
}
return true
case map[string]any:
bv, ok := b.(map[string]any)
if !ok || len(av) != len(bv) {
return false
}
for k, v := range av {
other, ok := bv[k]
if !ok || !tomlEqual(v, other) {
return false
}
}
return true
default:
return reflect.DeepEqual(a, b)
}
}
+5 -66
View File
@@ -21,7 +21,6 @@ package interpres
import ( import (
"context" "context"
"errors"
"fmt" "fmt"
"unicode/utf8" "unicode/utf8"
) )
@@ -37,58 +36,6 @@ func (e *SyntaxError) Error() string {
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg) return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
} }
// A DecodeError wraps a decoding failure with the key path at which it
// happened. Path lists one segment per level from the document root, the
// outermost key first: a key contributes its name and an array element its
// bracketed index, so the path of the weight field in the first item reads
// ["items", "[0]", "weight"]. The rendered message is unchanged by the type;
// read it programmatically with errors.AsType:
//
// if de, ok := errors.AsType[*interpres.DecodeError](err); ok {
// fmt.Println(de.Path, de.Err)
// }
type DecodeError struct {
// Path is the key path from the document root, outermost key first.
Path []string
// Err is the failure at that path.
Err error
}
func (e *DecodeError) Error() string { return e.Path[0] + ": " + e.Err.Error() }
// Unwrap returns the failure the path points at.
func (e *DecodeError) Unwrap() error { return e.Err }
// newDecodeError wraps err with one path segment. The rest of the path comes
// from the DecodeError err already carries, if any: the decoder wraps each
// key and index on its way down, so the innermost wrap holds the deepest
// segments and each outer wrap prepends one.
func newDecodeError(key string, err error) *DecodeError {
path := make([]string, 0, 4)
path = append(path, key)
if de, ok := errors.AsType[*DecodeError](err); ok {
path = append(path, de.Path...)
}
return &DecodeError{Path: path, Err: err}
}
// An EncodeError wraps an encoding failure with the key path of the value
// that failed, in the notation of a TOML document: fields join with dots and
// an array element carries its bracketed index, so the path of the third
// port under server reads "server.ports[2]". The rendered message is
// unchanged by the type; read it programmatically with errors.AsType.
type EncodeError struct {
// Path is the key path of the failing value.
Path string
// Err is the failure at that path.
Err error
}
func (e *EncodeError) Error() string { return "interpres: " + e.Path + ": " + e.Err.Error() }
// Unwrap returns the failure the path points at.
func (e *EncodeError) Unwrap() error { return e.Err }
// Parse decodes a TOML document into a nested map[string]any. // Parse decodes a TOML document into a nested map[string]any.
// //
// Values are mapped to Go types as follows: strings to string, integers to // Values are mapped to Go types as follows: strings to string, integers to
@@ -110,9 +57,7 @@ func ParseContext(ctx context.Context, data []byte) (map[string]any, error) {
if !utf8.Valid(data) { if !utf8.Valid(data) {
return nil, &SyntaxError{Line: 1, Msg: "input is not valid UTF-8"} return nil, &SyntaxError{Line: 1, Msg: "input is not valid UTF-8"}
} }
// The parser scans data in place; it only reads the buffer, and every p := &parser{src: []rune(string(data)), line: 1, ctx: ctx}
// string it stores in the tree is copied out of it.
p := &parser{src: data, line: 1, ctx: ctx}
return p.parse() return p.parse()
} }
@@ -198,26 +143,20 @@ type Unmarshaler interface {
UnmarshalTOML(data any) error UnmarshalTOML(data any) error
} }
// Marshal returns the TOML encoding of v. The output stays within TOML 1.0, // Marshal returns the TOML 1.0 encoding of v.
// so it is valid under both TOML 1.0 and 1.1.
// //
// Marshal traverses v using reflection and applies the following rules: // Marshal traverses v using reflection and applies the following rules:
// //
// - The top-level value must be a struct or a map[string]V. Pointers are // - The top-level value must be a struct or a map[string]V. Pointers are
// followed; a nil top-level pointer is an error. // followed; a nil top-level pointer is an error.
// - Struct fields are matched by `toml:"name"` tag (case-insensitive // - Struct fields are matched by `toml:"name"` tag (case-insensitive
// fallback to field name; `-` skips). The tag options `omitzero` (skip // fallback to field name; `-` skips). Anonymous (embedded) fields without
// the zero value of the field's type) and `omitempty` (skip an empty // a tag are inlined.
// slice, array, or map) drop a field from the output on encode; the
// decoder ignores them. Anonymous (embedded) fields without a tag are
// inlined.
// - Maps use sorted keys for deterministic output. // - Maps use sorted keys for deterministic output.
// - Slices and arrays of structs or maps become TOML arrays of tables; a // - Slices and arrays of structs or maps become TOML arrays of tables; a
// nil or empty array of tables is omitted (TOML forbids an empty `[[a]]`), // nil or empty array of tables is omitted (TOML forbids an empty `[[a]]`),
// while other empty arrays emit as `key = []`. // while other empty arrays emit as `key = []`.
// - Other slices and arrays become TOML arrays; a table element inside a // - Other slices and arrays become TOML arrays.
// value array (for example an inline table in a mixed array) emits as an
// inline table.
// - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time // - Scalars encode as TOML scalars: bool, int64, float64, string, time.Time
// (offset date-time), and LocalDateTime/LocalDate/LocalTime (local // (offset date-time), and LocalDateTime/LocalDate/LocalTime (local
// variants). // variants).
+1 -191
View File
@@ -5,7 +5,6 @@ package interpres
import ( import (
"math" "math"
"strings"
"testing" "testing"
"time" "time"
) )
@@ -317,26 +316,6 @@ func TestDisallowUnknownFields(t *testing.T) {
} }
} }
func TestDisallowUnknownFieldsReportsSmallestKey(t *testing.T) {
// Map iteration order is random, so the reported key must be chosen
// deterministically: the smallest unknown key, whichever order the map
// iterates in.
type C struct {
Known string `toml:"known"`
}
data := []byte("known = \"x\"\nzeta = 1\nalpha = 2\nmu = 3\n")
for range 20 {
var c C
err := NewDecoder().DisallowUnknownFields().Decode(data, &c)
if err == nil {
t.Fatal("expected error for unknown fields")
}
if !strings.Contains(err.Error(), `unknown field "alpha"`) {
t.Fatalf("err = %v, want the smallest unknown key alpha", err)
}
}
}
func TestSkippedFieldTag(t *testing.T) { func TestSkippedFieldTag(t *testing.T) {
type C struct { type C struct {
Keep string `toml:"keep"` Keep string `toml:"keep"`
@@ -391,7 +370,6 @@ func TestRejectsInvalidNumbers(t *testing.T) {
"01", "-01", "00", "01", "-01", "00",
"1__0", "_1", "1_", "0x_1", "1_.0", "1__0", "_1", "1_", "0x_1", "1_.0",
"1.", ".5", "1.2.3", "1.e2", "1.", ".5", "1.2.3", "1.e2",
"1e", "1e+", "1e-", "0.0E", "0.0e", "1.5e+",
"0x", "0o", "0b", "0b2", "0o8", "0xG", "0x", "0o", "0b", "0b2", "0o8", "0xG",
"+0x1", "+0x1",
} { } {
@@ -401,34 +379,6 @@ func TestRejectsInvalidNumbers(t *testing.T) {
} }
} }
func TestParseRejectsOffsetOutOfRange(t *testing.T) {
for _, tok := range []string{
"1979-05-27T07:32:00+00:60",
"1979-05-27T07:32:00-00:99",
"1979-05-27T07:32:00+24:00",
"1979-05-27T07:32:00+99:99",
} {
if _, err := Parse([]byte("v = " + tok + "\n")); err == nil {
t.Errorf("%q: expected an error, got none", tok)
}
}
}
func TestParseAcceptsOffsetBounds(t *testing.T) {
tree, err := Parse([]byte("a = 1979-05-27T07:32:00+23:59\nb = 1979-05-27T07:32:00-23:59\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
a := tree["a"].(time.Time)
if _, offset := a.Zone(); offset != 23*3600+59*60 {
t.Fatalf("a offset = %d, want %d", offset, 23*3600+59*60)
}
b := tree["b"].(time.Time)
if _, offset := b.Zone(); offset != -(23*3600 + 59*60) {
t.Fatalf("b offset = %d", offset)
}
}
func TestAcceptsNumberEdgeCases(t *testing.T) { func TestAcceptsNumberEdgeCases(t *testing.T) {
cases := map[string]any{ cases := map[string]any{
"0": int64(0), "0": int64(0),
@@ -442,10 +392,6 @@ func TestAcceptsNumberEdgeCases(t *testing.T) {
"3.14": 3.14, "3.14": 3.14,
"6.022e23": 6.022e23, "6.022e23": 6.022e23,
"1e10": 1e10, "1e10": 1e10,
"1e0": 1.0,
"1e06": 1e6,
"0e00": 0.0,
"2E-3": 2e-3,
"-2.5E-3": -2.5e-3, "-2.5E-3": -2.5e-3,
} }
for tok, want := range cases { for tok, want := range cases {
@@ -512,9 +458,6 @@ func TestRejectsInlineTableExtension(t *testing.T) {
"by header": "a = { b = 1 }\n[a.c]\nx = 2\n", "by header": "a = { b = 1 }\n[a.c]\nx = 2\n",
"by dotted key": "a = { b = 1 }\na.c = 2\n", "by dotted key": "a = { b = 1 }\na.c = 2\n",
"header over it": "a = { b = 1 }\n[a]\nx = 2\n", "header over it": "a = { b = 1 }\n[a]\nx = 2\n",
// The frozen check must cover the intermediate steps of an array-of-tables
// header, not only the leaf: [[a.b.c]] walks through a and a.b.
"by nested array header": "a = { b = {} }\n[[a.b.c]]\nx = 2\n",
} }
for name, doc := range cases { for name, doc := range cases {
if _, err := Parse([]byte(doc)); err == nil { if _, err := Parse([]byte(doc)); err == nil {
@@ -523,36 +466,6 @@ func TestRejectsInlineTableExtension(t *testing.T) {
} }
} }
// A new element of an array of tables starts a fresh scope: sub-table headers,
// nested arrays of tables, and dotted-key paths recorded for the previous
// element must not block the same paths in the next one.
func TestArrayOfTablesFreshScopePerElement(t *testing.T) {
cases := map[string]string{
"nested array of tables": "[[a]]\n[[a.b]]\nx = 1\n[[a]]\n[a.b]\ny = 2\n",
"dotted key": "[[a]]\nb.c = 1\n[[a]]\n[a.b]\nd = 2\n",
}
for name, doc := range cases {
tree, err := Parse([]byte(doc))
if err != nil {
t.Errorf("%s: %v", name, err)
continue
}
elements := tree["a"].([]map[string]any)
if len(elements) != 2 {
t.Errorf("%s: len(a) = %d, want 2", name, len(elements))
}
}
// Within one element the redefinition rules keep applying.
for name, doc := range map[string]string{
"header over dotted in one element": "[[a]]\nb.c = 1\n[a.b]\nd = 2\n",
"table over nested array": "[[a]]\n[[a.b]]\n[a.b]\nx = 1\n",
} {
if _, err := Parse([]byte(doc)); err == nil {
t.Errorf("%s: expected an error, got none", name)
}
}
}
func TestRejectsSpecInvalid(t *testing.T) { func TestRejectsSpecInvalid(t *testing.T) {
cases := map[string]string{ cases := map[string]string{
"single-digit hour": "a = 2023-10-01T1:32:00Z\n", "single-digit hour": "a = 2023-10-01T1:32:00Z\n",
@@ -561,8 +474,7 @@ func TestRejectsSpecInvalid(t *testing.T) {
"dotted over header": "[a.b]\nx = 1\n[a]\nb.y = 2\n", "dotted over header": "[a.b]\nx = 1\n[a]\nb.y = 2\n",
"table over array": "[[t]]\n[t]\n", "table over array": "[[t]]\n[t]\n",
"truncated datetime": "a = 2026-01-02T\n", "truncated datetime": "a = 2026-01-02T\n",
// "datetime no seconds" moved to the acceptance tests: TOML 1.1 "datetime no seconds": "a = 2026-01-02T07:32\n",
// makes the seconds optional.
} }
for name, doc := range cases { for name, doc := range cases {
if _, err := Parse([]byte(doc)); err == nil { if _, err := Parse([]byte(doc)); err == nil {
@@ -599,105 +511,3 @@ host = "h2"
t.Errorf("forms[1].smtp.host = %v", h) t.Errorf("forms[1].smtp.host = %v", h)
} }
} }
// --- TOML 1.1 --------------------------------------------------------------
func TestParseAcceptsNoSecondsDatetimes(t *testing.T) {
tree, err := Parse([]byte(`t = 13:37
dt = 1979-05-27T07:32
odt1 = 1979-05-27 07:32Z
odt2 = 1979-05-27 07:32-07:00
`))
if err != nil {
t.Fatalf("parse: %v", err)
}
if got := tree["t"].(LocalTime).String(); got != "13:37:00" {
t.Errorf("t = %q, want %q", got, "13:37:00")
}
if got := tree["dt"].(LocalDateTime).String(); got != "1979-05-27T07:32:00" {
t.Errorf("dt = %q, want %q", got, "1979-05-27T07:32:00")
}
if got := tree["odt1"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00Z" {
t.Errorf("odt1 = %q", got)
}
if got := tree["odt2"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00-07:00" {
t.Errorf("odt2 = %q", got)
}
// The fraction still requires the seconds it belongs to.
if _, err := Parse([]byte("a = 07:32.5\n")); err == nil {
t.Error("07:32.5: expected an error, got none")
}
}
func TestParseAcceptsEscapeAndHexEscapes(t *testing.T) {
tree, err := Parse([]byte(`esc = "\e"
hex = "\x20\x7f\xf8"
nul = "\x00"
multi = """\x68\x65"""
lit = '\x20'
`))
if err != nil {
t.Fatalf("parse: %v", err)
}
if got := tree["esc"].(string); got != "\x1b" {
t.Errorf("esc = %q, want the escape character", got)
}
if got := tree["hex"].(string); got != " \x7f\u00f8" {
t.Errorf("hex = %q", got)
}
if got := tree["nul"].(string); got != "\x00" {
t.Errorf("nul = %q", got)
}
if got := tree["multi"].(string); got != "he" {
t.Errorf("multi = %q", got)
}
// A literal string carries the sequence verbatim.
if got := tree["lit"].(string); got != `\x20` {
t.Errorf("lit = %q, want the verbatim sequence", got)
}
// Two digits exactly; a short or non-hex escape is an error.
for _, doc := range []string{`a = "\x4"`, `a = "\x"`, `a = "\xgg"`} {
if _, err := Parse([]byte(doc)); err == nil {
t.Errorf("%s: expected an error, got none", doc)
}
}
}
func TestParseAcceptsMultilineInlineTables(t *testing.T) {
tree, err := Parse([]byte("tbl = {\n\thello = \"world\",\n\tarr = [1,\n\t\t2,\n\t],\n\tsub = {\n\t\tk = 1,\n\t},\n\tbare = 2}\n"))
if err != nil {
t.Fatalf("parse: %v", err)
}
tbl := tree["tbl"].(map[string]any)
if tbl["hello"] != "world" || tbl["bare"] != int64(2) {
t.Fatalf("tbl = %#v", tbl)
}
if arr := tbl["arr"].([]any); len(arr) != 2 {
t.Errorf("arr = %#v", tbl["arr"])
}
if sub := tbl["sub"].(map[string]any); sub["k"] != int64(1) {
t.Errorf("sub = %#v", tbl["sub"])
}
// Comments inside the table, and a trailing comma at both depths.
tree, err = Parse([]byte("m = { # one\n\t# two\n\ta = 1, # three\n\t# four\n}\n"))
if err != nil {
t.Fatalf("parse with comments: %v", err)
}
if m := tree["m"].(map[string]any); m["a"] != int64(1) {
t.Errorf("m = %#v", m)
}
// The old single-line shapes keep working, with and without the comma.
if _, err := Parse([]byte("a = { b = 1, c = 2 }\n")); err != nil {
t.Errorf("single line: %v", err)
}
// Still rejected: two commas, a missing value, and an unclosed table.
for name, doc := range map[string]string{
"double comma": "a = { b = 1,, c = 2 }\n",
"missing value": "a = {\n\tb =\n}\n",
"unterminated": "a = { b = 1,\n",
} {
if _, err := Parse([]byte(doc)); err == nil {
t.Errorf("%s: expected an error, got none", name)
}
}
}
+2 -2
View File
@@ -92,9 +92,9 @@ run:
dev: dev:
go run -buildvcs=true {{package}} go run -buildvcs=true {{package}}
# Runs the official toml-test compliance suite against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite. # Runs the official toml-test compliance suite against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/cmd/toml-test@v1.6.0); not standard because no canonical recipe covers a domain compliance suite.
toml-test: build toml-test: build
toml-test test -decoder=bin/interpres-decode -toml=1.1 toml-test bin/interpres-decode
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate. # Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
coverage-html: test coverage-html: test
+15 -44
View File
@@ -74,16 +74,15 @@ func decodeFloat(tok string) (any, error) {
sign, s := splitSign(tok) sign, s := splitSign(tok)
mantissa, exp := s, "" mantissa, exp := s, ""
hasExp := false
if i := strings.IndexAny(s, "eE"); i >= 0 { if i := strings.IndexAny(s, "eE"); i >= 0 {
mantissa, exp, hasExp = s[:i], s[i+1:], true mantissa, exp = s[:i], s[i+1:]
} }
intPart, frac, hasDot := mantissa, "", false intPart, frac, hasDot := mantissa, "", false
if i := strings.IndexByte(mantissa, '.'); i >= 0 { if i := strings.IndexByte(mantissa, '.'); i >= 0 {
intPart, frac, hasDot = mantissa[:i], mantissa[i+1:], true intPart, frac, hasDot = mantissa[:i], mantissa[i+1:], true
} }
if !hasDot && !hasExp { if !hasDot && exp == "" {
return nil, fmt.Errorf("invalid float %q", tok) return nil, fmt.Errorf("invalid float %q", tok)
} }
@@ -94,43 +93,24 @@ func decodeFloat(tok string) (any, error) {
if err := checkNoLeadingZero(ip); err != nil { if err := checkNoLeadingZero(ip); err != nil {
return nil, err return nil, err
} }
fp := ""
if hasDot {
if fp, err = joinDigits(frac, isDecDigit); err != nil {
return nil, err
}
}
// The ABNF requires at least one digit after the exponent marker, so a
// trailing e or E is an error even though strconv would accept it. The
// digits are a zero-prefixable integer, so leading zeros are fine here
// (the corpus holds valid cases such as 1e06 and 0e00).
esign, ed := "", ""
if hasExp {
var digits string
esign, digits = splitSign(exp)
if ed, err = joinDigits(digits, isDecDigit); err != nil {
return nil, err
}
}
// The checks above validated the token's shape, and every character a
// valid token may carry is one strconv.ParseFloat accepts in place, so
// only a token with underscores needs the stripped rebuild.
if !strings.ContainsRune(tok, '_') {
f, err := strconv.ParseFloat(tok, 64)
if err != nil {
return nil, fmt.Errorf("invalid float %q", tok)
}
return f, nil
}
build := sign + ip build := sign + ip
if hasDot { if hasDot {
fp, err := joinDigits(frac, isDecDigit)
if err != nil {
return nil, err
}
build += "." + fp build += "." + fp
} }
if hasExp { if exp != "" {
esign, edigits := splitSign(exp)
ed, err := joinDigits(edigits, isDecDigit)
if err != nil {
return nil, err
}
build += "e" + esign + ed build += "e" + esign + ed
} }
f, err := strconv.ParseFloat(build, 64) f, err := strconv.ParseFloat(build, 64)
if err != nil { if err != nil {
return nil, fmt.Errorf("invalid float %q", tok) return nil, fmt.Errorf("invalid float %q", tok)
@@ -140,20 +120,11 @@ func decodeFloat(tok string) (any, error) {
// joinDigits validates that every rune is a digit (per isDigit) and that each // joinDigits validates that every rune is a digit (per isDigit) and that each
// underscore sits between two digits, returning the digits with underscores // underscore sits between two digits, returning the digits with underscores
// removed. A token without underscores, the common case, is validated in // removed.
// place and returned without a copy.
func joinDigits(s string, isDigit func(byte) bool) (string, error) { func joinDigits(s string, isDigit func(byte) bool) (string, error) {
if s == "" { if s == "" {
return "", fmt.Errorf("number is missing digits") return "", fmt.Errorf("number is missing digits")
} }
if !strings.ContainsRune(s, '_') {
for i := range len(s) {
if !isDigit(s[i]) {
return "", fmt.Errorf("invalid character %q in number", string(s[i]))
}
}
return s, nil
}
var b strings.Builder var b strings.Builder
for i := range len(s) { for i := range len(s) {
c := s[i] c := s[i]
+106 -163
View File
@@ -8,7 +8,6 @@ import (
"fmt" "fmt"
"strconv" "strconv"
"strings" "strings"
"unicode/utf8"
) )
// ctxCheckInterval is the number of top-level parser iterations between // ctxCheckInterval is the number of top-level parser iterations between
@@ -17,15 +16,8 @@ import (
const ctxCheckInterval = 64 const ctxCheckInterval = 64
// parser is a recursive-descent TOML parser producing a map[string]any tree. // parser is a recursive-descent TOML parser producing a map[string]any tree.
//
// The scanner works on bytes, not runes: the input is validated UTF-8 before
// the parser runs, every character that drives the grammar (quotes,
// separators, newlines, bare-key characters) is ASCII, and multi-byte runes
// matter only as string content, where they are decoded on the spot. Holding
// the source as []rune instead would cost a conversion pass plus four bytes
// per rune of extra memory before parsing even starts.
type parser struct { type parser struct {
src []byte src []rune
pos int pos int
line int line int
ctx context.Context ctx context.Context
@@ -93,10 +85,10 @@ func (p *parser) checkCtx() error {
func (p *parser) parseTableHeader() error { func (p *parser) parseTableHeader() error {
array := false array := false
p.pos++ // consume '[' p.next() // consume '['
if !p.eof() && p.peek() == '[' { if !p.eof() && p.peek() == '[' {
array = true array = true
p.pos++ p.next()
} }
key, err := p.parseKeyPath() key, err := p.parseKeyPath()
@@ -108,12 +100,12 @@ func (p *parser) parseTableHeader() error {
if p.eof() || p.peek() != ']' { if p.eof() || p.peek() != ']' {
return p.errf("expected ']' to close table header") return p.errf("expected ']' to close table header")
} }
p.pos++ p.next()
if array { if array {
if p.eof() || p.peek() != ']' { if p.eof() || p.peek() != ']' {
return p.errf("expected ']]' to close array-of-tables header") return p.errf("expected ']]' to close array-of-tables header")
} }
p.pos++ p.next()
} }
if array { if array {
@@ -179,12 +171,7 @@ func (p *parser) tableAt(key []string) (map[string]any, error) {
func (p *parser) appendArrayTable(key []string) (map[string]any, error) { func (p *parser) appendArrayTable(key []string) (map[string]any, error) {
parent := p.root parent := p.root
path := make([]string, 0, len(key))
for _, k := range key[:len(key)-1] { for _, k := range key[:len(key)-1] {
path = append(path, k)
if p.frozen[pathKey(path)] {
return nil, p.errf("cannot extend inline table %q", strings.Join(path, "."))
}
existing, ok := parent[k] existing, ok := parent[k]
if !ok { if !ok {
next := map[string]any{} next := map[string]any{}
@@ -226,7 +213,7 @@ func (p *parser) parseKeyValue() error {
if p.eof() || p.peek() != '=' { if p.eof() || p.peek() != '=' {
return p.errf("expected '=' after key") return p.errf("expected '=' after key")
} }
p.pos++ p.next()
p.skipInline() p.skipInline()
val, err := p.parseValue() val, err := p.parseValue()
@@ -235,10 +222,7 @@ func (p *parser) parseKeyValue() error {
} }
dest := p.current dest := p.current
// One allocation covers the current section plus the dotted key; a abs := append([]string{}, p.currentPath...)
// top-level statement reuses it for the leaf.
abs := make([]string, 0, len(p.currentPath)+len(key))
abs = append(abs, p.currentPath...)
for _, k := range key[:len(key)-1] { for _, k := range key[:len(key)-1] {
abs = append(abs, k) abs = append(abs, k)
if p.frozen[pathKey(abs)] { if p.frozen[pathKey(abs)] {
@@ -285,17 +269,18 @@ func (p *parser) freezeInline(path []string, val any) {
} }
} }
// resetScopeUnder forgets the definition records nested under key, which // resetScopeUnder forgets the header and freeze records nested under key, which
// belong to the previous element of an array of tables: headers, frozen // belong to the previous element of an array of tables.
// inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element.
func (p *parser) resetScopeUnder(key []string) { func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00" prefix := pathKey(key) + "\x00"
for _, m := range []map[string]bool{p.headers, p.frozen, p.dotted, p.arrays} { for k := range p.headers {
for k := range m { if strings.HasPrefix(k, prefix) {
if strings.HasPrefix(k, prefix) { delete(p.headers, k)
delete(m, k) }
} }
for k := range p.frozen {
if strings.HasPrefix(k, prefix) {
delete(p.frozen, k)
} }
} }
} }
@@ -312,7 +297,7 @@ func (p *parser) parseKeyPath() ([]string, error) {
parts = append(parts, part) parts = append(parts, part)
p.skipInline() p.skipInline()
if !p.eof() && p.peek() == '.' { if !p.eof() && p.peek() == '.' {
p.pos++ p.next()
continue continue
} }
break break
@@ -324,7 +309,7 @@ func (p *parser) parseKeyComponent() (string, error) {
if p.eof() { if p.eof() {
return "", p.errf("expected a key") return "", p.errf("expected a key")
} }
switch p.peek() { switch c := p.peek(); c {
case '"': case '"':
if p.lookahead(`"""`) { if p.lookahead(`"""`) {
return "", p.errf("multiline strings are not allowed in keys") return "", p.errf("multiline strings are not allowed in keys")
@@ -341,14 +326,13 @@ func (p *parser) parseKeyComponent() (string, error) {
c := p.peek() c := p.peek()
if (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') || if (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') ||
(c >= '0' && c <= '9') || c == '_' || c == '-' { (c >= '0' && c <= '9') || c == '_' || c == '-' {
p.pos++ p.next()
continue continue
} }
break break
} }
if p.pos == start { if p.pos == start {
r, _ := utf8.DecodeRune(p.src[p.pos:]) return "", p.errf("invalid key character %q", string(p.peek()))
return "", p.errf("invalid key character %q", string(r))
} }
return string(p.src[start:p.pos]), nil return string(p.src[start:p.pos]), nil
} }
@@ -397,7 +381,7 @@ func (p *parser) parseAtom() (any, error) {
// A date may be followed by a space and a time, forming one date-time. // A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' { if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
if next, ok := p.peekAt(1); ok && next >= '0' && next <= '9' { if next, ok := p.peekAt(1); ok && next >= '0' && next <= '9' {
p.pos++ // consume the separating space p.next() // consume the separating space
timeStart := p.pos timeStart := p.pos
p.scanBareToken() p.scanBareToken()
tok = tok + " " + string(p.src[timeStart:p.pos]) tok = tok + " " + string(p.src[timeStart:p.pos])
@@ -422,7 +406,7 @@ func (p *parser) scanBareToken() {
c == ',' || c == ']' || c == '}' || c == '#' { c == ',' || c == ']' || c == '}' || c == '#' {
return return
} }
p.pos++ p.next()
} }
} }
@@ -432,32 +416,31 @@ func (p *parser) parseBasicString() (string, error) {
if p.lookahead(`"""`) { if p.lookahead(`"""`) {
return p.parseMultilineString('"', true) return p.parseMultilineString('"', true)
} }
p.pos++ // opening quote p.next() // opening quote
var b strings.Builder var b strings.Builder
for { for {
if p.eof() { if p.eof() {
return "", p.errf("unterminated string") return "", p.errf("unterminated string")
} }
c := p.peek() c := p.next()
switch c { switch c {
case '"': case '"':
p.pos++
return b.String(), nil return b.String(), nil
case '\n': case '\n':
return "", p.errf("unterminated string") return "", p.errf("unterminated string")
case '\r': case '\r':
return "", p.errf("bare carriage return is not allowed in a string") return "", p.errf("bare carriage return is not allowed in a string")
case '\\': case '\\':
p.pos++
r, err := p.readEscape() r, err := p.readEscape()
if err != nil { if err != nil {
return "", err return "", err
} }
b.WriteRune(r) b.WriteRune(r)
default: default:
if err := p.writeContentRune(&b); err != nil { if isControlRune(c) {
return "", err return "", p.errf("control character U+%04X is not allowed in a string", c)
} }
b.WriteRune(c)
} }
} }
} }
@@ -466,58 +449,38 @@ func (p *parser) parseLiteralString() (string, error) {
if p.lookahead(`'''`) { if p.lookahead(`'''`) {
return p.parseMultilineString('\'', false) return p.parseMultilineString('\'', false)
} }
p.pos++ // opening quote p.next() // opening quote
var b strings.Builder var b strings.Builder
for { for {
if p.eof() { if p.eof() {
return "", p.errf("unterminated literal string") return "", p.errf("unterminated literal string")
} }
c := p.peek() c := p.next()
switch c { if c == '\'' {
case '\'':
p.pos++
return b.String(), nil return b.String(), nil
case '\n': }
if c == '\n' {
return "", p.errf("unterminated literal string") return "", p.errf("unterminated literal string")
case '\r': }
if c == '\r' {
return "", p.errf("bare carriage return is not allowed in a string") return "", p.errf("bare carriage return is not allowed in a string")
default:
if err := p.writeContentRune(&b); err != nil {
return "", err
}
} }
if isControlRune(c) {
return "", p.errf("control character U+%04X is not allowed in a string", c)
}
b.WriteRune(c)
} }
} }
// writeContentRune appends the rune at the cursor to b and advances past it. func (p *parser) parseMultilineString(quote rune, escapes bool) (string, error) {
// An ASCII byte, which includes every control character the grammar forbids,
// is checked and written directly; a multi-byte rune is decoded and can never
// be a control character.
func (p *parser) writeContentRune(b *strings.Builder) error {
c := p.peek()
if c < utf8.RuneSelf {
if isControlRune(rune(c)) {
return p.errf("control character U+%04X is not allowed in a string", c)
}
p.pos++
b.WriteByte(c)
return nil
}
r, size := utf8.DecodeRune(p.src[p.pos:])
p.pos += size
b.WriteRune(r)
return nil
}
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed. // A newline immediately after the opening delimiter is trimmed.
if !p.eof() && p.peek() == '\r' { if !p.eof() && p.peek() == '\r' {
p.pos++ p.next()
} }
if !p.eof() && p.peek() == '\n' { if !p.eof() && p.peek() == '\n' {
p.line++ p.line++
p.pos++ p.next()
} }
var b strings.Builder var b strings.Builder
@@ -537,32 +500,31 @@ func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error)
return "", p.errf("too many '%c' before the closing delimiter", quote) return "", p.errf("too many '%c' before the closing delimiter", quote)
} }
for range n - 3 { for range n - 3 {
b.WriteByte(quote) b.WriteRune(quote)
} }
p.skipN(n) p.skipN(n)
return b.String(), nil return b.String(), nil
} }
for range n { for range n {
b.WriteByte(quote) b.WriteRune(quote)
p.pos++ p.next()
} }
continue continue
} }
c := p.peek() c := p.next()
switch { if c == '\n' {
case c == '\n':
p.line++ p.line++
p.pos++ b.WriteRune(c)
b.WriteByte(c) continue
case c == '\r': }
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' { if c == '\r' {
b.WriteByte(c) if !p.eof() && p.peek() == '\n' {
p.pos++ b.WriteRune(c)
continue continue
} }
return "", p.errf("bare carriage return is not allowed in a string") return "", p.errf("bare carriage return is not allowed in a string")
case escapes && c == '\\': }
p.pos++ if escapes && c == '\\' {
// Line-ending backslash trims the following whitespace/newlines. // Line-ending backslash trims the following whitespace/newlines.
if p.trimLineEndingBackslash() { if p.trimLineEndingBackslash() {
continue continue
@@ -572,11 +534,12 @@ func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error)
return "", err return "", err
} }
b.WriteRune(r) b.WriteRune(r)
default: continue
if err := p.writeContentRune(&b); err != nil {
return "", err
}
} }
if isControlRune(c) {
return "", p.errf("control character U+%04X is not allowed in a string", c)
}
b.WriteRune(c)
} }
} }
@@ -588,7 +551,7 @@ func (p *parser) trimLineEndingBackslash() bool {
for !p.eof() { for !p.eof() {
c := p.peek() c := p.peek()
if c == ' ' || c == '\t' || c == '\r' { if c == ' ' || c == '\t' || c == '\r' {
p.pos++ p.next()
continue continue
} }
if c == '\n' { if c == '\n' {
@@ -607,11 +570,11 @@ func (p *parser) trimLineEndingBackslash() bool {
c := p.peek() c := p.peek()
if c == '\n' { if c == '\n' {
p.line++ p.line++
p.pos++ p.next()
continue continue
} }
if c == ' ' || c == '\t' || c == '\r' { if c == ' ' || c == '\t' || c == '\r' {
p.pos++ p.next()
continue continue
} }
break break
@@ -635,25 +598,16 @@ func (p *parser) readEscape() (rune, error) {
return '\f', nil return '\f', nil
case 'r': case 'r':
return '\r', nil return '\r', nil
case 'e':
// TOML 1.1: the escape character.
return '\x1b', nil
case '"': case '"':
return '"', nil return '"', nil
case '\\': case '\\':
return '\\', nil return '\\', nil
case 'x':
// TOML 1.1: two hex digits, code points 0x00 through 0xFF.
return p.readUnicode(2)
case 'u': case 'u':
return p.readUnicode(4) return p.readUnicode(4)
case 'U': case 'U':
return p.readUnicode(8) return p.readUnicode(8)
default: default:
// The byte just consumed starts a rune: the backslash before it is a return 0, p.errf("invalid escape sequence \\%c", c)
// boundary, and the input is valid UTF-8.
r, _ := utf8.DecodeRune(p.src[p.pos-1:])
return 0, p.errf("invalid escape sequence \\%c", r)
} }
} }
@@ -676,17 +630,17 @@ func (p *parser) readUnicode(n int) (rune, error) {
// --- arrays and inline tables --------------------------------------------- // --- arrays and inline tables ---------------------------------------------
func (p *parser) parseArray() (any, error) { func (p *parser) parseArray() (any, error) {
p.pos++ // '[' p.next() // '['
arr := []any{} arr := []any{}
for { for {
if err := p.skipNestedSpace(); err != nil { if err := p.skipArraySpace(); err != nil {
return nil, err return nil, err
} }
if p.eof() { if p.eof() {
return nil, p.errf("unterminated array") return nil, p.errf("unterminated array")
} }
if p.peek() == ']' { if p.peek() == ']' {
p.pos++ p.next()
return arr, nil return arr, nil
} }
v, err := p.parseValue() v, err := p.parseValue()
@@ -694,7 +648,7 @@ func (p *parser) parseArray() (any, error) {
return nil, err return nil, err
} }
arr = append(arr, v) arr = append(arr, v)
if err := p.skipNestedSpace(); err != nil { if err := p.skipArraySpace(); err != nil {
return nil, err return nil, err
} }
if p.eof() { if p.eof() {
@@ -702,9 +656,9 @@ func (p *parser) parseArray() (any, error) {
} }
switch p.peek() { switch p.peek() {
case ',': case ',':
p.pos++ p.next()
case ']': case ']':
p.pos++ p.next()
return arr, nil return arr, nil
default: default:
return nil, p.errf("expected ',' or ']' in array") return nil, p.errf("expected ',' or ']' in array")
@@ -713,23 +667,16 @@ func (p *parser) parseArray() (any, error) {
} }
func (p *parser) parseInlineTable() (any, error) { func (p *parser) parseInlineTable() (any, error) {
p.pos++ // '{' p.next() // '{'
tbl := map[string]any{} tbl := map[string]any{}
assigned := map[string]bool{} assigned := map[string]bool{}
// TOML 1.1 lets an inline table span lines: interior whitespace includes p.skipInline()
// newlines and comments, and a trailing comma is allowed before the
// closing brace.
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if !p.eof() && p.peek() == '}' { if !p.eof() && p.peek() == '}' {
p.pos++ p.next()
return tbl, nil return tbl, nil
} }
for { for {
if err := p.skipNestedSpace(); err != nil { p.skipInline()
return nil, err
}
key, err := p.parseKeyPath() key, err := p.parseKeyPath()
if err != nil { if err != nil {
return nil, err return nil, err
@@ -738,7 +685,7 @@ func (p *parser) parseInlineTable() (any, error) {
if p.eof() || p.peek() != '=' { if p.eof() || p.peek() != '=' {
return nil, p.errf("expected '=' in inline table") return nil, p.errf("expected '=' in inline table")
} }
p.pos++ p.next()
p.skipInline() p.skipInline()
val, err := p.parseValue() val, err := p.parseValue()
if err != nil { if err != nil {
@@ -773,24 +720,15 @@ func (p *parser) parseInlineTable() (any, error) {
dest[leaf] = val dest[leaf] = val
assigned[pathKey(path)] = true assigned[pathKey(path)] = true
if err := p.skipNestedSpace(); err != nil { p.skipInline()
return nil, err
}
if p.eof() { if p.eof() {
return nil, p.errf("unterminated inline table") return nil, p.errf("unterminated inline table")
} }
switch p.peek() { switch p.peek() {
case ',': case ',':
p.pos++ p.next()
if err := p.skipNestedSpace(); err != nil {
return nil, err
}
if !p.eof() && p.peek() == '}' {
p.pos++
return tbl, nil
}
case '}': case '}':
p.pos++ p.next()
return tbl, nil return tbl, nil
default: default:
return nil, p.errf("expected ',' or '}' in inline table") return nil, p.errf("expected ',' or '}' in inline table")
@@ -801,12 +739,12 @@ func (p *parser) parseInlineTable() (any, error) {
// --- scanning helpers ------------------------------------------------------ // --- scanning helpers ------------------------------------------------------
func (p *parser) eof() bool { return p.pos >= len(p.src) } func (p *parser) eof() bool { return p.pos >= len(p.src) }
func (p *parser) peek() byte { return p.src[p.pos] } func (p *parser) peek() rune { return p.src[p.pos] }
// peekAt returns the byte at offset n from the current position and whether the // peekAt returns the rune at offset n from the current position and whether the
// offset is within the source. Use it instead of indexing p.src directly when // offset is within the source. Use it instead of indexing p.src directly when
// the offset may sit past the end. // the offset may sit past the end.
func (p *parser) peekAt(n int) (byte, bool) { func (p *parser) peekAt(n int) (rune, bool) {
i := p.pos + n i := p.pos + n
if i < 0 || i >= len(p.src) { if i < 0 || i >= len(p.src) {
return 0, false return 0, false
@@ -814,7 +752,7 @@ func (p *parser) peekAt(n int) (byte, bool) {
return p.src[i], true return p.src[i], true
} }
func (p *parser) next() byte { func (p *parser) next() rune {
c := p.src[p.pos] c := p.src[p.pos]
p.pos++ p.pos++
return c return c
@@ -828,43 +766,49 @@ func (p *parser) skipN(n int) {
func (p *parser) match(word string) bool { func (p *parser) match(word string) bool {
if p.lookahead(word) { if p.lookahead(word) {
p.skipN(len(word)) p.skipN(len([]rune(word)))
return true return true
} }
return false return false
} }
// lookahead reports whether s follows the cursor. Every lookahead argument in
// the grammar is ASCII, so comparing bytes is exact.
func (p *parser) lookahead(s string) bool { func (p *parser) lookahead(s string) bool {
return p.pos+len(s) <= len(p.src) && string(p.src[p.pos:p.pos+len(s)]) == s r := []rune(s)
if p.pos+len(r) > len(p.src) {
return false
}
for i, c := range r {
if p.src[p.pos+i] != c {
return false
}
}
return true
} }
// skipInline consumes spaces and tabs only. // skipInline consumes spaces and tabs only.
func (p *parser) skipInline() { func (p *parser) skipInline() {
for !p.eof() { for !p.eof() {
if c := p.peek(); c == ' ' || c == '\t' { if c := p.peek(); c == ' ' || c == '\t' {
p.pos++ p.next()
continue continue
} }
break break
} }
} }
// skipNestedSpace consumes whitespace, newlines, and comments inside a value // skipArraySpace consumes whitespace, newlines, and comments inside arrays.
// container (an array, or an inline table under TOML 1.1). func (p *parser) skipArraySpace() error {
func (p *parser) skipNestedSpace() error {
for !p.eof() { for !p.eof() {
switch p.peek() { switch p.peek() {
case ' ', '\t': case ' ', '\t':
p.pos++ p.next()
case '\r': case '\r':
if err := p.expectCRLF(); err != nil { if err := p.expectCRLF(); err != nil {
return err return err
} }
case '\n': case '\n':
p.line++ p.line++
p.pos++ p.next()
case '#': case '#':
if err := p.skipComment(); err != nil { if err := p.skipComment(); err != nil {
return err return err
@@ -881,14 +825,14 @@ func (p *parser) skipBlank() error {
for !p.eof() { for !p.eof() {
switch p.peek() { switch p.peek() {
case ' ', '\t': case ' ', '\t':
p.pos++ p.next()
case '\r': case '\r':
if err := p.expectCRLF(); err != nil { if err := p.expectCRLF(); err != nil {
return err return err
} }
case '\n': case '\n':
p.line++ p.line++
p.pos++ p.next()
case '#': case '#':
if err := p.skipComment(); err != nil { if err := p.skipComment(); err != nil {
return err return err
@@ -901,7 +845,7 @@ func (p *parser) skipBlank() error {
} }
func (p *parser) skipComment() error { func (p *parser) skipComment() error {
p.pos++ // consume '#' p.next() // consume '#'
for !p.eof() { for !p.eof() {
c := p.peek() c := p.peek()
switch { switch {
@@ -913,11 +857,11 @@ func (p *parser) skipComment() error {
} }
return p.errf("bare carriage return is not allowed") return p.errf("bare carriage return is not allowed")
case c == '\t': case c == '\t':
p.pos++ p.next()
case c < 0x20 || c == 0x7f: case c < 0x20 || c == 0x7f:
return p.errf("control character U+%04X is not allowed in a comment", c) return p.errf("control character U+%04X is not allowed in a comment", c)
default: default:
p.pos++ p.next()
} }
} }
return nil return nil
@@ -927,7 +871,7 @@ func (p *parser) skipComment() error {
// line feed; a bare CR is invalid. // line feed; a bare CR is invalid.
func (p *parser) expectCRLF() error { func (p *parser) expectCRLF() error {
if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' { if p.pos+1 < len(p.src) && p.src[p.pos+1] == '\n' {
p.pos++ // consume CR; the LF is handled by the caller p.next() // consume CR; the LF is handled by the caller
return nil return nil
} }
return p.errf("bare carriage return is not allowed") return p.errf("bare carriage return is not allowed")
@@ -958,11 +902,10 @@ func (p *parser) expectLineEnd() error {
} }
if p.peek() == '\n' { if p.peek() == '\n' {
p.line++ p.line++
p.pos++ p.next()
return nil return nil
} }
r, _ := utf8.DecodeRune(p.src[p.pos:]) return p.errf("unexpected %q after value", string(p.peek()))
return p.errf("unexpected %q after value", string(r))
} }
func (p *parser) errf(format string, args ...any) error { func (p *parser) errf(format string, args ...any) error {
-2
View File
@@ -1,2 +0,0 @@
go test fuzz v1
[]byte("0=[{}]")