From d2fc31d2601f6ff1bf3ea22d564d8b45209fa221 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Thu, 17 Sep 2026 22:06:26 +0200 Subject: [PATCH] feat: support TOML 1.1 Assisted-by: GLM 5.3 Flash --- .gitea/workflows/test.yml | 5 +- CHANGELOG.md | 6 +++ README.md | 19 +++---- datetime.go | 15 ++++-- docs/API.md | 2 +- docs/ARCHITECTURE.md | 6 +-- fuzz_test.go | 4 ++ interpres.go | 3 +- interpres_test.go | 105 +++++++++++++++++++++++++++++++++++++- justfile | 2 +- parser.go | 37 +++++++++++--- 11 files changed, 175 insertions(+), 29 deletions(-) diff --git a/.gitea/workflows/test.yml b/.gitea/workflows/test.yml index 87aa637..12ef529 100644 --- a/.gitea/workflows/test.yml +++ b/.gitea/workflows/test.yml @@ -105,5 +105,6 @@ jobs: run: go build -o bin/interpres-decode ./cmd/interpres-decode - name: Compliance suite - # interpres implements TOML 1.0; v2 tests 1.1 by default, so the mode is pinned. - run: bin/toml-test test -decoder=bin/interpres-decode -toml=1.0 + # interpres implements TOML 1.0 and 1.1; the mode is pinned so an upstream + # default change cannot silently move the corpus. + run: bin/toml-test test -decoder=bin/interpres-decode -toml=1.1 diff --git a/CHANGELOG.md b/CHANGELOG.md index b8309b9..83ecbcd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- TOML 1.1 support, on by default: date-times and times without seconds + (`07:32`, `1979-05-27T07:32`, normalised to full seconds on output), the + `\e` and `\xHH` escape sequences, and multi-line inline tables with + comments and trailing commas. The compliance suite runs in TOML 1.1 mode: + 214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document + parses exactly as before. - `interpres-decode -validate [file ...]`: a validate mode beside the toml-test adapter. It parses each named file, or stdin when none are named, prints one line per invalid document to stderr, and exits 0 when all are diff --git a/README.md b/README.md index 235a7bb..1865a71 100644 --- a/README.md +++ b/README.md @@ -1,17 +1,18 @@ # interpres -A TOML 1.0 parser and encoder for Go, written with the standard library alone. -`interpres` (Latin for *interpreter*) gives zero-dependency programs an -`encoding/json`-style API for reading and writing TOML, and passes the entire -official [toml-test](https://github.com/toml-lang/toml-test) suite: 205 valid -and 474 invalid cases, zero failures. +A TOML 1.0 and 1.1 parser and encoder for Go, written with the standard +library alone. `interpres` (Latin for *interpreter*) gives zero-dependency +programs an `encoding/json`-style API for reading and writing TOML, and passes +the entire official [toml-test](https://github.com/toml-lang/toml-test) suite: +214 valid and 467 invalid cases, zero failures. ## Features -- **Full TOML 1.0**: bare, quoted and dotted keys; tables and arrays of tables; - basic and literal strings including multiline; integers in the four radixes - with `_` separators; floats with exponents, `inf` and `nan`; booleans; the - four date-time kinds; arrays and inline tables. +- **Full TOML 1.0 and 1.1**: bare, quoted and dotted keys; tables and arrays of + tables; basic and literal strings including multiline, with the 1.1 `\e` and + `\xHH` escapes; integers in the four radixes with `_` separators; floats with + exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds + optional as of 1.1; arrays and inline tables, multi-line as of 1.1. - **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and `Marshal` for structs and maps, mirroring `encoding/json`. - **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that diff --git a/datetime.go b/datetime.go index 6710b1d..ccfaeab 100644 --- a/datetime.go +++ b/datetime.go @@ -59,24 +59,31 @@ var ( "2006-01-02T15:04:05Z07:00", "2006-01-02 15:04:05.999999999Z07:00", "2006-01-02 15:04:05Z07:00", + // TOML 1.1 makes the seconds optional. + "2006-01-02T15:04Z07:00", + "2006-01-02 15:04Z07:00", } localDateTimeLayouts = []string{ "2006-01-02T15:04:05.999999999", "2006-01-02T15:04:05", "2006-01-02 15:04:05.999999999", "2006-01-02 15:04:05", + "2006-01-02T15:04", + "2006-01-02 15:04", } localTimeLayouts = []string{ "15:04:05.999999999", "15:04:05", + "15:04", } ) -// dateTimeShape enforces the strict TOML grammar (two-digit components) that -// time.Parse would otherwise accept loosely (e.g. a single-digit hour). +// dateTimeShape enforces the strict TOML grammar (two-digit components, +// seconds optional since 1.1, a fraction only after seconds) that time.Parse +// would otherwise accept loosely (e.g. a single-digit hour). var dateTimeShape = regexp.MustCompile( - `^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}:\d{2}(\.\d+)?([Zz]|[+-]\d{2}:\d{2})?)?$` + - `|^\d{2}:\d{2}:\d{2}(\.\d+)?$`, + `^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}(:\d{2}(\.\d+)?)?([Zz]|[+-]\d{2}:\d{2})?)?$` + + `|^\d{2}:\d{2}(:\d{2}(\.\d+)?)?$`, ) // offsetBounds extracts the numeric offset of a date-time. The ABNF bounds it diff --git a/docs/API.md b/docs/API.md index 601c99a..7718ed5 100644 --- a/docs/API.md +++ b/docs/API.md @@ -46,7 +46,7 @@ The cancellable variant of `Unmarshal`. ### `func Marshal(v any) ([]byte, error)` Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a -TOML 1.0 document. The emission rules are in the [Encoding](#encoding) section +TOML document. The emission rules are in the [Encoding](#encoding) section below. Equivalent to `MarshalContext(context.Background(), v)`. ```go diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 1bcff50..82378d2 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -6,9 +6,9 @@ source tree; nothing is aspirational. ## Overview interpres is one public library package, one command, and one example. The -library implements the whole of TOML 1.0, decoding and encoding, in the +library implements the whole of TOML 1.0 and 1.1, decoding and encoding, in the standard library alone; the command wraps the parser for the toml-test -compliance harness, against which it stands at 185 valid and 371 invalid cases +compliance harness, against which it stands at 214 valid and 467 invalid cases with zero failures; the example demonstrates the API. ```mermaid @@ -43,7 +43,7 @@ Inside the library package, one file owns one concern: | File | Responsibility | |---|---| -| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 (table redefinitions, dotted keys, arrays of tables). Reports a 1-based line on failure. | +| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 and 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. | | `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. | | `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. | | `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. | diff --git a/fuzz_test.go b/fuzz_test.go index a95c9be..46e0b4b 100644 --- a/fuzz_test.go +++ b/fuzz_test.go @@ -32,6 +32,10 @@ func FuzzParse(f *testing.F) { "x = \"unterminated\n", "[a]\n[a]\n", "n = 0x1_0000_0000_0000_0000\n", + // TOML 1.1 forms. + "t = 13:37\ndt = 1979-05-27T07:32\nodt = 1979-05-27 07:32Z\n", + "esc = \"\\e\\x41\\x7f\\x00\"\n", + "m = {\n\ta = 1,\n\tb = [1, 2,],\n\tc = { d = 2 },\n} # close\n", } for _, s := range seeds { f.Add([]byte(s)) diff --git a/interpres.go b/interpres.go index 8d4b3d3..6495df4 100644 --- a/interpres.go +++ b/interpres.go @@ -196,7 +196,8 @@ type Unmarshaler interface { UnmarshalTOML(data any) error } -// Marshal returns the TOML 1.0 encoding of v. +// Marshal returns the TOML encoding of v. The output stays within TOML 1.0, +// so it is valid under both TOML 1.0 and 1.1. // // Marshal traverses v using reflection and applies the following rules: // diff --git a/interpres_test.go b/interpres_test.go index 7dc416f..7b3e3fa 100644 --- a/interpres_test.go +++ b/interpres_test.go @@ -503,7 +503,8 @@ func TestRejectsSpecInvalid(t *testing.T) { "dotted over header": "[a.b]\nx = 1\n[a]\nb.y = 2\n", "table over array": "[[t]]\n[t]\n", "truncated datetime": "a = 2026-01-02T\n", - "datetime no seconds": "a = 2026-01-02T07:32\n", + // "datetime no seconds" moved to the acceptance tests: TOML 1.1 + // makes the seconds optional. } for name, doc := range cases { if _, err := Parse([]byte(doc)); err == nil { @@ -540,3 +541,105 @@ host = "h2" t.Errorf("forms[1].smtp.host = %v", h) } } + +// --- TOML 1.1 -------------------------------------------------------------- + +func TestParseAcceptsNoSecondsDatetimes(t *testing.T) { + tree, err := Parse([]byte(`t = 13:37 +dt = 1979-05-27T07:32 +odt1 = 1979-05-27 07:32Z +odt2 = 1979-05-27 07:32-07:00 +`)) + if err != nil { + t.Fatalf("parse: %v", err) + } + if got := tree["t"].(LocalTime).String(); got != "13:37:00" { + t.Errorf("t = %q, want %q", got, "13:37:00") + } + if got := tree["dt"].(LocalDateTime).String(); got != "1979-05-27T07:32:00" { + t.Errorf("dt = %q, want %q", got, "1979-05-27T07:32:00") + } + if got := tree["odt1"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00Z" { + t.Errorf("odt1 = %q", got) + } + if got := tree["odt2"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00-07:00" { + t.Errorf("odt2 = %q", got) + } + // The fraction still requires the seconds it belongs to. + if _, err := Parse([]byte("a = 07:32.5\n")); err == nil { + t.Error("07:32.5: expected an error, got none") + } +} + +func TestParseAcceptsEscapeAndHexEscapes(t *testing.T) { + tree, err := Parse([]byte(`esc = "\e" +hex = "\x20\x7f\xf8" +nul = "\x00" +multi = """\x68\x65""" +lit = '\x20' +`)) + if err != nil { + t.Fatalf("parse: %v", err) + } + if got := tree["esc"].(string); got != "\x1b" { + t.Errorf("esc = %q, want the escape character", got) + } + if got := tree["hex"].(string); got != " \x7f\u00f8" { + t.Errorf("hex = %q", got) + } + if got := tree["nul"].(string); got != "\x00" { + t.Errorf("nul = %q", got) + } + if got := tree["multi"].(string); got != "he" { + t.Errorf("multi = %q", got) + } + // A literal string carries the sequence verbatim. + if got := tree["lit"].(string); got != `\x20` { + t.Errorf("lit = %q, want the verbatim sequence", got) + } + // Two digits exactly; a short or non-hex escape is an error. + for _, doc := range []string{`a = "\x4"`, `a = "\x"`, `a = "\xgg"`} { + if _, err := Parse([]byte(doc)); err == nil { + t.Errorf("%s: expected an error, got none", doc) + } + } +} + +func TestParseAcceptsMultilineInlineTables(t *testing.T) { + tree, err := Parse([]byte("tbl = {\n\thello = \"world\",\n\tarr = [1,\n\t\t2,\n\t],\n\tsub = {\n\t\tk = 1,\n\t},\n\tbare = 2}\n")) + if err != nil { + t.Fatalf("parse: %v", err) + } + tbl := tree["tbl"].(map[string]any) + if tbl["hello"] != "world" || tbl["bare"] != int64(2) { + t.Fatalf("tbl = %#v", tbl) + } + if arr := tbl["arr"].([]any); len(arr) != 2 { + t.Errorf("arr = %#v", tbl["arr"]) + } + if sub := tbl["sub"].(map[string]any); sub["k"] != int64(1) { + t.Errorf("sub = %#v", tbl["sub"]) + } + // Comments inside the table, and a trailing comma at both depths. + tree, err = Parse([]byte("m = { # one\n\t# two\n\ta = 1, # three\n\t# four\n}\n")) + if err != nil { + t.Fatalf("parse with comments: %v", err) + } + if m := tree["m"].(map[string]any); m["a"] != int64(1) { + t.Errorf("m = %#v", m) + } + // The old single-line shapes keep working, with and without the comma. + if _, err := Parse([]byte("a = { b = 1, c = 2 }\n")); err != nil { + t.Errorf("single line: %v", err) + } + // Still rejected: two commas, a missing value, and an unclosed table. + for name, doc := range map[string]string{ + "double comma": "a = { b = 1,, c = 2 }\n", + "missing value": "a = {\n\tb =\n}\n", + "unterminated": "a = { b = 1,\n", + } { + if _, err := Parse([]byte(doc)); err == nil { + t.Errorf("%s: expected an error, got none", name) + } + } +} diff --git a/justfile b/justfile index 4f4f91e..0751b64 100644 --- a/justfile +++ b/justfile @@ -94,7 +94,7 @@ dev: # Runs the official toml-test compliance suite against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite. toml-test: build - toml-test test -decoder=bin/interpres-decode -toml=1.0 + toml-test test -decoder=bin/interpres-decode -toml=1.1 # Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate. coverage-html: test diff --git a/parser.go b/parser.go index cc40128..fb0715e 100644 --- a/parser.go +++ b/parser.go @@ -598,10 +598,16 @@ func (p *parser) readEscape() (rune, error) { return '\f', nil case 'r': return '\r', nil + case 'e': + // TOML 1.1: the escape character. + return '\x1b', nil case '"': return '"', nil case '\\': return '\\', nil + case 'x': + // TOML 1.1: two hex digits, code points 0x00 through 0xFF. + return p.readUnicode(2) case 'u': return p.readUnicode(4) case 'U': @@ -633,7 +639,7 @@ func (p *parser) parseArray() (any, error) { p.next() // '[' arr := []any{} for { - if err := p.skipArraySpace(); err != nil { + if err := p.skipNestedSpace(); err != nil { return nil, err } if p.eof() { @@ -648,7 +654,7 @@ func (p *parser) parseArray() (any, error) { return nil, err } arr = append(arr, v) - if err := p.skipArraySpace(); err != nil { + if err := p.skipNestedSpace(); err != nil { return nil, err } if p.eof() { @@ -670,13 +676,20 @@ func (p *parser) parseInlineTable() (any, error) { p.next() // '{' tbl := map[string]any{} assigned := map[string]bool{} - p.skipInline() + // TOML 1.1 lets an inline table span lines: interior whitespace includes + // newlines and comments, and a trailing comma is allowed before the + // closing brace. + if err := p.skipNestedSpace(); err != nil { + return nil, err + } if !p.eof() && p.peek() == '}' { p.next() return tbl, nil } for { - p.skipInline() + if err := p.skipNestedSpace(); err != nil { + return nil, err + } key, err := p.parseKeyPath() if err != nil { return nil, err @@ -720,13 +733,22 @@ func (p *parser) parseInlineTable() (any, error) { dest[leaf] = val assigned[pathKey(path)] = true - p.skipInline() + if err := p.skipNestedSpace(); err != nil { + return nil, err + } if p.eof() { return nil, p.errf("unterminated inline table") } switch p.peek() { case ',': p.next() + if err := p.skipNestedSpace(); err != nil { + return nil, err + } + if !p.eof() && p.peek() == '}' { + p.next() + return tbl, nil + } case '}': p.next() return tbl, nil @@ -796,8 +818,9 @@ func (p *parser) skipInline() { } } -// skipArraySpace consumes whitespace, newlines, and comments inside arrays. -func (p *parser) skipArraySpace() error { +// skipNestedSpace consumes whitespace, newlines, and comments inside a value +// container (an array, or an inline table under TOML 1.1). +func (p *parser) skipNestedSpace() error { for !p.eof() { switch p.peek() { case ' ', '\t':