Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -105,5 +105,6 @@ jobs:
|
||||
run: go build -o bin/interpres-decode ./cmd/interpres-decode
|
||||
|
||||
- name: Compliance suite
|
||||
# interpres implements TOML 1.0; v2 tests 1.1 by default, so the mode is pinned.
|
||||
run: bin/toml-test test -decoder=bin/interpres-decode -toml=1.0
|
||||
# interpres implements TOML 1.0 and 1.1; the mode is pinned so an upstream
|
||||
# default change cannot silently move the corpus.
|
||||
run: bin/toml-test test -decoder=bin/interpres-decode -toml=1.1
|
||||
|
||||
@@ -9,6 +9,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
### Added
|
||||
|
||||
- TOML 1.1 support, on by default: date-times and times without seconds
|
||||
(`07:32`, `1979-05-27T07:32`, normalised to full seconds on output), the
|
||||
`\e` and `\xHH` escape sequences, and multi-line inline tables with
|
||||
comments and trailing commas. The compliance suite runs in TOML 1.1 mode:
|
||||
214 valid and 467 invalid cases, zero failures. Every TOML 1.0 document
|
||||
parses exactly as before.
|
||||
- `interpres-decode -validate [file ...]`: a validate mode beside the
|
||||
toml-test adapter. It parses each named file, or stdin when none are named,
|
||||
prints one line per invalid document to stderr, and exits 0 when all are
|
||||
|
||||
@@ -1,17 +1,18 @@
|
||||
# interpres
|
||||
|
||||
A TOML 1.0 parser and encoder for Go, written with the standard library alone.
|
||||
`interpres` (Latin for *interpreter*) gives zero-dependency programs an
|
||||
`encoding/json`-style API for reading and writing TOML, and passes the entire
|
||||
official [toml-test](https://github.com/toml-lang/toml-test) suite: 205 valid
|
||||
and 474 invalid cases, zero failures.
|
||||
A TOML 1.0 and 1.1 parser and encoder for Go, written with the standard
|
||||
library alone. `interpres` (Latin for *interpreter*) gives zero-dependency
|
||||
programs an `encoding/json`-style API for reading and writing TOML, and passes
|
||||
the entire official [toml-test](https://github.com/toml-lang/toml-test) suite:
|
||||
214 valid and 467 invalid cases, zero failures.
|
||||
|
||||
## Features
|
||||
|
||||
- **Full TOML 1.0**: bare, quoted and dotted keys; tables and arrays of tables;
|
||||
basic and literal strings including multiline; integers in the four radixes
|
||||
with `_` separators; floats with exponents, `inf` and `nan`; booleans; the
|
||||
four date-time kinds; arrays and inline tables.
|
||||
- **Full TOML 1.0 and 1.1**: bare, quoted and dotted keys; tables and arrays of
|
||||
tables; basic and literal strings including multiline, with the 1.1 `\e` and
|
||||
`\xHH` escapes; integers in the four radixes with `_` separators; floats with
|
||||
exponents, `inf` and `nan`; booleans; the four date-time kinds, seconds
|
||||
optional as of 1.1; arrays and inline tables, multi-line as of 1.1.
|
||||
- **Decoding and encoding**: `Parse` for an untyped tree, `Unmarshal` and
|
||||
`Marshal` for structs and maps, mirroring `encoding/json`.
|
||||
- **Strict decoding**: `NewDecoder().DisallowUnknownFields()` rejects keys that
|
||||
|
||||
+11
-4
@@ -59,24 +59,31 @@ var (
|
||||
"2006-01-02T15:04:05Z07:00",
|
||||
"2006-01-02 15:04:05.999999999Z07:00",
|
||||
"2006-01-02 15:04:05Z07:00",
|
||||
// TOML 1.1 makes the seconds optional.
|
||||
"2006-01-02T15:04Z07:00",
|
||||
"2006-01-02 15:04Z07:00",
|
||||
}
|
||||
localDateTimeLayouts = []string{
|
||||
"2006-01-02T15:04:05.999999999",
|
||||
"2006-01-02T15:04:05",
|
||||
"2006-01-02 15:04:05.999999999",
|
||||
"2006-01-02 15:04:05",
|
||||
"2006-01-02T15:04",
|
||||
"2006-01-02 15:04",
|
||||
}
|
||||
localTimeLayouts = []string{
|
||||
"15:04:05.999999999",
|
||||
"15:04:05",
|
||||
"15:04",
|
||||
}
|
||||
)
|
||||
|
||||
// dateTimeShape enforces the strict TOML grammar (two-digit components) that
|
||||
// time.Parse would otherwise accept loosely (e.g. a single-digit hour).
|
||||
// dateTimeShape enforces the strict TOML grammar (two-digit components,
|
||||
// seconds optional since 1.1, a fraction only after seconds) that time.Parse
|
||||
// would otherwise accept loosely (e.g. a single-digit hour).
|
||||
var dateTimeShape = regexp.MustCompile(
|
||||
`^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}:\d{2}(\.\d+)?([Zz]|[+-]\d{2}:\d{2})?)?$` +
|
||||
`|^\d{2}:\d{2}:\d{2}(\.\d+)?$`,
|
||||
`^\d{4}-\d{2}-\d{2}([Tt ]\d{2}:\d{2}(:\d{2}(\.\d+)?)?([Zz]|[+-]\d{2}:\d{2})?)?$` +
|
||||
`|^\d{2}:\d{2}(:\d{2}(\.\d+)?)?$`,
|
||||
)
|
||||
|
||||
// offsetBounds extracts the numeric offset of a date-time. The ABNF bounds it
|
||||
|
||||
+1
-1
@@ -46,7 +46,7 @@ The cancellable variant of `Unmarshal`.
|
||||
### `func Marshal(v any) ([]byte, error)`
|
||||
|
||||
Encodes a `struct` or `map[string]V` value, or a non-nil pointer to one, into a
|
||||
TOML 1.0 document. The emission rules are in the [Encoding](#encoding) section
|
||||
TOML document. The emission rules are in the [Encoding](#encoding) section
|
||||
below. Equivalent to `MarshalContext(context.Background(), v)`.
|
||||
|
||||
```go
|
||||
|
||||
@@ -6,9 +6,9 @@ source tree; nothing is aspirational.
|
||||
## Overview
|
||||
|
||||
interpres is one public library package, one command, and one example. The
|
||||
library implements the whole of TOML 1.0, decoding and encoding, in the
|
||||
library implements the whole of TOML 1.0 and 1.1, decoding and encoding, in the
|
||||
standard library alone; the command wraps the parser for the toml-test
|
||||
compliance harness, against which it stands at 185 valid and 371 invalid cases
|
||||
compliance harness, against which it stands at 214 valid and 467 invalid cases
|
||||
with zero failures; the example demonstrates the API.
|
||||
|
||||
```mermaid
|
||||
@@ -43,7 +43,7 @@ Inside the library package, one file owns one concern:
|
||||
|
||||
| File | Responsibility |
|
||||
|---|---|
|
||||
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 (table redefinitions, dotted keys, arrays of tables). Reports a 1-based line on failure. |
|
||||
| `parser.go` | The recursive-descent parser. Produces the `map[string]any` tree and enforces the structural rules of TOML 1.0 and 1.1 (table redefinitions, dotted keys, arrays of tables, multi-line inline tables). Reports a 1-based line on failure. |
|
||||
| `number.go` | Strict numeric tokens: integers in the four radixes with `_` separators, and floats including `inf` and `nan`. Rejects leading zeros, misplaced underscores and malformed fractions. |
|
||||
| `datetime.go` | The three local date-time wrapper types and `parseDateTime`, which classifies a token into the four date-time kinds under the strict TOML grammar. |
|
||||
| `decode.go` | Maps the parsed tree onto Go values by reflection: struct fields, maps, slices, scalar conversion with overflow checks, `Unmarshaler` dispatch. |
|
||||
|
||||
@@ -32,6 +32,10 @@ func FuzzParse(f *testing.F) {
|
||||
"x = \"unterminated\n",
|
||||
"[a]\n[a]\n",
|
||||
"n = 0x1_0000_0000_0000_0000\n",
|
||||
// TOML 1.1 forms.
|
||||
"t = 13:37\ndt = 1979-05-27T07:32\nodt = 1979-05-27 07:32Z\n",
|
||||
"esc = \"\\e\\x41\\x7f\\x00\"\n",
|
||||
"m = {\n\ta = 1,\n\tb = [1, 2,],\n\tc = { d = 2 },\n} # close\n",
|
||||
}
|
||||
for _, s := range seeds {
|
||||
f.Add([]byte(s))
|
||||
|
||||
+2
-1
@@ -196,7 +196,8 @@ type Unmarshaler interface {
|
||||
UnmarshalTOML(data any) error
|
||||
}
|
||||
|
||||
// Marshal returns the TOML 1.0 encoding of v.
|
||||
// Marshal returns the TOML encoding of v. The output stays within TOML 1.0,
|
||||
// so it is valid under both TOML 1.0 and 1.1.
|
||||
//
|
||||
// Marshal traverses v using reflection and applies the following rules:
|
||||
//
|
||||
|
||||
+104
-1
@@ -503,7 +503,8 @@ func TestRejectsSpecInvalid(t *testing.T) {
|
||||
"dotted over header": "[a.b]\nx = 1\n[a]\nb.y = 2\n",
|
||||
"table over array": "[[t]]\n[t]\n",
|
||||
"truncated datetime": "a = 2026-01-02T\n",
|
||||
"datetime no seconds": "a = 2026-01-02T07:32\n",
|
||||
// "datetime no seconds" moved to the acceptance tests: TOML 1.1
|
||||
// makes the seconds optional.
|
||||
}
|
||||
for name, doc := range cases {
|
||||
if _, err := Parse([]byte(doc)); err == nil {
|
||||
@@ -540,3 +541,105 @@ host = "h2"
|
||||
t.Errorf("forms[1].smtp.host = %v", h)
|
||||
}
|
||||
}
|
||||
|
||||
// --- TOML 1.1 --------------------------------------------------------------
|
||||
|
||||
func TestParseAcceptsNoSecondsDatetimes(t *testing.T) {
|
||||
tree, err := Parse([]byte(`t = 13:37
|
||||
dt = 1979-05-27T07:32
|
||||
odt1 = 1979-05-27 07:32Z
|
||||
odt2 = 1979-05-27 07:32-07:00
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if got := tree["t"].(LocalTime).String(); got != "13:37:00" {
|
||||
t.Errorf("t = %q, want %q", got, "13:37:00")
|
||||
}
|
||||
if got := tree["dt"].(LocalDateTime).String(); got != "1979-05-27T07:32:00" {
|
||||
t.Errorf("dt = %q, want %q", got, "1979-05-27T07:32:00")
|
||||
}
|
||||
if got := tree["odt1"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00Z" {
|
||||
t.Errorf("odt1 = %q", got)
|
||||
}
|
||||
if got := tree["odt2"].(time.Time).Format(time.RFC3339Nano); got != "1979-05-27T07:32:00-07:00" {
|
||||
t.Errorf("odt2 = %q", got)
|
||||
}
|
||||
// The fraction still requires the seconds it belongs to.
|
||||
if _, err := Parse([]byte("a = 07:32.5\n")); err == nil {
|
||||
t.Error("07:32.5: expected an error, got none")
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAcceptsEscapeAndHexEscapes(t *testing.T) {
|
||||
tree, err := Parse([]byte(`esc = "\e"
|
||||
hex = "\x20\x7f\xf8"
|
||||
nul = "\x00"
|
||||
multi = """\x68\x65"""
|
||||
lit = '\x20'
|
||||
`))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if got := tree["esc"].(string); got != "\x1b" {
|
||||
t.Errorf("esc = %q, want the escape character", got)
|
||||
}
|
||||
if got := tree["hex"].(string); got != " \x7f\u00f8" {
|
||||
t.Errorf("hex = %q", got)
|
||||
}
|
||||
if got := tree["nul"].(string); got != "\x00" {
|
||||
t.Errorf("nul = %q", got)
|
||||
}
|
||||
if got := tree["multi"].(string); got != "he" {
|
||||
t.Errorf("multi = %q", got)
|
||||
}
|
||||
// A literal string carries the sequence verbatim.
|
||||
if got := tree["lit"].(string); got != `\x20` {
|
||||
t.Errorf("lit = %q, want the verbatim sequence", got)
|
||||
}
|
||||
// Two digits exactly; a short or non-hex escape is an error.
|
||||
for _, doc := range []string{`a = "\x4"`, `a = "\x"`, `a = "\xgg"`} {
|
||||
if _, err := Parse([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", doc)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseAcceptsMultilineInlineTables(t *testing.T) {
|
||||
tree, err := Parse([]byte("tbl = {\n\thello = \"world\",\n\tarr = [1,\n\t\t2,\n\t],\n\tsub = {\n\t\tk = 1,\n\t},\n\tbare = 2}\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
tbl := tree["tbl"].(map[string]any)
|
||||
if tbl["hello"] != "world" || tbl["bare"] != int64(2) {
|
||||
t.Fatalf("tbl = %#v", tbl)
|
||||
}
|
||||
if arr := tbl["arr"].([]any); len(arr) != 2 {
|
||||
t.Errorf("arr = %#v", tbl["arr"])
|
||||
}
|
||||
if sub := tbl["sub"].(map[string]any); sub["k"] != int64(1) {
|
||||
t.Errorf("sub = %#v", tbl["sub"])
|
||||
}
|
||||
// Comments inside the table, and a trailing comma at both depths.
|
||||
tree, err = Parse([]byte("m = { # one\n\t# two\n\ta = 1, # three\n\t# four\n}\n"))
|
||||
if err != nil {
|
||||
t.Fatalf("parse with comments: %v", err)
|
||||
}
|
||||
if m := tree["m"].(map[string]any); m["a"] != int64(1) {
|
||||
t.Errorf("m = %#v", m)
|
||||
}
|
||||
// The old single-line shapes keep working, with and without the comma.
|
||||
if _, err := Parse([]byte("a = { b = 1, c = 2 }\n")); err != nil {
|
||||
t.Errorf("single line: %v", err)
|
||||
}
|
||||
// Still rejected: two commas, a missing value, and an unclosed table.
|
||||
for name, doc := range map[string]string{
|
||||
"double comma": "a = { b = 1,, c = 2 }\n",
|
||||
"missing value": "a = {\n\tb =\n}\n",
|
||||
"unterminated": "a = { b = 1,\n",
|
||||
} {
|
||||
if _, err := Parse([]byte(doc)); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -94,7 +94,7 @@ dev:
|
||||
|
||||
# Runs the official toml-test compliance suite against the built adapter; toml-test must be on PATH (go install github.com/toml-lang/toml-test/v2/cmd/toml-test@v2.2.0); not standard because no canonical recipe covers a domain compliance suite.
|
||||
toml-test: build
|
||||
toml-test test -decoder=bin/interpres-decode -toml=1.0
|
||||
toml-test test -decoder=bin/interpres-decode -toml=1.1
|
||||
|
||||
# Coverage report as an HTML map from the gate's profile; not standard because the gate needs only the numeric floor, and a browser artefact is exploration, not a gate.
|
||||
coverage-html: test
|
||||
|
||||
@@ -598,10 +598,16 @@ func (p *parser) readEscape() (rune, error) {
|
||||
return '\f', nil
|
||||
case 'r':
|
||||
return '\r', nil
|
||||
case 'e':
|
||||
// TOML 1.1: the escape character.
|
||||
return '\x1b', nil
|
||||
case '"':
|
||||
return '"', nil
|
||||
case '\\':
|
||||
return '\\', nil
|
||||
case 'x':
|
||||
// TOML 1.1: two hex digits, code points 0x00 through 0xFF.
|
||||
return p.readUnicode(2)
|
||||
case 'u':
|
||||
return p.readUnicode(4)
|
||||
case 'U':
|
||||
@@ -633,7 +639,7 @@ func (p *parser) parseArray() (any, error) {
|
||||
p.next() // '['
|
||||
arr := []any{}
|
||||
for {
|
||||
if err := p.skipArraySpace(); err != nil {
|
||||
if err := p.skipNestedSpace(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if p.eof() {
|
||||
@@ -648,7 +654,7 @@ func (p *parser) parseArray() (any, error) {
|
||||
return nil, err
|
||||
}
|
||||
arr = append(arr, v)
|
||||
if err := p.skipArraySpace(); err != nil {
|
||||
if err := p.skipNestedSpace(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if p.eof() {
|
||||
@@ -670,13 +676,20 @@ func (p *parser) parseInlineTable() (any, error) {
|
||||
p.next() // '{'
|
||||
tbl := map[string]any{}
|
||||
assigned := map[string]bool{}
|
||||
p.skipInline()
|
||||
// TOML 1.1 lets an inline table span lines: interior whitespace includes
|
||||
// newlines and comments, and a trailing comma is allowed before the
|
||||
// closing brace.
|
||||
if err := p.skipNestedSpace(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !p.eof() && p.peek() == '}' {
|
||||
p.next()
|
||||
return tbl, nil
|
||||
}
|
||||
for {
|
||||
p.skipInline()
|
||||
if err := p.skipNestedSpace(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
key, err := p.parseKeyPath()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -720,13 +733,22 @@ func (p *parser) parseInlineTable() (any, error) {
|
||||
dest[leaf] = val
|
||||
assigned[pathKey(path)] = true
|
||||
|
||||
p.skipInline()
|
||||
if err := p.skipNestedSpace(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if p.eof() {
|
||||
return nil, p.errf("unterminated inline table")
|
||||
}
|
||||
switch p.peek() {
|
||||
case ',':
|
||||
p.next()
|
||||
if err := p.skipNestedSpace(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !p.eof() && p.peek() == '}' {
|
||||
p.next()
|
||||
return tbl, nil
|
||||
}
|
||||
case '}':
|
||||
p.next()
|
||||
return tbl, nil
|
||||
@@ -796,8 +818,9 @@ func (p *parser) skipInline() {
|
||||
}
|
||||
}
|
||||
|
||||
// skipArraySpace consumes whitespace, newlines, and comments inside arrays.
|
||||
func (p *parser) skipArraySpace() error {
|
||||
// skipNestedSpace consumes whitespace, newlines, and comments inside a value
|
||||
// container (an array, or an inline table under TOML 1.1).
|
||||
func (p *parser) skipNestedSpace() error {
|
||||
for !p.eof() {
|
||||
switch p.peek() {
|
||||
case ' ', '\t':
|
||||
|
||||
Reference in New Issue
Block a user