feat: carry the byte offset and column in SyntaxError
Test / test (push) Canceled after 2m28s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 23:55:58 +02:00
parent 3cc168f39a
commit 10d49fbe60
5 changed files with 120 additions and 17 deletions
+6
View File
@@ -51,6 +51,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
failure alike. `Valid(data)` reports whether a document parses, nil on
success and the parse error on failure, the library call the `-validate`
mode of interpres-decode is built on.
- `SyntaxError` carries the byte `Offset` the scan stopped at and the 1-based
`Column` on the line, beside the line it always had, and `SourceLine(src)`
renders that line with a caret under the position, for messages shown under
the input. An input that is not valid UTF-8 names the offset of the first
invalid byte in its message. The new fields are additive: a `SyntaxError`
built from a line and a message alone is unchanged.
- `Decoder.UseNumber()` decodes the integers and floats of the document into
`Number`, which carries the literal the document wrote, so `0x1f`, `1_000`,
`+1.0` and `inf` survive a round trip with their spelling instead of the
+53
View File
@@ -1220,3 +1220,56 @@ func TestNumberMethods(t *testing.T) {
}
}
}
func TestSyntaxErrorPosition(t *testing.T) {
src := []byte("alpha = 1\nbeta x = 2\n")
_, err := ParseMap(src)
se, ok := errors.AsType[*SyntaxError](err)
if !ok {
t.Fatalf("err = %v, want a SyntaxError", err)
}
if se.Line != 2 {
t.Errorf("Line = %d, want 2", se.Line)
}
if want := strings.Index(string(src), "x"); se.Offset != want {
t.Errorf("Offset = %d, want %d", se.Offset, want)
}
if se.Column != 6 {
t.Errorf("Column = %d, want 6", se.Column)
}
want := "beta x = 2\n ^"
if got := se.SourceLine(src); got != want {
t.Errorf("SourceLine =\n%s\nwant:\n%s", got, want)
}
}
func TestSyntaxErrorUTF8Offset(t *testing.T) {
src := []byte("a = \"ok\"\nb = \"\xff\xfe\"\n")
_, err := ParseMap(src)
se, ok := errors.AsType[*SyntaxError](err)
if !ok {
t.Fatalf("err = %v, want a SyntaxError", err)
}
if !strings.Contains(se.Msg, "byte offset 14") {
t.Errorf("Msg = %q, want it to name byte offset 14", se.Msg)
}
if se.Offset != 14 {
t.Errorf("Offset = %d, want 14", se.Offset)
}
want := "b = \"\xff\xfe\"\n ^"
if got := se.SourceLine(src); got != want {
t.Errorf("SourceLine =\n%q\nwant:\n%q", got, want)
}
}
func TestSourceLineEdgePositions(t *testing.T) {
src := []byte("a = 1\n")
e := &SyntaxError{Line: 1, Msg: "no position"}
if got, want := e.SourceLine(src), "a = 1\n^"; got != want {
t.Errorf("SourceLine(zero offset) =\n%q\nwant:\n%q", got, want)
}
e = &SyntaxError{Line: 2, Offset: 100, Msg: "past the end"}
if got, want := e.SourceLine(src), "\n^"; got != want {
t.Errorf("SourceLine(offset past end) =\n%q\nwant:\n%q", got, want)
}
}
+14 -7
View File
@@ -642,20 +642,27 @@ sequenceDiagram
## Types
### `type SyntaxError struct{ Line int; Msg string }`
### `type SyntaxError struct{ Line, Offset, Column int; Msg string }`
Describes a document the parser rejected, with the 1-based `Line` at which it
gave up and `Error()` rendering as `interpres: line N: msg`. A malformed
document is the usual cause; the nesting limit and an input that is not valid
UTF-8 report through the same type. Read the structured fields with a type
assertion or `errors.AsType`:
Describes a document the parser rejected: the 1-based `Line` at which it gave
up, the `Offset` in bytes the scan stopped at, the 1-based `Column` on that
line, and `Error()` rendering as `interpres: line N: msg`. A malformed document
is the usual cause; the nesting limit and an input that is not valid UTF-8
report through the same type, with the UTF-8 message naming the offset of the
first invalid byte. Read the structured fields with a type assertion or
`errors.AsType`:
```go
if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
fmt.Println(se.Line, se.Msg)
fmt.Println(se.Line, se.Offset, se.Column, se.Msg)
fmt.Println(se.SourceLine(data)) // the line, with a caret under Offset
}
```
`SourceLine(src)` renders the source line the error points at from `src`,
followed by a caret line marking the column, for messages the reader sees
under the input.
### `type DecodeError struct{ Path []string; Err error }`
Wraps a decoding failure with the key path at which it happened. `Path` lists
+32 -4
View File
@@ -21,24 +21,52 @@
package interpres
import (
"bytes"
"context"
"errors"
"fmt"
"os"
"slices"
"strings"
)
// A SyntaxError describes a malformed TOML document, including the 1-based
// line on which the problem was detected.
// A SyntaxError describes a malformed TOML document. Line is the 1-based line
// the problem was detected on. Offset is the byte offset in the input the scan
// stopped at, and Column is the 1-based column on that line; both are new in
// 2.0 and a struct literal that names Line and Msg alone still builds.
type SyntaxError struct {
Line int
Msg string
Line int
Offset int
Column int
Msg string
}
func (e *SyntaxError) Error() string {
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
}
// SourceLine returns the source line the error points at, rendered from src,
// followed by a caret line marking the column. It is meant for a message the
// reader sees under the input:
//
// port = = 8080
// ^
//
// The caret sits at Offset when it falls inside src, and at the start of the
// line when the error carries no position.
func (e *SyntaxError) SourceLine(src []byte) string {
off := min(e.Offset, len(src))
start := 0
if i := bytes.LastIndexByte(src[:off], '\n'); i >= 0 {
start = i + 1
}
end := len(src)
if i := bytes.IndexByte(src[start:], '\n'); i >= 0 {
end = start + i
}
return string(src[start:end]) + "\n" + strings.Repeat(" ", off-start) + "^"
}
// A DecodeError wraps a decoding failure with the key path at which it
// happened. Path lists one segment per level from the document root, the
// outermost key first: a key contributes its name and an array element its
+15 -6
View File
@@ -4,6 +4,7 @@
package interpres
import (
"bytes"
"context"
"fmt"
"strconv"
@@ -652,7 +653,7 @@ func (p *parser) parseKeyComponent() (string, error) {
// does not decode names that, before any grammar message can.
if !p.eof() && p.peek() >= utf8.RuneSelf {
if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in key")
return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos)
}
}
if p.pos == start {
@@ -707,7 +708,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value")
}
if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value")
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
}
// A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -885,7 +886,7 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 in string")
return p.errf("invalid UTF-8 in string at byte offset %d", p.pos)
}
p.pos += size
b.WriteRune(r)
@@ -1393,7 +1394,7 @@ func (p *parser) skipComment() (string, error) {
default:
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in comment")
return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos)
}
p.pos += size
}
@@ -1451,13 +1452,21 @@ func (p *parser) expectLineEnd() error {
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 after value")
return p.errf("invalid UTF-8 after value at byte offset %d", p.pos)
}
return p.errf("unexpected %q after value", string(r))
}
// errf builds the SyntaxError with the position the scan stopped at: the line,
// the byte offset in the input, and the 1-based column on that line. The
// offset is the cursor, which on an escape or a delimiter run sits just after
// the bytes that caused the complaint; SourceLine renders the caret there.
func (p *parser) errf(format string, args ...any) error {
return &SyntaxError{Line: p.line, Msg: fmt.Sprintf(format, args...)}
col := p.pos + 1
if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 {
col = p.pos - start
}
return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)}
}
// pathKey joins key components with a NUL separator so a dotted path can be