feat: carry the byte offset and column in SyntaxError
Test / test (push) Canceled after 2m28s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 23:55:58 +02:00
parent 3cc168f39a
commit 10d49fbe60
5 changed files with 120 additions and 17 deletions
+6
View File
@@ -51,6 +51,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
failure alike. `Valid(data)` reports whether a document parses, nil on failure alike. `Valid(data)` reports whether a document parses, nil on
success and the parse error on failure, the library call the `-validate` success and the parse error on failure, the library call the `-validate`
mode of interpres-decode is built on. mode of interpres-decode is built on.
- `SyntaxError` carries the byte `Offset` the scan stopped at and the 1-based
`Column` on the line, beside the line it always had, and `SourceLine(src)`
renders that line with a caret under the position, for messages shown under
the input. An input that is not valid UTF-8 names the offset of the first
invalid byte in its message. The new fields are additive: a `SyntaxError`
built from a line and a message alone is unchanged.
- `Decoder.UseNumber()` decodes the integers and floats of the document into - `Decoder.UseNumber()` decodes the integers and floats of the document into
`Number`, which carries the literal the document wrote, so `0x1f`, `1_000`, `Number`, which carries the literal the document wrote, so `0x1f`, `1_000`,
`+1.0` and `inf` survive a round trip with their spelling instead of the `+1.0` and `inf` survive a round trip with their spelling instead of the
+53
View File
@@ -1220,3 +1220,56 @@ func TestNumberMethods(t *testing.T) {
} }
} }
} }
func TestSyntaxErrorPosition(t *testing.T) {
src := []byte("alpha = 1\nbeta x = 2\n")
_, err := ParseMap(src)
se, ok := errors.AsType[*SyntaxError](err)
if !ok {
t.Fatalf("err = %v, want a SyntaxError", err)
}
if se.Line != 2 {
t.Errorf("Line = %d, want 2", se.Line)
}
if want := strings.Index(string(src), "x"); se.Offset != want {
t.Errorf("Offset = %d, want %d", se.Offset, want)
}
if se.Column != 6 {
t.Errorf("Column = %d, want 6", se.Column)
}
want := "beta x = 2\n ^"
if got := se.SourceLine(src); got != want {
t.Errorf("SourceLine =\n%s\nwant:\n%s", got, want)
}
}
func TestSyntaxErrorUTF8Offset(t *testing.T) {
src := []byte("a = \"ok\"\nb = \"\xff\xfe\"\n")
_, err := ParseMap(src)
se, ok := errors.AsType[*SyntaxError](err)
if !ok {
t.Fatalf("err = %v, want a SyntaxError", err)
}
if !strings.Contains(se.Msg, "byte offset 14") {
t.Errorf("Msg = %q, want it to name byte offset 14", se.Msg)
}
if se.Offset != 14 {
t.Errorf("Offset = %d, want 14", se.Offset)
}
want := "b = \"\xff\xfe\"\n ^"
if got := se.SourceLine(src); got != want {
t.Errorf("SourceLine =\n%q\nwant:\n%q", got, want)
}
}
func TestSourceLineEdgePositions(t *testing.T) {
src := []byte("a = 1\n")
e := &SyntaxError{Line: 1, Msg: "no position"}
if got, want := e.SourceLine(src), "a = 1\n^"; got != want {
t.Errorf("SourceLine(zero offset) =\n%q\nwant:\n%q", got, want)
}
e = &SyntaxError{Line: 2, Offset: 100, Msg: "past the end"}
if got, want := e.SourceLine(src), "\n^"; got != want {
t.Errorf("SourceLine(offset past end) =\n%q\nwant:\n%q", got, want)
}
}
+14 -7
View File
@@ -642,20 +642,27 @@ sequenceDiagram
## Types ## Types
### `type SyntaxError struct{ Line int; Msg string }` ### `type SyntaxError struct{ Line, Offset, Column int; Msg string }`
Describes a document the parser rejected, with the 1-based `Line` at which it Describes a document the parser rejected: the 1-based `Line` at which it gave
gave up and `Error()` rendering as `interpres: line N: msg`. A malformed up, the `Offset` in bytes the scan stopped at, the 1-based `Column` on that
document is the usual cause; the nesting limit and an input that is not valid line, and `Error()` rendering as `interpres: line N: msg`. A malformed document
UTF-8 report through the same type. Read the structured fields with a type is the usual cause; the nesting limit and an input that is not valid UTF-8
assertion or `errors.AsType`: report through the same type, with the UTF-8 message naming the offset of the
first invalid byte. Read the structured fields with a type assertion or
`errors.AsType`:
```go ```go
if se, ok := errors.AsType[*interpres.SyntaxError](err); ok { if se, ok := errors.AsType[*interpres.SyntaxError](err); ok {
fmt.Println(se.Line, se.Msg) fmt.Println(se.Line, se.Offset, se.Column, se.Msg)
fmt.Println(se.SourceLine(data)) // the line, with a caret under Offset
} }
``` ```
`SourceLine(src)` renders the source line the error points at from `src`,
followed by a caret line marking the column, for messages the reader sees
under the input.
### `type DecodeError struct{ Path []string; Err error }` ### `type DecodeError struct{ Path []string; Err error }`
Wraps a decoding failure with the key path at which it happened. `Path` lists Wraps a decoding failure with the key path at which it happened. `Path` lists
+32 -4
View File
@@ -21,24 +21,52 @@
package interpres package interpres
import ( import (
"bytes"
"context" "context"
"errors" "errors"
"fmt" "fmt"
"os" "os"
"slices" "slices"
"strings"
) )
// A SyntaxError describes a malformed TOML document, including the 1-based // A SyntaxError describes a malformed TOML document. Line is the 1-based line
// line on which the problem was detected. // the problem was detected on. Offset is the byte offset in the input the scan
// stopped at, and Column is the 1-based column on that line; both are new in
// 2.0 and a struct literal that names Line and Msg alone still builds.
type SyntaxError struct { type SyntaxError struct {
Line int Line int
Msg string Offset int
Column int
Msg string
} }
func (e *SyntaxError) Error() string { func (e *SyntaxError) Error() string {
return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg) return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg)
} }
// SourceLine returns the source line the error points at, rendered from src,
// followed by a caret line marking the column. It is meant for a message the
// reader sees under the input:
//
// port = = 8080
// ^
//
// The caret sits at Offset when it falls inside src, and at the start of the
// line when the error carries no position.
func (e *SyntaxError) SourceLine(src []byte) string {
off := min(e.Offset, len(src))
start := 0
if i := bytes.LastIndexByte(src[:off], '\n'); i >= 0 {
start = i + 1
}
end := len(src)
if i := bytes.IndexByte(src[start:], '\n'); i >= 0 {
end = start + i
}
return string(src[start:end]) + "\n" + strings.Repeat(" ", off-start) + "^"
}
// A DecodeError wraps a decoding failure with the key path at which it // A DecodeError wraps a decoding failure with the key path at which it
// happened. Path lists one segment per level from the document root, the // happened. Path lists one segment per level from the document root, the
// outermost key first: a key contributes its name and an array element its // outermost key first: a key contributes its name and an array element its
+15 -6
View File
@@ -4,6 +4,7 @@
package interpres package interpres
import ( import (
"bytes"
"context" "context"
"fmt" "fmt"
"strconv" "strconv"
@@ -652,7 +653,7 @@ func (p *parser) parseKeyComponent() (string, error) {
// does not decode names that, before any grammar message can. // does not decode names that, before any grammar message can.
if !p.eof() && p.peek() >= utf8.RuneSelf { if !p.eof() && p.peek() >= utf8.RuneSelf {
if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 { if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in key") return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos)
} }
} }
if p.pos == start { if p.pos == start {
@@ -707,7 +708,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value") return nil, p.errf("expected a value")
} }
if hasHighByte(tok) && !utf8.ValidString(tok) { if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value") return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
} }
// A date may be followed by a space and a time, forming one date-time. // A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' { if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -885,7 +886,7 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
} }
r, size := utf8.DecodeRune(p.src[p.pos:]) r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 { if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 in string") return p.errf("invalid UTF-8 in string at byte offset %d", p.pos)
} }
p.pos += size p.pos += size
b.WriteRune(r) b.WriteRune(r)
@@ -1393,7 +1394,7 @@ func (p *parser) skipComment() (string, error) {
default: default:
r, size := utf8.DecodeRune(p.src[p.pos:]) r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 { if r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in comment") return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos)
} }
p.pos += size p.pos += size
} }
@@ -1451,13 +1452,21 @@ func (p *parser) expectLineEnd() error {
} }
r, size := utf8.DecodeRune(p.src[p.pos:]) r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 { if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 after value") return p.errf("invalid UTF-8 after value at byte offset %d", p.pos)
} }
return p.errf("unexpected %q after value", string(r)) return p.errf("unexpected %q after value", string(r))
} }
// errf builds the SyntaxError with the position the scan stopped at: the line,
// the byte offset in the input, and the 1-based column on that line. The
// offset is the cursor, which on an escape or a delimiter run sits just after
// the bytes that caused the complaint; SourceLine renders the caret there.
func (p *parser) errf(format string, args ...any) error { func (p *parser) errf(format string, args ...any) error {
return &SyntaxError{Line: p.line, Msg: fmt.Sprintf(format, args...)} col := p.pos + 1
if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 {
col = p.pos - start
}
return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)}
} }
// pathKey joins key components with a NUL separator so a dotted path can be // pathKey joins key components with a NUL separator so a dotted path can be