From 10d49fbe607c76924c7bc2338624e075e725945a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Mon, 21 Sep 2026 23:55:58 +0200 Subject: [PATCH] feat: carry the byte offset and column in SyntaxError Assisted-by: GLM 5.3 Flash --- CHANGELOG.md | 6 ++++++ decode_test.go | 53 ++++++++++++++++++++++++++++++++++++++++++++++++++ docs/API.md | 21 +++++++++++++------- interpres.go | 36 ++++++++++++++++++++++++++++++---- parser.go | 21 ++++++++++++++------ 5 files changed, 120 insertions(+), 17 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3cd9903..465b809 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -51,6 +51,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 failure alike. `Valid(data)` reports whether a document parses, nil on success and the parse error on failure, the library call the `-validate` mode of interpres-decode is built on. +- `SyntaxError` carries the byte `Offset` the scan stopped at and the 1-based + `Column` on the line, beside the line it always had, and `SourceLine(src)` + renders that line with a caret under the position, for messages shown under + the input. An input that is not valid UTF-8 names the offset of the first + invalid byte in its message. The new fields are additive: a `SyntaxError` + built from a line and a message alone is unchanged. - `Decoder.UseNumber()` decodes the integers and floats of the document into `Number`, which carries the literal the document wrote, so `0x1f`, `1_000`, `+1.0` and `inf` survive a round trip with their spelling instead of the diff --git a/decode_test.go b/decode_test.go index dc43d7a..4271748 100644 --- a/decode_test.go +++ b/decode_test.go @@ -1220,3 +1220,56 @@ func TestNumberMethods(t *testing.T) { } } } + +func TestSyntaxErrorPosition(t *testing.T) { + src := []byte("alpha = 1\nbeta x = 2\n") + _, err := ParseMap(src) + se, ok := errors.AsType[*SyntaxError](err) + if !ok { + t.Fatalf("err = %v, want a SyntaxError", err) + } + if se.Line != 2 { + t.Errorf("Line = %d, want 2", se.Line) + } + if want := strings.Index(string(src), "x"); se.Offset != want { + t.Errorf("Offset = %d, want %d", se.Offset, want) + } + if se.Column != 6 { + t.Errorf("Column = %d, want 6", se.Column) + } + want := "beta x = 2\n ^" + if got := se.SourceLine(src); got != want { + t.Errorf("SourceLine =\n%s\nwant:\n%s", got, want) + } +} + +func TestSyntaxErrorUTF8Offset(t *testing.T) { + src := []byte("a = \"ok\"\nb = \"\xff\xfe\"\n") + _, err := ParseMap(src) + se, ok := errors.AsType[*SyntaxError](err) + if !ok { + t.Fatalf("err = %v, want a SyntaxError", err) + } + if !strings.Contains(se.Msg, "byte offset 14") { + t.Errorf("Msg = %q, want it to name byte offset 14", se.Msg) + } + if se.Offset != 14 { + t.Errorf("Offset = %d, want 14", se.Offset) + } + want := "b = \"\xff\xfe\"\n ^" + if got := se.SourceLine(src); got != want { + t.Errorf("SourceLine =\n%q\nwant:\n%q", got, want) + } +} + +func TestSourceLineEdgePositions(t *testing.T) { + src := []byte("a = 1\n") + e := &SyntaxError{Line: 1, Msg: "no position"} + if got, want := e.SourceLine(src), "a = 1\n^"; got != want { + t.Errorf("SourceLine(zero offset) =\n%q\nwant:\n%q", got, want) + } + e = &SyntaxError{Line: 2, Offset: 100, Msg: "past the end"} + if got, want := e.SourceLine(src), "\n^"; got != want { + t.Errorf("SourceLine(offset past end) =\n%q\nwant:\n%q", got, want) + } +} diff --git a/docs/API.md b/docs/API.md index 1d0bb70..a6ed4c7 100644 --- a/docs/API.md +++ b/docs/API.md @@ -642,20 +642,27 @@ sequenceDiagram ## Types -### `type SyntaxError struct{ Line int; Msg string }` +### `type SyntaxError struct{ Line, Offset, Column int; Msg string }` -Describes a document the parser rejected, with the 1-based `Line` at which it -gave up and `Error()` rendering as `interpres: line N: msg`. A malformed -document is the usual cause; the nesting limit and an input that is not valid -UTF-8 report through the same type. Read the structured fields with a type -assertion or `errors.AsType`: +Describes a document the parser rejected: the 1-based `Line` at which it gave +up, the `Offset` in bytes the scan stopped at, the 1-based `Column` on that +line, and `Error()` rendering as `interpres: line N: msg`. A malformed document +is the usual cause; the nesting limit and an input that is not valid UTF-8 +report through the same type, with the UTF-8 message naming the offset of the +first invalid byte. Read the structured fields with a type assertion or +`errors.AsType`: ```go if se, ok := errors.AsType[*interpres.SyntaxError](err); ok { - fmt.Println(se.Line, se.Msg) + fmt.Println(se.Line, se.Offset, se.Column, se.Msg) + fmt.Println(se.SourceLine(data)) // the line, with a caret under Offset } ``` +`SourceLine(src)` renders the source line the error points at from `src`, +followed by a caret line marking the column, for messages the reader sees +under the input. + ### `type DecodeError struct{ Path []string; Err error }` Wraps a decoding failure with the key path at which it happened. `Path` lists diff --git a/interpres.go b/interpres.go index 2ca543b..2307123 100644 --- a/interpres.go +++ b/interpres.go @@ -21,24 +21,52 @@ package interpres import ( + "bytes" "context" "errors" "fmt" "os" "slices" + "strings" ) -// A SyntaxError describes a malformed TOML document, including the 1-based -// line on which the problem was detected. +// A SyntaxError describes a malformed TOML document. Line is the 1-based line +// the problem was detected on. Offset is the byte offset in the input the scan +// stopped at, and Column is the 1-based column on that line; both are new in +// 2.0 and a struct literal that names Line and Msg alone still builds. type SyntaxError struct { - Line int - Msg string + Line int + Offset int + Column int + Msg string } func (e *SyntaxError) Error() string { return fmt.Sprintf("interpres: line %d: %s", e.Line, e.Msg) } +// SourceLine returns the source line the error points at, rendered from src, +// followed by a caret line marking the column. It is meant for a message the +// reader sees under the input: +// +// port = = 8080 +// ^ +// +// The caret sits at Offset when it falls inside src, and at the start of the +// line when the error carries no position. +func (e *SyntaxError) SourceLine(src []byte) string { + off := min(e.Offset, len(src)) + start := 0 + if i := bytes.LastIndexByte(src[:off], '\n'); i >= 0 { + start = i + 1 + } + end := len(src) + if i := bytes.IndexByte(src[start:], '\n'); i >= 0 { + end = start + i + } + return string(src[start:end]) + "\n" + strings.Repeat(" ", off-start) + "^" +} + // A DecodeError wraps a decoding failure with the key path at which it // happened. Path lists one segment per level from the document root, the // outermost key first: a key contributes its name and an array element its diff --git a/parser.go b/parser.go index 87c0214..9daa0f2 100644 --- a/parser.go +++ b/parser.go @@ -4,6 +4,7 @@ package interpres import ( + "bytes" "context" "fmt" "strconv" @@ -652,7 +653,7 @@ func (p *parser) parseKeyComponent() (string, error) { // does not decode names that, before any grammar message can. if !p.eof() && p.peek() >= utf8.RuneSelf { if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 { - return "", p.errf("invalid UTF-8 in key") + return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos) } } if p.pos == start { @@ -707,7 +708,7 @@ func (p *parser) parseAtom() (any, error) { return nil, p.errf("expected a value") } if hasHighByte(tok) && !utf8.ValidString(tok) { - return nil, p.errf("invalid UTF-8 in value") + return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos) } // A date may be followed by a space and a time, forming one date-time. if isDateToken(tok) && !p.eof() && p.peek() == ' ' { @@ -885,7 +886,7 @@ func (p *parser) writeContentRune(b *strings.Builder) error { } r, size := utf8.DecodeRune(p.src[p.pos:]) if r == utf8.RuneError && size == 1 { - return p.errf("invalid UTF-8 in string") + return p.errf("invalid UTF-8 in string at byte offset %d", p.pos) } p.pos += size b.WriteRune(r) @@ -1393,7 +1394,7 @@ func (p *parser) skipComment() (string, error) { default: r, size := utf8.DecodeRune(p.src[p.pos:]) if r == utf8.RuneError && size == 1 { - return "", p.errf("invalid UTF-8 in comment") + return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos) } p.pos += size } @@ -1451,13 +1452,21 @@ func (p *parser) expectLineEnd() error { } r, size := utf8.DecodeRune(p.src[p.pos:]) if r == utf8.RuneError && size == 1 { - return p.errf("invalid UTF-8 after value") + return p.errf("invalid UTF-8 after value at byte offset %d", p.pos) } return p.errf("unexpected %q after value", string(r)) } +// errf builds the SyntaxError with the position the scan stopped at: the line, +// the byte offset in the input, and the 1-based column on that line. The +// offset is the cursor, which on an escape or a delimiter run sits just after +// the bytes that caused the complaint; SourceLine renders the caret there. func (p *parser) errf(format string, args ...any) error { - return &SyntaxError{Line: p.line, Msg: fmt.Sprintf(format, args...)} + col := p.pos + 1 + if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 { + col = p.pos - start + } + return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)} } // pathKey joins key components with a NUL separator so a dotted path can be