feat: carry the byte offset and column in SyntaxError
Test / test (push) Canceled after 2m28s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-21 23:55:58 +02:00
parent 3cc168f39a
commit 10d49fbe60
5 changed files with 120 additions and 17 deletions
+15 -6
View File
@@ -4,6 +4,7 @@
package interpres
import (
"bytes"
"context"
"fmt"
"strconv"
@@ -652,7 +653,7 @@ func (p *parser) parseKeyComponent() (string, error) {
// does not decode names that, before any grammar message can.
if !p.eof() && p.peek() >= utf8.RuneSelf {
if r, size := utf8.DecodeRune(p.src[p.pos:]); r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in key")
return "", p.errf("invalid UTF-8 in key at byte offset %d", p.pos)
}
}
if p.pos == start {
@@ -707,7 +708,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value")
}
if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value")
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
}
// A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -885,7 +886,7 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 in string")
return p.errf("invalid UTF-8 in string at byte offset %d", p.pos)
}
p.pos += size
b.WriteRune(r)
@@ -1393,7 +1394,7 @@ func (p *parser) skipComment() (string, error) {
default:
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return "", p.errf("invalid UTF-8 in comment")
return "", p.errf("invalid UTF-8 in comment at byte offset %d", p.pos)
}
p.pos += size
}
@@ -1451,13 +1452,21 @@ func (p *parser) expectLineEnd() error {
}
r, size := utf8.DecodeRune(p.src[p.pos:])
if r == utf8.RuneError && size == 1 {
return p.errf("invalid UTF-8 after value")
return p.errf("invalid UTF-8 after value at byte offset %d", p.pos)
}
return p.errf("unexpected %q after value", string(r))
}
// errf builds the SyntaxError with the position the scan stopped at: the line,
// the byte offset in the input, and the 1-based column on that line. The
// offset is the cursor, which on an escape or a delimiter run sits just after
// the bytes that caused the complaint; SourceLine renders the caret there.
func (p *parser) errf(format string, args ...any) error {
return &SyntaxError{Line: p.line, Msg: fmt.Sprintf(format, args...)}
col := p.pos + 1
if start := bytes.LastIndexByte(p.src[:p.pos], '\n'); start >= 0 {
col = p.pos - start
}
return &SyntaxError{Line: p.line, Offset: p.pos, Column: col, Msg: fmt.Sprintf(format, args...)}
}
// pathKey joins key components with a NUL separator so a dotted path can be