fix(parse): reject the lenient grammar edges and name out-of-range date-times

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-22 21:15:00 +02:00
parent 8f2b26bd33
commit 2efdb2d059
3 changed files with 212 additions and 36 deletions
+72 -21
View File
@@ -48,6 +48,13 @@ type parser struct {
dotted map[string]bool
arrays map[string]bool
// scopeMarks records the definition-map entries added under an array of
// tables, keyed by that array's path, so a new element's reset drops
// exactly what the previous element added. Without it the reset scans
// every map for the prefix, which a document with many elements and many
// definitions outside them turns quadratic.
scopeMarks map[string][]string
currentPath []string
// keys interns key strings: a document that repeats a key across
@@ -196,6 +203,7 @@ func (p *parser) markHeader(pk string) {
p.headers = make(map[string]bool, 4)
}
p.headers[pk] = true
p.trackScope(pk)
}
func (p *parser) markFrozen(pk string) {
@@ -203,6 +211,7 @@ func (p *parser) markFrozen(pk string) {
p.frozen = make(map[string]bool, 4)
}
p.frozen[pk] = true
p.trackScope(pk)
}
func (p *parser) markDotted(pk string) {
@@ -210,6 +219,7 @@ func (p *parser) markDotted(pk string) {
p.dotted = make(map[string]bool, 4)
}
p.dotted[pk] = true
p.trackScope(pk)
}
func (p *parser) markArray(pk string) {
@@ -217,6 +227,22 @@ func (p *parser) markArray(pk string) {
p.arrays = make(map[string]bool, 2)
}
p.arrays[pk] = true
p.trackScope(pk)
}
// trackScope records a definition entry under every array of tables it falls
// inside, so resetScopeUnder can drop it when a later element opens. An entry
// under no array, such as every definition before the first header, needs no
// record: no reset can ever name it.
func (p *parser) trackScope(pk string) {
for arr := range p.arrays {
if strings.HasPrefix(pk, arr+"\x00") {
if p.scopeMarks == nil {
p.scopeMarks = make(map[string][]string, 2)
}
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
}
}
}
// internKey returns the shared string for key bytes. The lookup works on the
@@ -482,9 +508,14 @@ func (p *parser) parseKeyValue() error {
if p.doc != nil {
node := p.currentNode
if len(rest) > 0 {
// The tables a dotted key builds hold the position of a line, so
// the write side marks them and gives each leaf back as a dotted
// key rather than a header that would swallow the lines after it.
node = node.addTable(first, dests[0])
node.dotted = true
for i, k := range rest[:len(rest)-1] {
node = node.addTable(k, dests[i+1])
node.dotted = true
}
}
_, inline := val.(map[string]any)
@@ -570,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) {
// resetScopeUnder forgets the definition records nested under key, which
// belong to the previous element of an array of tables: headers, frozen
// inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element.
// fresh in the new element. The records to drop are the ones the element
// added, which scopeMarks holds; the array's own entry, and everything
// outside it, keep their place.
func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00"
p.resetMapUnder(p.headers, prefix)
p.resetMapUnder(p.frozen, prefix)
p.resetMapUnder(p.dotted, prefix)
p.resetMapUnder(p.arrays, prefix)
}
// resetMapUnder deletes the entries m holds under prefix. An empty or
// unallocated map holds none, so the common case walks nothing.
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
if len(m) == 0 {
return
pk := pathKey(key)
for _, k := range p.scopeMarks[pk] {
delete(p.headers, k)
delete(p.frozen, k)
delete(p.dotted, k)
delete(p.arrays, k)
}
for k := range m {
if strings.HasPrefix(k, prefix) {
delete(m, k)
}
if p.scopeMarks != nil {
p.scopeMarks[pk] = nil
}
}
@@ -708,7 +733,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value")
}
if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
}
// A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -719,7 +744,11 @@ func (p *parser) parseAtom() (any, error) {
tok = tok + " " + string(p.src[timeStart:p.pos])
}
}
if v, ok := parseDateTime(tok); ok {
v, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
return v, nil
}
v, err := decodeNumber(tok)
@@ -758,6 +787,20 @@ func hasHighByte(s string) bool {
return false
}
// invalidUTF8Offset returns the offset of the first byte in s that does not
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
// that is invalid rather than the end of the token around it.
func invalidUTF8Offset(s string) int {
for i := 0; i < len(s); {
r, size := utf8.DecodeRuneInString(s[i:])
if r == utf8.RuneError && size == 1 {
return i
}
i += size
}
return -1
}
// --- strings ---------------------------------------------------------------
func (p *parser) parseBasicString() (string, error) {
@@ -895,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed.
// A newline immediately after the opening delimiter is trimmed, and it is
// a newline: a bare CR here is the bare-CR error like anywhere else, not
// a newline to trim.
if !p.eof() && p.peek() == '\r' {
if next, ok := p.peekAt(1); !ok || next != '\n' {
return "", p.errf("bare carriage return is not allowed in a string")
}
p.pos++
}
if !p.eof() && p.peek() == '\n' {
@@ -1047,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) {
}
hex := string(p.src[p.pos : p.pos+n])
p.pos += n
v, err := strconv.ParseInt(hex, 16, 64)
// ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
// read would let "\U-0000001" through the range checks below only to
// write U+FFFD for a document the grammar rejects.
v, err := strconv.ParseUint(hex, 16, 32)
if err != nil {
return 0, p.errf("invalid unicode escape \\%s", hex)
}