fix(parse): reject the lenient grammar edges and name out-of-range date-times
Assisted-by: GLM 5.3
This commit is contained in:
@@ -48,6 +48,13 @@ type parser struct {
|
||||
dotted map[string]bool
|
||||
arrays map[string]bool
|
||||
|
||||
// scopeMarks records the definition-map entries added under an array of
|
||||
// tables, keyed by that array's path, so a new element's reset drops
|
||||
// exactly what the previous element added. Without it the reset scans
|
||||
// every map for the prefix, which a document with many elements and many
|
||||
// definitions outside them turns quadratic.
|
||||
scopeMarks map[string][]string
|
||||
|
||||
currentPath []string
|
||||
|
||||
// keys interns key strings: a document that repeats a key across
|
||||
@@ -196,6 +203,7 @@ func (p *parser) markHeader(pk string) {
|
||||
p.headers = make(map[string]bool, 4)
|
||||
}
|
||||
p.headers[pk] = true
|
||||
p.trackScope(pk)
|
||||
}
|
||||
|
||||
func (p *parser) markFrozen(pk string) {
|
||||
@@ -203,6 +211,7 @@ func (p *parser) markFrozen(pk string) {
|
||||
p.frozen = make(map[string]bool, 4)
|
||||
}
|
||||
p.frozen[pk] = true
|
||||
p.trackScope(pk)
|
||||
}
|
||||
|
||||
func (p *parser) markDotted(pk string) {
|
||||
@@ -210,6 +219,7 @@ func (p *parser) markDotted(pk string) {
|
||||
p.dotted = make(map[string]bool, 4)
|
||||
}
|
||||
p.dotted[pk] = true
|
||||
p.trackScope(pk)
|
||||
}
|
||||
|
||||
func (p *parser) markArray(pk string) {
|
||||
@@ -217,6 +227,22 @@ func (p *parser) markArray(pk string) {
|
||||
p.arrays = make(map[string]bool, 2)
|
||||
}
|
||||
p.arrays[pk] = true
|
||||
p.trackScope(pk)
|
||||
}
|
||||
|
||||
// trackScope records a definition entry under every array of tables it falls
|
||||
// inside, so resetScopeUnder can drop it when a later element opens. An entry
|
||||
// under no array, such as every definition before the first header, needs no
|
||||
// record: no reset can ever name it.
|
||||
func (p *parser) trackScope(pk string) {
|
||||
for arr := range p.arrays {
|
||||
if strings.HasPrefix(pk, arr+"\x00") {
|
||||
if p.scopeMarks == nil {
|
||||
p.scopeMarks = make(map[string][]string, 2)
|
||||
}
|
||||
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// internKey returns the shared string for key bytes. The lookup works on the
|
||||
@@ -482,9 +508,14 @@ func (p *parser) parseKeyValue() error {
|
||||
if p.doc != nil {
|
||||
node := p.currentNode
|
||||
if len(rest) > 0 {
|
||||
// The tables a dotted key builds hold the position of a line, so
|
||||
// the write side marks them and gives each leaf back as a dotted
|
||||
// key rather than a header that would swallow the lines after it.
|
||||
node = node.addTable(first, dests[0])
|
||||
node.dotted = true
|
||||
for i, k := range rest[:len(rest)-1] {
|
||||
node = node.addTable(k, dests[i+1])
|
||||
node.dotted = true
|
||||
}
|
||||
}
|
||||
_, inline := val.(map[string]any)
|
||||
@@ -570,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) {
|
||||
// resetScopeUnder forgets the definition records nested under key, which
|
||||
// belong to the previous element of an array of tables: headers, frozen
|
||||
// inline tables, dotted-key paths, and nested arrays of tables all start
|
||||
// fresh in the new element.
|
||||
// fresh in the new element. The records to drop are the ones the element
|
||||
// added, which scopeMarks holds; the array's own entry, and everything
|
||||
// outside it, keep their place.
|
||||
func (p *parser) resetScopeUnder(key []string) {
|
||||
prefix := pathKey(key) + "\x00"
|
||||
p.resetMapUnder(p.headers, prefix)
|
||||
p.resetMapUnder(p.frozen, prefix)
|
||||
p.resetMapUnder(p.dotted, prefix)
|
||||
p.resetMapUnder(p.arrays, prefix)
|
||||
}
|
||||
|
||||
// resetMapUnder deletes the entries m holds under prefix. An empty or
|
||||
// unallocated map holds none, so the common case walks nothing.
|
||||
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
|
||||
if len(m) == 0 {
|
||||
return
|
||||
pk := pathKey(key)
|
||||
for _, k := range p.scopeMarks[pk] {
|
||||
delete(p.headers, k)
|
||||
delete(p.frozen, k)
|
||||
delete(p.dotted, k)
|
||||
delete(p.arrays, k)
|
||||
}
|
||||
for k := range m {
|
||||
if strings.HasPrefix(k, prefix) {
|
||||
delete(m, k)
|
||||
}
|
||||
if p.scopeMarks != nil {
|
||||
p.scopeMarks[pk] = nil
|
||||
}
|
||||
}
|
||||
|
||||
@@ -708,7 +733,7 @@ func (p *parser) parseAtom() (any, error) {
|
||||
return nil, p.errf("expected a value")
|
||||
}
|
||||
if hasHighByte(tok) && !utf8.ValidString(tok) {
|
||||
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos)
|
||||
return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
|
||||
}
|
||||
// A date may be followed by a space and a time, forming one date-time.
|
||||
if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
|
||||
@@ -719,7 +744,11 @@ func (p *parser) parseAtom() (any, error) {
|
||||
tok = tok + " " + string(p.src[timeStart:p.pos])
|
||||
}
|
||||
}
|
||||
if v, ok := parseDateTime(tok); ok {
|
||||
v, isDT, dterr := parseDateTime(tok)
|
||||
if dterr != nil {
|
||||
return nil, p.errf("%s", dterr)
|
||||
}
|
||||
if isDT {
|
||||
return v, nil
|
||||
}
|
||||
v, err := decodeNumber(tok)
|
||||
@@ -758,6 +787,20 @@ func hasHighByte(s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// invalidUTF8Offset returns the offset of the first byte in s that does not
|
||||
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
|
||||
// that is invalid rather than the end of the token around it.
|
||||
func invalidUTF8Offset(s string) int {
|
||||
for i := 0; i < len(s); {
|
||||
r, size := utf8.DecodeRuneInString(s[i:])
|
||||
if r == utf8.RuneError && size == 1 {
|
||||
return i
|
||||
}
|
||||
i += size
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// --- strings ---------------------------------------------------------------
|
||||
|
||||
func (p *parser) parseBasicString() (string, error) {
|
||||
@@ -895,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
|
||||
|
||||
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
|
||||
p.skipN(3) // opening delimiter
|
||||
// A newline immediately after the opening delimiter is trimmed.
|
||||
// A newline immediately after the opening delimiter is trimmed, and it is
|
||||
// a newline: a bare CR here is the bare-CR error like anywhere else, not
|
||||
// a newline to trim.
|
||||
if !p.eof() && p.peek() == '\r' {
|
||||
if next, ok := p.peekAt(1); !ok || next != '\n' {
|
||||
return "", p.errf("bare carriage return is not allowed in a string")
|
||||
}
|
||||
p.pos++
|
||||
}
|
||||
if !p.eof() && p.peek() == '\n' {
|
||||
@@ -1047,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) {
|
||||
}
|
||||
hex := string(p.src[p.pos : p.pos+n])
|
||||
p.pos += n
|
||||
v, err := strconv.ParseInt(hex, 16, 64)
|
||||
// ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
|
||||
// read would let "\U-0000001" through the range checks below only to
|
||||
// write U+FFFD for a document the grammar rejects.
|
||||
v, err := strconv.ParseUint(hex, 16, 32)
|
||||
if err != nil {
|
||||
return 0, p.errf("invalid unicode escape \\%s", hex)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user