fix(parse): reject the lenient grammar edges and name out-of-range date-times

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-22 21:15:00 +02:00
parent 8f2b26bd33
commit 2efdb2d059
3 changed files with 212 additions and 36 deletions
+32 -15
View File
@@ -79,7 +79,9 @@ func clockString(t time.Time) string {
// offsetString renders an offset date-time, the fourth TOML kind, in the same // offsetString renders an offset date-time, the fourth TOML kind, in the same
// shape: no zero seconds, no trailing zeros in the fraction, and the offset // shape: no zero seconds, no trailing zeros in the fraction, and the offset
// written as "Z" when it is zero. // written as "Z" when it is zero. A zone offset that is not a whole number of
// minutes loses its seconds to this rendering, which is why Marshal refuses
// such a value rather than writing it.
func offsetString(t time.Time) string { func offsetString(t time.Time) string {
buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T") buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T")
buf = appendClock(buf, t) buf = appendClock(buf, t)
@@ -235,17 +237,21 @@ func normaliseDateTimeToken(tok string, kind dateTimeKind) string {
// parseDateTime classifies and parses a bare token as a TOML date-time value. // parseDateTime classifies and parses a bare token as a TOML date-time value.
// It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or // It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or
// LocalTime) and whether the token was a date-time at all. // LocalTime), whether the token was a date-time at all, and an error for a
func parseDateTime(tok string) (any, bool) { // token whose shape is a date-time a component of which lies outside its
// range: an hour of 24, a day the month does not hold. Such a token is a
// broken date-time, not some other value, so the error names it instead of
// leaving it to the number decoder's complaint.
func parseDateTime(tok string) (any, bool, error) {
if tok == "" || tok[0] < '0' || tok[0] > '9' { if tok == "" || tok[0] < '0' || tok[0] > '9' {
return nil, false return nil, false, nil
} }
if !strings.ContainsAny(tok, "-:") { if !strings.ContainsAny(tok, "-:") {
return nil, false return nil, false, nil
} }
kind, seconds := scanDateTimeShape(tok) kind, seconds := scanDateTimeShape(tok)
if kind == dateTimeNone { if kind == dateTimeNone {
return nil, false return nil, false, nil
} }
norm := normaliseDateTimeToken(tok, kind) norm := normaliseDateTimeToken(tok, kind)
switch kind { switch kind {
@@ -256,7 +262,7 @@ func parseDateTime(tok string) (any, bool) {
} }
t, err := time.Parse(layout, norm) t, err := time.Parse(layout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
// A zero offset carries its own anonymous location from time.Parse, // A zero offset carries its own anonymous location from time.Parse,
// while the written form is "Z" either way; normalising to UTC keeps // while the written form is "Z" either way; normalising to UTC keeps
@@ -264,7 +270,7 @@ func parseDateTime(tok string) (any, bool) {
if _, off := t.Zone(); off == 0 { if _, off := t.Zone(); off == 0 {
t = t.In(time.UTC) t = t.In(time.UTC)
} }
return OffsetDateTime{t}, true return OffsetDateTime{t}, true, nil
case dateTimeLocal: case dateTimeLocal:
layout := localClockLayout layout := localClockLayout
if seconds { if seconds {
@@ -272,15 +278,15 @@ func parseDateTime(tok string) (any, bool) {
} }
t, err := time.Parse(layout, norm) t, err := time.Parse(layout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
return LocalDateTime{t}, true return LocalDateTime{t}, true, nil
case dateTimeDate: case dateTimeDate:
t, err := time.Parse(localDateOnlyLayout, norm) t, err := time.Parse(localDateOnlyLayout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
return LocalDate{t}, true return LocalDate{t}, true, nil
case dateTimeClock: case dateTimeClock:
layout := localTimeClockLayout layout := localTimeClockLayout
if seconds { if seconds {
@@ -288,11 +294,22 @@ func parseDateTime(tok string) (any, bool) {
} }
t, err := time.Parse(layout, norm) t, err := time.Parse(layout, norm)
if err != nil { if err != nil {
return nil, false return nil, false, fmt.Errorf("invalid date-time %q", tok)
} }
return LocalTime{t}, true return LocalTime{t}, true, nil
} }
return nil, false return nil, false, nil
}
// wholeMinuteOffset reports an error when the zone offset carries seconds, a
// shape no TOML offset can hold: writing only the minutes would silently
// shift the instant on the way back, so the encoder refuses the value rather
// than corrupting it.
func wholeMinuteOffset(t time.Time) error {
if _, off := t.Zone(); off%60 != 0 {
return fmt.Errorf("interpres: date-time offset of %d seconds is not a whole number of minutes, which TOML cannot write", off)
}
return nil
} }
// isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a // isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a
+108
View File
@@ -895,3 +895,111 @@ func TestParseCRLFDocument(t *testing.T) {
t.Errorf("tree = %v", tree) t.Errorf("tree = %v", tree)
} }
} }
// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U:
// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the
// greatest scalar value parses.
func TestParseUnicodeEscapeBoundaries(t *testing.T) {
bad := []struct {
name string
in string
}{
{"high surrogate", `a = "\ud800"`},
{"low surrogate", `a = "\udfff"`},
{"past the greatest scalar", `a = "\U00110000"`},
{"signed short escape", `a = "\u+041"`},
{"negative long escape", `a = "\U-0000001"`},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := Parse([]byte(tt.in))
if err == nil {
t.Fatalf("Parse accepted %q", tt.in)
}
})
}
tree, err := ParseMap([]byte("a = \"\\U0010FFFF\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "􏿿" {
t.Errorf("a = %q", tree["a"])
}
}
// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a
// date-time with a component out of range is rejected as a date-time, not
// left to the number decoder's complaint.
func TestParseRejectsOutOfRangeDateTimes(t *testing.T) {
bad := []struct {
name string
in string
}{
{"hour 24", "a = 1979-05-27T24:00:00Z"},
{"minute 60", "a = 1979-05-27T07:60:00Z"},
{"second 60", "a = 1979-05-27T07:32:60Z"},
{"month 13", "a = 1979-13-27T07:32:00Z"},
{"day 32", "a = 1979-05-32T07:32:00Z"},
{"february the thirtieth", "a = 1979-02-30"},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseMap([]byte(tt.in))
if err == nil {
t.Fatalf("ParseMap accepted %q", tt.in)
}
if !strings.Contains(err.Error(), "invalid date-time") {
t.Errorf("err = %v, want the date-time complaint", err)
}
})
}
}
// TestParseMultilineStringEdges pins the carriage-return and delimiter rules
// of multi-line strings: a bare CR right after the opening delimiter is the
// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the
// content survives.
func TestParseMultilineStringEdges(t *testing.T) {
_, err := ParseMap([]byte("a = \"\"\"\rX\"\"\""))
if err == nil || !strings.Contains(err.Error(), "bare carriage return") {
t.Errorf("err = %v, want the bare-CR error after the delimiter", err)
}
tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "X\r\nY" {
t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"])
}
}
// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes
// before the closing delimiter of a basic multi-line string are content, and
// more than five are the error.
func TestParseMultilineBasicDelimiterRuns(t *testing.T) {
tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != `end"` {
t.Errorf("a = %q", tree["a"])
}
_, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\""))
if err == nil || !strings.Contains(err.Error(), "too many") {
t.Errorf("err = %v, want the too-many-delimiters error", err)
}
}
// TestParseLineEndingBackslashEdges pins the line-ending backslash at the
// very end of the input and before a bare CR.
func TestParseLineEndingBackslashEdges(t *testing.T) {
bad := []string{
"a = \"\"\"x \\\\",
"a = \"\"\"x \\\\\rZ\"\"\"",
}
for _, in := range bad {
if _, err := ParseMap([]byte(in)); err == nil {
t.Errorf("ParseMap accepted %q", in)
}
}
}
+72 -21
View File
@@ -48,6 +48,13 @@ type parser struct {
dotted map[string]bool dotted map[string]bool
arrays map[string]bool arrays map[string]bool
// scopeMarks records the definition-map entries added under an array of
// tables, keyed by that array's path, so a new element's reset drops
// exactly what the previous element added. Without it the reset scans
// every map for the prefix, which a document with many elements and many
// definitions outside them turns quadratic.
scopeMarks map[string][]string
currentPath []string currentPath []string
// keys interns key strings: a document that repeats a key across // keys interns key strings: a document that repeats a key across
@@ -196,6 +203,7 @@ func (p *parser) markHeader(pk string) {
p.headers = make(map[string]bool, 4) p.headers = make(map[string]bool, 4)
} }
p.headers[pk] = true p.headers[pk] = true
p.trackScope(pk)
} }
func (p *parser) markFrozen(pk string) { func (p *parser) markFrozen(pk string) {
@@ -203,6 +211,7 @@ func (p *parser) markFrozen(pk string) {
p.frozen = make(map[string]bool, 4) p.frozen = make(map[string]bool, 4)
} }
p.frozen[pk] = true p.frozen[pk] = true
p.trackScope(pk)
} }
func (p *parser) markDotted(pk string) { func (p *parser) markDotted(pk string) {
@@ -210,6 +219,7 @@ func (p *parser) markDotted(pk string) {
p.dotted = make(map[string]bool, 4) p.dotted = make(map[string]bool, 4)
} }
p.dotted[pk] = true p.dotted[pk] = true
p.trackScope(pk)
} }
func (p *parser) markArray(pk string) { func (p *parser) markArray(pk string) {
@@ -217,6 +227,22 @@ func (p *parser) markArray(pk string) {
p.arrays = make(map[string]bool, 2) p.arrays = make(map[string]bool, 2)
} }
p.arrays[pk] = true p.arrays[pk] = true
p.trackScope(pk)
}
// trackScope records a definition entry under every array of tables it falls
// inside, so resetScopeUnder can drop it when a later element opens. An entry
// under no array, such as every definition before the first header, needs no
// record: no reset can ever name it.
func (p *parser) trackScope(pk string) {
for arr := range p.arrays {
if strings.HasPrefix(pk, arr+"\x00") {
if p.scopeMarks == nil {
p.scopeMarks = make(map[string][]string, 2)
}
p.scopeMarks[arr] = append(p.scopeMarks[arr], pk)
}
}
} }
// internKey returns the shared string for key bytes. The lookup works on the // internKey returns the shared string for key bytes. The lookup works on the
@@ -482,9 +508,14 @@ func (p *parser) parseKeyValue() error {
if p.doc != nil { if p.doc != nil {
node := p.currentNode node := p.currentNode
if len(rest) > 0 { if len(rest) > 0 {
// The tables a dotted key builds hold the position of a line, so
// the write side marks them and gives each leaf back as a dotted
// key rather than a header that would swallow the lines after it.
node = node.addTable(first, dests[0]) node = node.addTable(first, dests[0])
node.dotted = true
for i, k := range rest[:len(rest)-1] { for i, k := range rest[:len(rest)-1] {
node = node.addTable(k, dests[i+1]) node = node.addTable(k, dests[i+1])
node.dotted = true
} }
} }
_, inline := val.(map[string]any) _, inline := val.(map[string]any)
@@ -570,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) {
// resetScopeUnder forgets the definition records nested under key, which // resetScopeUnder forgets the definition records nested under key, which
// belong to the previous element of an array of tables: headers, frozen // belong to the previous element of an array of tables: headers, frozen
// inline tables, dotted-key paths, and nested arrays of tables all start // inline tables, dotted-key paths, and nested arrays of tables all start
// fresh in the new element. // fresh in the new element. The records to drop are the ones the element
// added, which scopeMarks holds; the array's own entry, and everything
// outside it, keep their place.
func (p *parser) resetScopeUnder(key []string) { func (p *parser) resetScopeUnder(key []string) {
prefix := pathKey(key) + "\x00" pk := pathKey(key)
p.resetMapUnder(p.headers, prefix) for _, k := range p.scopeMarks[pk] {
p.resetMapUnder(p.frozen, prefix) delete(p.headers, k)
p.resetMapUnder(p.dotted, prefix) delete(p.frozen, k)
p.resetMapUnder(p.arrays, prefix) delete(p.dotted, k)
} delete(p.arrays, k)
// resetMapUnder deletes the entries m holds under prefix. An empty or
// unallocated map holds none, so the common case walks nothing.
func (p *parser) resetMapUnder(m map[string]bool, prefix string) {
if len(m) == 0 {
return
} }
for k := range m { if p.scopeMarks != nil {
if strings.HasPrefix(k, prefix) { p.scopeMarks[pk] = nil
delete(m, k)
}
} }
} }
@@ -708,7 +733,7 @@ func (p *parser) parseAtom() (any, error) {
return nil, p.errf("expected a value") return nil, p.errf("expected a value")
} }
if hasHighByte(tok) && !utf8.ValidString(tok) { if hasHighByte(tok) && !utf8.ValidString(tok) {
return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos) return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok))
} }
// A date may be followed by a space and a time, forming one date-time. // A date may be followed by a space and a time, forming one date-time.
if isDateToken(tok) && !p.eof() && p.peek() == ' ' { if isDateToken(tok) && !p.eof() && p.peek() == ' ' {
@@ -719,7 +744,11 @@ func (p *parser) parseAtom() (any, error) {
tok = tok + " " + string(p.src[timeStart:p.pos]) tok = tok + " " + string(p.src[timeStart:p.pos])
} }
} }
if v, ok := parseDateTime(tok); ok { v, isDT, dterr := parseDateTime(tok)
if dterr != nil {
return nil, p.errf("%s", dterr)
}
if isDT {
return v, nil return v, nil
} }
v, err := decodeNumber(tok) v, err := decodeNumber(tok)
@@ -758,6 +787,20 @@ func hasHighByte(s string) bool {
return false return false
} }
// invalidUTF8Offset returns the offset of the first byte in s that does not
// decode as UTF-8, or -1 when all of it does, so an error can name the byte
// that is invalid rather than the end of the token around it.
func invalidUTF8Offset(s string) int {
for i := 0; i < len(s); {
r, size := utf8.DecodeRuneInString(s[i:])
if r == utf8.RuneError && size == 1 {
return i
}
i += size
}
return -1
}
// --- strings --------------------------------------------------------------- // --- strings ---------------------------------------------------------------
func (p *parser) parseBasicString() (string, error) { func (p *parser) parseBasicString() (string, error) {
@@ -895,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error {
func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) { func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) {
p.skipN(3) // opening delimiter p.skipN(3) // opening delimiter
// A newline immediately after the opening delimiter is trimmed. // A newline immediately after the opening delimiter is trimmed, and it is
// a newline: a bare CR here is the bare-CR error like anywhere else, not
// a newline to trim.
if !p.eof() && p.peek() == '\r' { if !p.eof() && p.peek() == '\r' {
if next, ok := p.peekAt(1); !ok || next != '\n' {
return "", p.errf("bare carriage return is not allowed in a string")
}
p.pos++ p.pos++
} }
if !p.eof() && p.peek() == '\n' { if !p.eof() && p.peek() == '\n' {
@@ -1047,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) {
} }
hex := string(p.src[p.pos : p.pos+n]) hex := string(p.src[p.pos : p.pos+n])
p.pos += n p.pos += n
v, err := strconv.ParseInt(hex, 16, 64) // ParseUint rather than ParseInt: a sign is not a hex digit, and a signed
// read would let "\U-0000001" through the range checks below only to
// write U+FFFD for a document the grammar rejects.
v, err := strconv.ParseUint(hex, 16, 32)
if err != nil { if err != nil {
return 0, p.errf("invalid unicode escape \\%s", hex) return 0, p.errf("invalid unicode escape \\%s", hex)
} }