diff --git a/datetime.go b/datetime.go index 7d54649..55f9de0 100644 --- a/datetime.go +++ b/datetime.go @@ -79,7 +79,9 @@ func clockString(t time.Time) string { // offsetString renders an offset date-time, the fourth TOML kind, in the same // shape: no zero seconds, no trailing zeros in the fraction, and the offset -// written as "Z" when it is zero. +// written as "Z" when it is zero. A zone offset that is not a whole number of +// minutes loses its seconds to this rendering, which is why Marshal refuses +// such a value rather than writing it. func offsetString(t time.Time) string { buf := t.AppendFormat(make([]byte, 0, 32), "2006-01-02T") buf = appendClock(buf, t) @@ -235,17 +237,21 @@ func normaliseDateTimeToken(tok string, kind dateTimeKind) string { // parseDateTime classifies and parses a bare token as a TOML date-time value. // It returns the decoded value (OffsetDateTime, LocalDateTime, LocalDate or -// LocalTime) and whether the token was a date-time at all. -func parseDateTime(tok string) (any, bool) { +// LocalTime), whether the token was a date-time at all, and an error for a +// token whose shape is a date-time a component of which lies outside its +// range: an hour of 24, a day the month does not hold. Such a token is a +// broken date-time, not some other value, so the error names it instead of +// leaving it to the number decoder's complaint. +func parseDateTime(tok string) (any, bool, error) { if tok == "" || tok[0] < '0' || tok[0] > '9' { - return nil, false + return nil, false, nil } if !strings.ContainsAny(tok, "-:") { - return nil, false + return nil, false, nil } kind, seconds := scanDateTimeShape(tok) if kind == dateTimeNone { - return nil, false + return nil, false, nil } norm := normaliseDateTimeToken(tok, kind) switch kind { @@ -256,7 +262,7 @@ func parseDateTime(tok string) (any, bool) { } t, err := time.Parse(layout, norm) if err != nil { - return nil, false + return nil, false, fmt.Errorf("invalid date-time %q", tok) } // A zero offset carries its own anonymous location from time.Parse, // while the written form is "Z" either way; normalising to UTC keeps @@ -264,7 +270,7 @@ func parseDateTime(tok string) (any, bool) { if _, off := t.Zone(); off == 0 { t = t.In(time.UTC) } - return OffsetDateTime{t}, true + return OffsetDateTime{t}, true, nil case dateTimeLocal: layout := localClockLayout if seconds { @@ -272,15 +278,15 @@ func parseDateTime(tok string) (any, bool) { } t, err := time.Parse(layout, norm) if err != nil { - return nil, false + return nil, false, fmt.Errorf("invalid date-time %q", tok) } - return LocalDateTime{t}, true + return LocalDateTime{t}, true, nil case dateTimeDate: t, err := time.Parse(localDateOnlyLayout, norm) if err != nil { - return nil, false + return nil, false, fmt.Errorf("invalid date-time %q", tok) } - return LocalDate{t}, true + return LocalDate{t}, true, nil case dateTimeClock: layout := localTimeClockLayout if seconds { @@ -288,11 +294,22 @@ func parseDateTime(tok string) (any, bool) { } t, err := time.Parse(layout, norm) if err != nil { - return nil, false + return nil, false, fmt.Errorf("invalid date-time %q", tok) } - return LocalTime{t}, true + return LocalTime{t}, true, nil } - return nil, false + return nil, false, nil +} + +// wholeMinuteOffset reports an error when the zone offset carries seconds, a +// shape no TOML offset can hold: writing only the minutes would silently +// shift the instant on the way back, so the encoder refuses the value rather +// than corrupting it. +func wholeMinuteOffset(t time.Time) error { + if _, off := t.Zone(); off%60 != 0 { + return fmt.Errorf("interpres: date-time offset of %d seconds is not a whole number of minutes, which TOML cannot write", off) + } + return nil } // isDateToken reports whether s is exactly a YYYY-MM-DD date, used to detect a diff --git a/interpres_test.go b/interpres_test.go index 78a762b..e04e77e 100644 --- a/interpres_test.go +++ b/interpres_test.go @@ -895,3 +895,111 @@ func TestParseCRLFDocument(t *testing.T) { t.Errorf("tree = %v", tree) } } + +// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U: +// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the +// greatest scalar value parses. +func TestParseUnicodeEscapeBoundaries(t *testing.T) { + bad := []struct { + name string + in string + }{ + {"high surrogate", `a = "\ud800"`}, + {"low surrogate", `a = "\udfff"`}, + {"past the greatest scalar", `a = "\U00110000"`}, + {"signed short escape", `a = "\u+041"`}, + {"negative long escape", `a = "\U-0000001"`}, + } + for _, tt := range bad { + t.Run(tt.name, func(t *testing.T) { + _, err := Parse([]byte(tt.in)) + if err == nil { + t.Fatalf("Parse accepted %q", tt.in) + } + }) + } + tree, err := ParseMap([]byte("a = \"\\U0010FFFF\"")) + if err != nil { + t.Fatalf("ParseMap: %v", err) + } + if tree["a"] != "􏿿" { + t.Errorf("a = %q", tree["a"]) + } +} + +// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a +// date-time with a component out of range is rejected as a date-time, not +// left to the number decoder's complaint. +func TestParseRejectsOutOfRangeDateTimes(t *testing.T) { + bad := []struct { + name string + in string + }{ + {"hour 24", "a = 1979-05-27T24:00:00Z"}, + {"minute 60", "a = 1979-05-27T07:60:00Z"}, + {"second 60", "a = 1979-05-27T07:32:60Z"}, + {"month 13", "a = 1979-13-27T07:32:00Z"}, + {"day 32", "a = 1979-05-32T07:32:00Z"}, + {"february the thirtieth", "a = 1979-02-30"}, + } + for _, tt := range bad { + t.Run(tt.name, func(t *testing.T) { + _, err := ParseMap([]byte(tt.in)) + if err == nil { + t.Fatalf("ParseMap accepted %q", tt.in) + } + if !strings.Contains(err.Error(), "invalid date-time") { + t.Errorf("err = %v, want the date-time complaint", err) + } + }) + } +} + +// TestParseMultilineStringEdges pins the carriage-return and delimiter rules +// of multi-line strings: a bare CR right after the opening delimiter is the +// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the +// content survives. +func TestParseMultilineStringEdges(t *testing.T) { + _, err := ParseMap([]byte("a = \"\"\"\rX\"\"\"")) + if err == nil || !strings.Contains(err.Error(), "bare carriage return") { + t.Errorf("err = %v, want the bare-CR error after the delimiter", err) + } + tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\"")) + if err != nil { + t.Fatalf("ParseMap: %v", err) + } + if tree["a"] != "X\r\nY" { + t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"]) + } +} + +// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes +// before the closing delimiter of a basic multi-line string are content, and +// more than five are the error. +func TestParseMultilineBasicDelimiterRuns(t *testing.T) { + tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\"")) + if err != nil { + t.Fatalf("ParseMap: %v", err) + } + if tree["a"] != `end"` { + t.Errorf("a = %q", tree["a"]) + } + _, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\"")) + if err == nil || !strings.Contains(err.Error(), "too many") { + t.Errorf("err = %v, want the too-many-delimiters error", err) + } +} + +// TestParseLineEndingBackslashEdges pins the line-ending backslash at the +// very end of the input and before a bare CR. +func TestParseLineEndingBackslashEdges(t *testing.T) { + bad := []string{ + "a = \"\"\"x \\\\", + "a = \"\"\"x \\\\\rZ\"\"\"", + } + for _, in := range bad { + if _, err := ParseMap([]byte(in)); err == nil { + t.Errorf("ParseMap accepted %q", in) + } + } +} diff --git a/parser.go b/parser.go index ae05190..7435928 100644 --- a/parser.go +++ b/parser.go @@ -48,6 +48,13 @@ type parser struct { dotted map[string]bool arrays map[string]bool + // scopeMarks records the definition-map entries added under an array of + // tables, keyed by that array's path, so a new element's reset drops + // exactly what the previous element added. Without it the reset scans + // every map for the prefix, which a document with many elements and many + // definitions outside them turns quadratic. + scopeMarks map[string][]string + currentPath []string // keys interns key strings: a document that repeats a key across @@ -196,6 +203,7 @@ func (p *parser) markHeader(pk string) { p.headers = make(map[string]bool, 4) } p.headers[pk] = true + p.trackScope(pk) } func (p *parser) markFrozen(pk string) { @@ -203,6 +211,7 @@ func (p *parser) markFrozen(pk string) { p.frozen = make(map[string]bool, 4) } p.frozen[pk] = true + p.trackScope(pk) } func (p *parser) markDotted(pk string) { @@ -210,6 +219,7 @@ func (p *parser) markDotted(pk string) { p.dotted = make(map[string]bool, 4) } p.dotted[pk] = true + p.trackScope(pk) } func (p *parser) markArray(pk string) { @@ -217,6 +227,22 @@ func (p *parser) markArray(pk string) { p.arrays = make(map[string]bool, 2) } p.arrays[pk] = true + p.trackScope(pk) +} + +// trackScope records a definition entry under every array of tables it falls +// inside, so resetScopeUnder can drop it when a later element opens. An entry +// under no array, such as every definition before the first header, needs no +// record: no reset can ever name it. +func (p *parser) trackScope(pk string) { + for arr := range p.arrays { + if strings.HasPrefix(pk, arr+"\x00") { + if p.scopeMarks == nil { + p.scopeMarks = make(map[string][]string, 2) + } + p.scopeMarks[arr] = append(p.scopeMarks[arr], pk) + } + } } // internKey returns the shared string for key bytes. The lookup works on the @@ -482,9 +508,14 @@ func (p *parser) parseKeyValue() error { if p.doc != nil { node := p.currentNode if len(rest) > 0 { + // The tables a dotted key builds hold the position of a line, so + // the write side marks them and gives each leaf back as a dotted + // key rather than a header that would swallow the lines after it. node = node.addTable(first, dests[0]) + node.dotted = true for i, k := range rest[:len(rest)-1] { node = node.addTable(k, dests[i+1]) + node.dotted = true } } _, inline := val.(map[string]any) @@ -570,25 +601,19 @@ func (p *parser) freezeInline(path []string, val any) { // resetScopeUnder forgets the definition records nested under key, which // belong to the previous element of an array of tables: headers, frozen // inline tables, dotted-key paths, and nested arrays of tables all start -// fresh in the new element. +// fresh in the new element. The records to drop are the ones the element +// added, which scopeMarks holds; the array's own entry, and everything +// outside it, keep their place. func (p *parser) resetScopeUnder(key []string) { - prefix := pathKey(key) + "\x00" - p.resetMapUnder(p.headers, prefix) - p.resetMapUnder(p.frozen, prefix) - p.resetMapUnder(p.dotted, prefix) - p.resetMapUnder(p.arrays, prefix) -} - -// resetMapUnder deletes the entries m holds under prefix. An empty or -// unallocated map holds none, so the common case walks nothing. -func (p *parser) resetMapUnder(m map[string]bool, prefix string) { - if len(m) == 0 { - return + pk := pathKey(key) + for _, k := range p.scopeMarks[pk] { + delete(p.headers, k) + delete(p.frozen, k) + delete(p.dotted, k) + delete(p.arrays, k) } - for k := range m { - if strings.HasPrefix(k, prefix) { - delete(m, k) - } + if p.scopeMarks != nil { + p.scopeMarks[pk] = nil } } @@ -708,7 +733,7 @@ func (p *parser) parseAtom() (any, error) { return nil, p.errf("expected a value") } if hasHighByte(tok) && !utf8.ValidString(tok) { - return nil, p.errf("invalid UTF-8 in value at byte offset %d", p.pos) + return nil, p.errf("invalid UTF-8 in value at byte offset %d", start+invalidUTF8Offset(tok)) } // A date may be followed by a space and a time, forming one date-time. if isDateToken(tok) && !p.eof() && p.peek() == ' ' { @@ -719,7 +744,11 @@ func (p *parser) parseAtom() (any, error) { tok = tok + " " + string(p.src[timeStart:p.pos]) } } - if v, ok := parseDateTime(tok); ok { + v, isDT, dterr := parseDateTime(tok) + if dterr != nil { + return nil, p.errf("%s", dterr) + } + if isDT { return v, nil } v, err := decodeNumber(tok) @@ -758,6 +787,20 @@ func hasHighByte(s string) bool { return false } +// invalidUTF8Offset returns the offset of the first byte in s that does not +// decode as UTF-8, or -1 when all of it does, so an error can name the byte +// that is invalid rather than the end of the token around it. +func invalidUTF8Offset(s string) int { + for i := 0; i < len(s); { + r, size := utf8.DecodeRuneInString(s[i:]) + if r == utf8.RuneError && size == 1 { + return i + } + i += size + } + return -1 +} + // --- strings --------------------------------------------------------------- func (p *parser) parseBasicString() (string, error) { @@ -895,8 +938,13 @@ func (p *parser) writeContentRune(b *strings.Builder) error { func (p *parser) parseMultilineString(quote byte, escapes bool) (string, error) { p.skipN(3) // opening delimiter - // A newline immediately after the opening delimiter is trimmed. + // A newline immediately after the opening delimiter is trimmed, and it is + // a newline: a bare CR here is the bare-CR error like anywhere else, not + // a newline to trim. if !p.eof() && p.peek() == '\r' { + if next, ok := p.peekAt(1); !ok || next != '\n' { + return "", p.errf("bare carriage return is not allowed in a string") + } p.pos++ } if !p.eof() && p.peek() == '\n' { @@ -1047,7 +1095,10 @@ func (p *parser) readUnicode(n int) (rune, error) { } hex := string(p.src[p.pos : p.pos+n]) p.pos += n - v, err := strconv.ParseInt(hex, 16, 64) + // ParseUint rather than ParseInt: a sign is not a hex digit, and a signed + // read would let "\U-0000001" through the range checks below only to + // write U+FFFD for a document the grammar rejects. + v, err := strconv.ParseUint(hex, 16, 32) if err != nil { return 0, p.errf("invalid unicode escape \\%s", hex) }