fix(parse): reject the lenient grammar edges and name out-of-range date-times

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-22 21:15:00 +02:00
parent 8f2b26bd33
commit 2efdb2d059
3 changed files with 212 additions and 36 deletions
+108
View File
@@ -895,3 +895,111 @@ func TestParseCRLFDocument(t *testing.T) {
t.Errorf("tree = %v", tree)
}
}
// TestParseUnicodeEscapeBoundaries pins the scalar-value checks of \u and \U:
// a surrogate, a value past U+10FFFF, and a sign are all rejected, and the
// greatest scalar value parses.
func TestParseUnicodeEscapeBoundaries(t *testing.T) {
bad := []struct {
name string
in string
}{
{"high surrogate", `a = "\ud800"`},
{"low surrogate", `a = "\udfff"`},
{"past the greatest scalar", `a = "\U00110000"`},
{"signed short escape", `a = "\u+041"`},
{"negative long escape", `a = "\U-0000001"`},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := Parse([]byte(tt.in))
if err == nil {
t.Fatalf("Parse accepted %q", tt.in)
}
})
}
tree, err := ParseMap([]byte("a = \"\\U0010FFFF\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "􏿿" {
t.Errorf("a = %q", tree["a"])
}
}
// TestParseRejectsOutOfRangeDateTimes pins that a token shaped like a
// date-time with a component out of range is rejected as a date-time, not
// left to the number decoder's complaint.
func TestParseRejectsOutOfRangeDateTimes(t *testing.T) {
bad := []struct {
name string
in string
}{
{"hour 24", "a = 1979-05-27T24:00:00Z"},
{"minute 60", "a = 1979-05-27T07:60:00Z"},
{"second 60", "a = 1979-05-27T07:32:60Z"},
{"month 13", "a = 1979-13-27T07:32:00Z"},
{"day 32", "a = 1979-05-32T07:32:00Z"},
{"february the thirtieth", "a = 1979-02-30"},
}
for _, tt := range bad {
t.Run(tt.name, func(t *testing.T) {
_, err := ParseMap([]byte(tt.in))
if err == nil {
t.Fatalf("ParseMap accepted %q", tt.in)
}
if !strings.Contains(err.Error(), "invalid date-time") {
t.Errorf("err = %v, want the date-time complaint", err)
}
})
}
}
// TestParseMultilineStringEdges pins the carriage-return and delimiter rules
// of multi-line strings: a bare CR right after the opening delimiter is the
// bare-CR error, a CRLF pair is the trimmed newline, and a CRLF inside the
// content survives.
func TestParseMultilineStringEdges(t *testing.T) {
_, err := ParseMap([]byte("a = \"\"\"\rX\"\"\""))
if err == nil || !strings.Contains(err.Error(), "bare carriage return") {
t.Errorf("err = %v, want the bare-CR error after the delimiter", err)
}
tree, err := ParseMap([]byte("a = \"\"\"\r\nX\r\nY\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != "X\r\nY" {
t.Errorf("a = %q, want the CRLF pairs preserved", tree["a"])
}
}
// TestParseMultilineBasicDelimiterRuns pins that up to two extra quotes
// before the closing delimiter of a basic multi-line string are content, and
// more than five are the error.
func TestParseMultilineBasicDelimiterRuns(t *testing.T) {
tree, err := ParseMap([]byte("a = \"\"\"end\"\"\"\""))
if err != nil {
t.Fatalf("ParseMap: %v", err)
}
if tree["a"] != `end"` {
t.Errorf("a = %q", tree["a"])
}
_, err = ParseMap([]byte("a = \"\"\"end\"\"\"\"\"\"\""))
if err == nil || !strings.Contains(err.Error(), "too many") {
t.Errorf("err = %v, want the too-many-delimiters error", err)
}
}
// TestParseLineEndingBackslashEdges pins the line-ending backslash at the
// very end of the input and before a bare CR.
func TestParseLineEndingBackslashEdges(t *testing.T) {
bad := []string{
"a = \"\"\"x \\\\",
"a = \"\"\"x \\\\\rZ\"\"\"",
}
for _, in := range bad {
if _, err := ParseMap([]byte(in)); err == nil {
t.Errorf("ParseMap accepted %q", in)
}
}
}