perf(parse): cut allocations and validate UTF-8 in the scan

This commit is contained in:
2026-09-20 22:15:10 +02:00
parent adf189aa2c
commit b4d564c682
6 changed files with 427 additions and 117 deletions
+3 -4
View File
@@ -24,7 +24,6 @@ import (
"context"
"errors"
"fmt"
"unicode/utf8"
)
// A SyntaxError describes a malformed TOML document, including the 1-based
@@ -143,9 +142,9 @@ func parseWithOptions(ctx context.Context, data []byte, opts parseOptions, wantD
if opts.maxInputSize > 0 && len(data) > opts.maxInputSize {
return nil, nil, fmt.Errorf("interpres: input is %d bytes, over the limit of %d", len(data), opts.maxInputSize)
}
if !utf8.Valid(data) {
return nil, nil, &SyntaxError{Line: 1, Msg: "input is not valid UTF-8"}
}
// UTF-8 validity is not checked in a pass of its own: the scanner
// validates the multi-byte sequences where it meets them, so an invalid
// byte is reported on its own line instead of always on line 1.
maxDepth := opts.maxDepth
if maxDepth <= 0 {
maxDepth = maxNestingDepth