perf(decode): cache struct schemas per type

Assisted-by: GLM 5.3
This commit is contained in:
2026-09-17 23:08:14 +02:00
parent a8d69d90d5
commit aaea68efc9
3 changed files with 29 additions and 3 deletions
+4
View File
@@ -38,6 +38,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
v2.2.0, run in TOML 1.0 mode. The new corpus holds 205 valid and 474 invalid v2.2.0, run in TOML 1.0 mode. The new corpus holds 205 valid and 474 invalid
cases (v1.6.0 had 185 and 371), and it caught the two documents the parser cases (v1.6.0 had 185 and 371), and it caught the two documents the parser
still accepted, fixed below. still accepted, fixed below.
- The flattened struct layout the decoder consults is cached per struct type
and shared with the encoder, which now resolves duplicate field keys with
it. Strict decoding of an array of tables of structs runs about a quarter
faster; marshalling structs gained the same layout without measurable cost.
### Fixed ### Fixed
+17 -1
View File
@@ -8,6 +8,7 @@ import (
"math" "math"
"reflect" "reflect"
"strings" "strings"
"sync"
"time" "time"
) )
@@ -105,7 +106,7 @@ func (d *decoder) assignTable(tbl map[string]any, dst reflect.Value) error {
} }
func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error { func (d *decoder) assignStruct(tbl map[string]any, dst reflect.Value) error {
schema := newStructSchema(dst.Type()) schema := cachedStructSchema(dst.Type())
for key, val := range tbl { for key, val := range tbl {
field, ok := schema.byName[strings.ToLower(key)] field, ok := schema.byName[strings.ToLower(key)]
if !ok { if !ok {
@@ -253,6 +254,21 @@ type structSchema struct {
embedMaps [][]int embedMaps [][]int
} }
// structSchemaCache holds one schema per struct type. A schema is immutable
// once published, so concurrent callers only race to build an identical value,
// the same trade-off encoding/json's field cache makes. The cache grows with
// the number of distinct types decoded or encoded, never per document.
var structSchemaCache sync.Map // reflect.Type -> structSchema
func cachedStructSchema(t reflect.Type) structSchema {
if s, ok := structSchemaCache.Load(t); ok {
return s.(structSchema)
}
s := newStructSchema(t)
actual, _ := structSchemaCache.LoadOrStore(t, s)
return actual.(structSchema)
}
func newStructSchema(t reflect.Type) structSchema { func newStructSchema(t reflect.Type) structSchema {
s := structSchema{byName: make(map[string]structFieldLoc, t.NumField())} s := structSchema{byName: make(map[string]structFieldLoc, t.NumField())}
// A struct may embed a pointer to itself, which is legal Go, so the walk // A struct may embed a pointer to itself, which is legal Go, so the walk
+8 -2
View File
@@ -101,9 +101,15 @@ sequenceDiagram
setter methods are not, and must finish before the value is shared. setter methods are not, and must finish before the value is shared.
- The parser is allocated per `ParseContext` call; nothing is cached between - The parser is allocated per `ParseContext` call; nothing is cached between
documents. documents.
- The one piece of shared state is the struct-schema cache in `decode.go`: a
`sync.Map` keyed by `reflect.Type`, holding the flattened field layout the
decoder and the encoder both consult. A schema is immutable once published,
so concurrent callers only race to build an identical value, the same
trade-off `encoding/json`'s field cache makes. The cache grows with the
number of distinct struct types, never with document size.
- The date-time wrappers are values, not pointers, and are immutable in use. - The date-time wrappers are values, not pointers, and are immutable in use.
- Nothing in the library starts goroutines or holds locks; concurrency safety - Nothing in the library starts goroutines; apart from the schema cache above,
comes from having no shared mutable state. which never mutates a published entry, there is no shared mutable state.
## Dependencies ## Dependencies