feat: initial release
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+786
@@ -0,0 +1,786 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package io
|
||||
|
||||
// HDF5 writing, the mirror of the reader in hdf5.go. The reader's
|
||||
// verified decoders are the specification: every structure here is
|
||||
// written in the shape the reader accepts and in the shape the
|
||||
// reference library writes, as the fixtures under testdata/h5 pin it.
|
||||
//
|
||||
// Written: superblock version 0 (the classic layout, the default) and
|
||||
// version 3 (the "latest" layout, whose superblock and object headers
|
||||
// carry lookup3 checksums), object headers versions 1 and 2, groups
|
||||
// stored as symbol tables (local heap, version 1 group B-tree, symbol
|
||||
// table nodes) in the classic layout or as link messages in the latest
|
||||
// one, datasets stored contiguously or in chunks through a version 1
|
||||
// chunk B-tree, the deflate and shuffle filters, fixed-point and
|
||||
// floating-point datatypes of the usual widths, the boolean
|
||||
// enumeration convention, fixed-length string datatypes, and
|
||||
// attributes in the object header.
|
||||
//
|
||||
// Every address and length is eight bytes, as in the fixtures. The
|
||||
// output is deterministic: children are written in sorted name order
|
||||
// and nothing depends on map iteration.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"compress/zlib"
|
||||
"encoding/binary"
|
||||
"maps"
|
||||
"math"
|
||||
"os"
|
||||
"slices"
|
||||
"strings"
|
||||
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
||||
)
|
||||
|
||||
// HDF5WriteOptions tunes SaveHDF5 and SaveHDF5Text. The zero value
|
||||
// writes the classic file layout with contiguous datasets, which every
|
||||
// reader of the format understands.
|
||||
type HDF5WriteOptions struct {
|
||||
// Latest writes superblock version 3 with version 2 object
|
||||
// headers: groups become link messages and every structure the
|
||||
// format checksums carries a lookup3 sum. Latest files hold
|
||||
// contiguous datasets only: the reference library stores filtered
|
||||
// chunks of a latest file in a version 2 B-tree, which LoadHDF5
|
||||
// does not read, so combining Latest with a filter is refused.
|
||||
Latest bool
|
||||
|
||||
// Gzip applies the deflate filter to every numeric dataset at the
|
||||
// given level: 0 (the default) disables it, -1 means the default
|
||||
// level and 1 to 9 are the levels of the format. A filtered
|
||||
// dataset is stored in chunks.
|
||||
Gzip int
|
||||
|
||||
// Shuffle applies the shuffle filter before deflate, which
|
||||
// regroups the bytes of each element so compression sees the
|
||||
// high-order bytes together. Shuffle alone also forces chunks.
|
||||
Shuffle bool
|
||||
|
||||
// ChunkBytes is the target size of one chunk in bytes for
|
||||
// filtered datasets; 0 selects a default of 64 KiB. Datasets
|
||||
// smaller than the target stay in one chunk.
|
||||
ChunkBytes int
|
||||
}
|
||||
|
||||
// SaveHDF5 writes the datasets as an HDF5 file: the mirror of
|
||||
// LoadHDF5. The paths build the group tree (the dataset "/g/f32" sits
|
||||
// in the group "/g"), so the file reads back with the same paths,
|
||||
// shapes, dtypes and values. Each dataset's Attrs are written on the
|
||||
// dataset itself; the attributes of the root and of the groups come
|
||||
// from groupAttrs, keyed by group path with the root keyed "/".
|
||||
//
|
||||
// Every dtype the writer stores lands the same dtype through LoadHDF5:
|
||||
// bool through the HDF5 boolean enumeration convention, the narrow
|
||||
// integers at their stored width and signedness, float32, float64 and
|
||||
// int64 directly. Float16 and complex arrays are refused: LoadHDF5
|
||||
// decodes neither a two-byte floating-point nor a complex datatype, so
|
||||
// the writer refuses them rather than write a file this package cannot
|
||||
// read back. Attribute values are parsed back into typed attributes: a
|
||||
// whole number becomes an int64 attribute, a decimal a float64 one, a
|
||||
// bracketed list an int64 or float64 array, and anything else a
|
||||
// fixed-length string, so a file written from LoadHDF5's own output
|
||||
// reads back with the same attribute text. When several option values
|
||||
// are passed the last one wins.
|
||||
func SaveHDF5(path string, datasets []HDF5Dataset, groupAttrs map[string]map[string]string, opts ...HDF5WriteOptions) error {
|
||||
const name = "SaveHDF5"
|
||||
options := HDF5WriteOptions{}
|
||||
for _, o := range opts {
|
||||
options = o
|
||||
}
|
||||
if err := hdf5CheckOptions(name, options); err != nil {
|
||||
return err
|
||||
}
|
||||
root, err := hdf5BuildPlan(name, datasets, nil, groupAttrs)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
w := &hdf5Writer{latest: options.Latest, opts: options}
|
||||
if err := w.write(root); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(path, w.buf, 0o644); err != nil {
|
||||
return base.Errf("%s: %w", name, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// HDF5TextDataset is one fixed-length string dataset for SaveHDF5Text:
|
||||
// Text holds the elements in row-major order, each padded to the
|
||||
// longest element in the file. The reader of this package refuses
|
||||
// string datasets (it reads numeric arrays only), so these files are
|
||||
// for other readers of the format.
|
||||
type HDF5TextDataset struct {
|
||||
Path string
|
||||
Shape []int
|
||||
Text []string
|
||||
}
|
||||
|
||||
// SaveHDF5Text writes fixed-length string datasets, the string side of
|
||||
// the datatype message the reader refuses for data but accepts for
|
||||
// attributes. The datasets are stored contiguously; the deflate and
|
||||
// shuffle filters, which are chunked-storage filters, are refused
|
||||
// here by name rather than silently dropped. When several option
|
||||
// values are passed the last one wins.
|
||||
func SaveHDF5Text(path string, texts []HDF5TextDataset, opts ...HDF5WriteOptions) error {
|
||||
const name = "SaveHDF5Text"
|
||||
options := HDF5WriteOptions{}
|
||||
for _, o := range opts {
|
||||
options = o
|
||||
}
|
||||
if options.Gzip != 0 || options.Shuffle {
|
||||
return base.Errf("%s: the deflate and shuffle filters apply to numeric datasets; string datasets are written contiguously", name)
|
||||
}
|
||||
if err := hdf5CheckOptions(name, options); err != nil {
|
||||
return err
|
||||
}
|
||||
root, err := hdf5BuildPlan(name, nil, texts, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
w := &hdf5Writer{latest: options.Latest, opts: options}
|
||||
if err := w.write(root); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(path, w.buf, 0o644); err != nil {
|
||||
return base.Errf("%s: %w", name, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// hdf5CheckOptions rejects the option combinations the writer cannot
|
||||
// honour, naming each one.
|
||||
func hdf5CheckOptions(name string, opts HDF5WriteOptions) error {
|
||||
switch opts.Gzip {
|
||||
case 0, -1, 1, 2, 3, 4, 5, 6, 7, 8, 9:
|
||||
default:
|
||||
return base.Errf("%s: gzip level %d: use 0 to disable the filter, -1 for the default level or 1 to 9", name, opts.Gzip)
|
||||
}
|
||||
if opts.ChunkBytes < 0 {
|
||||
return base.Errf("%s: a chunk target of %d bytes is negative", name, opts.ChunkBytes)
|
||||
}
|
||||
if opts.Latest && (opts.Gzip != 0 || opts.Shuffle) {
|
||||
return base.Errf("%s: the latest format stores filtered chunks through a version 2 B-tree, which LoadHDF5 does not read; write filtered datasets to a classic file", name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// The B-tree and symbol node fanouts the classic layout declares in
|
||||
// its superblock: a node of the format holds twice the K of its kind,
|
||||
// so a symbol table node holds eight entries, a group B-tree node
|
||||
// thirty-two children and a chunk B-tree node (whose K the format
|
||||
// fixes at thirty-two for superblock version 0) sixty-four chunks.
|
||||
const (
|
||||
hdf5GroupLeafK = 4
|
||||
hdf5GroupInnerK = 16
|
||||
hdf5IStoreK = 32
|
||||
hdf5ChunkTarget = 64 << 10
|
||||
hdf5MaxChunks = 4 << 20
|
||||
hdf5MaxAttrs = 4096
|
||||
hdf5MaxRank = 32
|
||||
hdf5MaxNameBytes = 4096
|
||||
)
|
||||
|
||||
// hdf5OutSet is one dataset of the plan: the source of its payload,
|
||||
// its shape and its datatype class (0 fixed-point, 1 floating-point,
|
||||
// 3 string, 8 the boolean enumeration).
|
||||
type hdf5OutSet struct {
|
||||
path string
|
||||
shape []int
|
||||
class byte
|
||||
width int
|
||||
// signed marks a fixed-point payload as two's complement: the class
|
||||
// bit field's bit 0x08, the bit the datatype message carries and
|
||||
// the reader keys its landing on.
|
||||
signed bool
|
||||
// nbytes is the payload's byte size, which the plan has bounded.
|
||||
nbytes int
|
||||
// The source lives in exactly one of the payload fields below, the
|
||||
// one its class, width and signedness name: a contiguous write
|
||||
// encodes the values straight into the image and a chunked write
|
||||
// stages them chunk by chunk, so no encoded copy of the payload is
|
||||
// built. Every numeric payload is serialised little-endian at its
|
||||
// own stored width, bool as one zero-or-one byte per element.
|
||||
bools []bool
|
||||
ints []int64
|
||||
i8s []int8
|
||||
u8s []uint8
|
||||
i16s []int16
|
||||
u16s []uint16
|
||||
i32s []int32
|
||||
u32s []uint32
|
||||
f32s []float32
|
||||
f64s []float64
|
||||
texts []string
|
||||
attrs []hdf5Attr
|
||||
// written state
|
||||
addr uint64
|
||||
}
|
||||
|
||||
// hdf5OutNode is one group of the plan: the root is the node whose
|
||||
// path is "/". Groups sort their children by name before writing so
|
||||
// the heap offsets and the B-tree order agree.
|
||||
type hdf5OutNode struct {
|
||||
path string
|
||||
name string
|
||||
attrs []hdf5Attr
|
||||
groups []*hdf5OutNode
|
||||
sets []*hdf5OutSet
|
||||
// written state: the object header address and, in the classic
|
||||
// layout, the group's B-tree and local heap.
|
||||
addr uint64
|
||||
btree uint64
|
||||
heap uint64
|
||||
}
|
||||
|
||||
// hdf5BuildPlan validates the paths, the dtypes and the attribute
|
||||
// texts and lays the file out as a tree: the datasets under their
|
||||
// groups, every attribute sorted by name. Anything the writer would
|
||||
// refuse it refuses here, before a byte is written.
|
||||
func hdf5BuildPlan(name string, datasets []HDF5Dataset, texts []HDF5TextDataset, groupAttrs map[string]map[string]string) (*hdf5OutNode, error) {
|
||||
root := &hdf5OutNode{path: "/"}
|
||||
groups := map[string]*hdf5OutNode{"/": root}
|
||||
used := map[string]bool{"/": true}
|
||||
var ensure func(path string) (*hdf5OutNode, error)
|
||||
ensure = func(path string) (*hdf5OutNode, error) {
|
||||
if g, ok := groups[path]; ok {
|
||||
return g, nil
|
||||
}
|
||||
segs, err := hdf5CheckPath(name, path, "group")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
parentPath := "/"
|
||||
if len(segs) > 1 {
|
||||
parentPath = "/" + strings.Join(segs[:len(segs)-1], "/")
|
||||
}
|
||||
parent, err := ensure(parentPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if used[path] {
|
||||
return nil, base.Errf("%s: %q names both a dataset and a group", name, path)
|
||||
}
|
||||
g := &hdf5OutNode{path: path, name: segs[len(segs)-1]}
|
||||
parent.groups = append(parent.groups, g)
|
||||
groups[path] = g
|
||||
used[path] = true
|
||||
return g, nil
|
||||
}
|
||||
place := func(path string) (*hdf5OutNode, error) {
|
||||
segs, err := hdf5CheckPath(name, path, "dataset")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if used[path] {
|
||||
return nil, base.Errf("%s: the path %q is written twice", name, path)
|
||||
}
|
||||
parentPath := "/"
|
||||
if len(segs) > 1 {
|
||||
parentPath = "/" + strings.Join(segs[:len(segs)-1], "/")
|
||||
}
|
||||
parent, err := ensure(parentPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
used[path] = true
|
||||
return parent, nil
|
||||
}
|
||||
for i := range datasets {
|
||||
d := &datasets[i]
|
||||
parent, err := place(d.Path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
s, err := hdf5PlanSet(name, d)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
parent.sets = append(parent.sets, s)
|
||||
}
|
||||
for i := range texts {
|
||||
tx := &texts[i]
|
||||
parent, err := place(tx.Path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
s, err := hdf5PlanText(name, tx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
parent.sets = append(parent.sets, s)
|
||||
}
|
||||
keys := slices.Sorted(maps.Keys(groupAttrs))
|
||||
for _, k := range keys {
|
||||
g, ok := groups[k]
|
||||
if !ok {
|
||||
return nil, base.Errf("%s: the attribute path %q does not name a group of this file", name, k)
|
||||
}
|
||||
attrs, err := hdf5PlanAttrs(name, k, groupAttrs[k])
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
g.attrs = attrs
|
||||
}
|
||||
hdf5SortNode(root)
|
||||
return root, nil
|
||||
}
|
||||
|
||||
// hdf5CheckPath splits an absolute object path into its segments and
|
||||
// refuses what no file should carry: a relative path, the root path
|
||||
// where an object is wanted, an empty segment, a segment holding a
|
||||
// NUL byte or a path longer than the bound a sane file keeps.
|
||||
func hdf5CheckPath(name, path, kind string) ([]string, error) {
|
||||
if path == "" || path[0] != '/' {
|
||||
return nil, base.Errf("%s: %s path %q is not absolute", name, kind, path)
|
||||
}
|
||||
if path == "/" {
|
||||
return nil, base.Errf("%s: the root path does not name a %s", name, kind)
|
||||
}
|
||||
if len(path) > hdf5MaxNameBytes {
|
||||
return nil, base.Errf("%s: %s path %q is longer than %d bytes", name, kind, path, hdf5MaxNameBytes)
|
||||
}
|
||||
segs := strings.Split(path[1:], "/")
|
||||
for _, s := range segs {
|
||||
if s == "" {
|
||||
return nil, base.Errf("%s: %s path %q has an empty segment", name, kind, path)
|
||||
}
|
||||
if strings.IndexByte(s, 0) >= 0 {
|
||||
return nil, base.Errf("%s: %s path %q holds a NUL byte", name, kind, path)
|
||||
}
|
||||
}
|
||||
return segs, nil
|
||||
}
|
||||
|
||||
// hdf5SortNode orders the children of a group by name, recursively, so
|
||||
// the written file is independent of the order the caller supplied.
|
||||
func hdf5SortNode(g *hdf5OutNode) {
|
||||
slices.SortFunc(g.groups, func(a, b *hdf5OutNode) int { return strings.Compare(a.name, b.name) })
|
||||
slices.SortFunc(g.sets, func(a, b *hdf5OutSet) int { return strings.Compare(a.path, b.path) })
|
||||
for _, sub := range g.groups {
|
||||
hdf5SortNode(sub)
|
||||
}
|
||||
}
|
||||
|
||||
// hdf5ImageEstimate bounds the byte size of the image the plan writes,
|
||||
// the number the buffer is allocated from. The bound is loose on
|
||||
// purpose: every structure the format wraps around a payload is
|
||||
// charged a fixed frame, a deflated chunk cannot grow past its own
|
||||
// bytes, and a chunked dataset is charged one chunk of padding plus a
|
||||
// node per sixty-four chunks. Over-estimating costs the memory the
|
||||
// write frees again; under-estimating costs one reallocation.
|
||||
func hdf5ImageEstimate(g *hdf5OutNode, opts HDF5WriteOptions) int {
|
||||
// The sum is kept in int64 and clamped to what an int holds, so no
|
||||
// pile of frames can wrap it into a negative capacity.
|
||||
return int(min(hdf5NodeEstimate(g, opts), maxInt))
|
||||
}
|
||||
|
||||
// maxInt is the largest value an int holds on this platform.
|
||||
const maxInt = int64(^uint(0) >> 1)
|
||||
|
||||
// hdf5NodeEstimate sums one group's own frame, its attributes and the
|
||||
// datasets and subgroups beneath it.
|
||||
func hdf5NodeEstimate(g *hdf5OutNode, opts HDF5WriteOptions) int64 {
|
||||
total := int64(4096+128*(len(g.groups)+len(g.sets))) + hdf5AttrsEstimate(g.attrs)
|
||||
for _, s := range g.sets {
|
||||
total += hdf5SetEstimate(s, opts)
|
||||
}
|
||||
for _, sub := range g.groups {
|
||||
total += hdf5NodeEstimate(sub, opts)
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// hdf5SetEstimate bounds the image bytes one dataset occupies: its
|
||||
// stored payload, its object header and, when it is chunked, the
|
||||
// padding, the chunk B-tree nodes and the filter that shrinks it.
|
||||
func hdf5SetEstimate(s *hdf5OutSet, opts HDF5WriteOptions) int64 {
|
||||
total := int64(s.nbytes+s.nbytes/64) + 4096
|
||||
if len(s.shape) == 0 || s.nbytes == 0 || s.class == 3 || (opts.Gzip == 0 && !opts.Shuffle) {
|
||||
return total
|
||||
}
|
||||
chunkTarget := hdf5ChunkTargetOf(opts)
|
||||
// The node count is a starting hint the write grows past by append,
|
||||
// never a bound it must honour, so it is capped at what a 512-byte
|
||||
// target would charge: a pathological chunk target of one or two
|
||||
// bytes would otherwise charge a node, and its kilobyte of image,
|
||||
// to every single data byte up front.
|
||||
nodes := 1 + min(s.nbytes/max(chunkTarget/2, 1), s.nbytes/512+1, 1<<20)
|
||||
return total + int64(chunkTarget) + int64(nodes)*int64(hdf5ChunkNodeSize(len(s.shape)))
|
||||
}
|
||||
|
||||
// hdf5AttrsEstimate bounds the attribute messages of one object: every
|
||||
// value's bytes are at most four times the text they are parsed from,
|
||||
// and the fields around them are charged a fixed frame.
|
||||
func hdf5AttrsEstimate(attrs []hdf5Attr) int64 {
|
||||
var total int64
|
||||
for _, a := range attrs {
|
||||
total += int64(4*(len(a.name)+len(a.text)) + 256)
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// hdf5PlanSet validates one numeric dataset and keeps its values for
|
||||
// the write, which encodes them little-endian, the byte order the
|
||||
// datatype message declares.
|
||||
func hdf5PlanSet(name string, d *HDF5Dataset) (*hdf5OutSet, error) {
|
||||
if d.Values == nil {
|
||||
return nil, base.Errf("%s: dataset %q has no values", name, d.Path)
|
||||
}
|
||||
var class byte
|
||||
var width int
|
||||
var signed bool
|
||||
var bools []bool
|
||||
var ints []int64
|
||||
var i8s []int8
|
||||
var u8s []uint8
|
||||
var i16s []int16
|
||||
var u16s []uint16
|
||||
var i32s []int32
|
||||
var u32s []uint32
|
||||
var f32s []float32
|
||||
var f64s []float64
|
||||
switch d.Values.Dtype() {
|
||||
case core.Bool:
|
||||
class, width, bools = 8, 1, d.Values.RawBools()
|
||||
case core.Int8:
|
||||
class, width, signed, i8s = 0, 1, true, d.Values.RawInt8s()
|
||||
case core.Uint8:
|
||||
class, width, u8s = 0, 1, d.Values.RawUint8s()
|
||||
case core.Int16:
|
||||
class, width, signed, i16s = 0, 2, true, d.Values.RawInt16s()
|
||||
case core.Uint16:
|
||||
class, width, u16s = 0, 2, d.Values.RawUint16s()
|
||||
case core.Int32:
|
||||
class, width, signed, i32s = 0, 4, true, d.Values.RawInt32s()
|
||||
case core.Uint32:
|
||||
class, width, u32s = 0, 4, d.Values.RawUint32s()
|
||||
case core.Int:
|
||||
class, width, signed, ints = 0, 8, true, d.Values.RawInts()
|
||||
case core.Float32:
|
||||
class, width, f32s = 1, 4, d.Values.RawFloat32s()
|
||||
case core.Float:
|
||||
class, width, f64s = 1, 8, d.Values.RawFloats()
|
||||
default:
|
||||
return nil, base.Errf("%s: dataset %q: dtype %s is not supported; the writer stores bool, int8, uint8, int16, uint16, int32, uint32, int64, float32 and float64", name, d.Path, d.Values.Dtype())
|
||||
}
|
||||
shape := d.Values.Shape()
|
||||
if d.Shape != nil && !slices.Equal(d.Shape, shape) {
|
||||
return nil, base.Errf("%s: dataset %q declares a shape of %v for values shaped %v", name, d.Path, d.Shape, shape)
|
||||
}
|
||||
if len(shape) > hdf5MaxRank {
|
||||
return nil, base.Errf("%s: dataset %q has %d dimensions, the format allows %d", name, d.Path, len(shape), hdf5MaxRank)
|
||||
}
|
||||
// The payload's byte size is bounded here, in the plan, so the write
|
||||
// can reserve it whole: this check is what stands between a wrapped
|
||||
// shape and a reservation past the budget.
|
||||
n, err := hdf5ByteExtent(shape, width, hdf5MaxDatasetBytes)
|
||||
if err != nil {
|
||||
return nil, base.Errf("%s: dataset %q: %w", name, d.Path, err)
|
||||
}
|
||||
attrs, err := hdf5PlanAttrs(name, d.Path, d.Attrs)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &hdf5OutSet{
|
||||
path: d.Path, shape: shape, class: class, width: width, signed: signed,
|
||||
nbytes: int(n), bools: bools, ints: ints, i8s: i8s, u8s: u8s,
|
||||
i16s: i16s, u16s: u16s, i32s: i32s, u32s: u32s,
|
||||
f32s: f32s, f64s: f64s, attrs: attrs,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// hdf5PlanText validates one string dataset and pads its elements to
|
||||
// the longest one, the fixed length the datatype message declares.
|
||||
func hdf5PlanText(name string, tx *HDF5TextDataset) (*hdf5OutSet, error) {
|
||||
if len(tx.Shape) > hdf5MaxRank {
|
||||
return nil, base.Errf("%s: dataset %q has %d dimensions, the format allows %d", name, tx.Path, len(tx.Shape), hdf5MaxRank)
|
||||
}
|
||||
n := 1
|
||||
for _, d := range tx.Shape {
|
||||
if d < 0 {
|
||||
return nil, base.Errf("%s: dataset %q has the negative extent %d", name, tx.Path, d)
|
||||
}
|
||||
n *= d
|
||||
}
|
||||
if len(tx.Text) != n {
|
||||
return nil, base.Errf("%s: dataset %q holds %d strings for a shape of %d elements", name, tx.Path, len(tx.Text), n)
|
||||
}
|
||||
width := 1
|
||||
for _, s := range tx.Text {
|
||||
if strings.IndexByte(s, 0) >= 0 {
|
||||
return nil, base.Errf("%s: dataset %q holds a string with a NUL byte, which a fixed-length element cannot carry", name, tx.Path)
|
||||
}
|
||||
width = max(width, len(s))
|
||||
}
|
||||
// The same byte budget the numeric plan answers to: a shape whose
|
||||
// extents wrap the element count onto len(nil) would otherwise pass
|
||||
// the length check and write a header declaring data it does not
|
||||
// store.
|
||||
if _, err := hdf5ByteExtent(tx.Shape, width, hdf5MaxDatasetBytes); err != nil {
|
||||
return nil, base.Errf("%s: dataset %q: %w", name, tx.Path, err)
|
||||
}
|
||||
// Every element occupies the fixed width; the write lays the
|
||||
// strings into the image itself, where the padding behind each is
|
||||
// the zero the buffer already holds.
|
||||
return &hdf5OutSet{path: tx.Path, shape: tx.Shape, class: 3, width: width, nbytes: n * width, texts: tx.Text}, nil
|
||||
}
|
||||
|
||||
// encode lays the dataset's payload into dst, which holds exactly the
|
||||
// nbytes the plan bounded: every numeric value little-endian at its own
|
||||
// stored width, bool as one zero-or-one byte per element, the strings
|
||||
// each into the fixed-width slot its index names. The zeros dst arrives
|
||||
// with are the padding behind every string, so the encoder writes only
|
||||
// the bytes the values themselves fill. A class or width the plan never
|
||||
// produces is a loud error, never a silent zero payload.
|
||||
func (s *hdf5OutSet) encode(dst []byte) error {
|
||||
switch s.class {
|
||||
case 0:
|
||||
switch {
|
||||
case s.width == 1 && s.signed:
|
||||
for i, v := range s.i8s {
|
||||
dst[i] = byte(v)
|
||||
}
|
||||
case s.width == 1:
|
||||
for i, v := range s.u8s {
|
||||
dst[i] = v
|
||||
}
|
||||
case s.width == 2 && s.signed:
|
||||
for i, v := range s.i16s {
|
||||
binary.LittleEndian.PutUint16(dst[i*2:], uint16(v))
|
||||
}
|
||||
case s.width == 2:
|
||||
for i, v := range s.u16s {
|
||||
binary.LittleEndian.PutUint16(dst[i*2:], v)
|
||||
}
|
||||
case s.width == 4 && s.signed:
|
||||
for i, v := range s.i32s {
|
||||
binary.LittleEndian.PutUint32(dst[i*4:], uint32(v))
|
||||
}
|
||||
case s.width == 4:
|
||||
for i, v := range s.u32s {
|
||||
binary.LittleEndian.PutUint32(dst[i*4:], v)
|
||||
}
|
||||
case s.width == 8:
|
||||
for i, v := range s.ints {
|
||||
binary.LittleEndian.PutUint64(dst[i*8:], uint64(v))
|
||||
}
|
||||
default:
|
||||
return s.payloadRefusal()
|
||||
}
|
||||
case 1:
|
||||
switch s.width {
|
||||
case 4:
|
||||
for i, v := range s.f32s {
|
||||
binary.LittleEndian.PutUint32(dst[i*4:], math.Float32bits(v))
|
||||
}
|
||||
case 8:
|
||||
for i, v := range s.f64s {
|
||||
binary.LittleEndian.PutUint64(dst[i*8:], math.Float64bits(v))
|
||||
}
|
||||
default:
|
||||
return s.payloadRefusal()
|
||||
}
|
||||
case 3:
|
||||
for i, t := range s.texts {
|
||||
copy(dst[i*s.width:], t)
|
||||
}
|
||||
case 8:
|
||||
// The enumeration's member values, one byte per element.
|
||||
for i, v := range s.bools {
|
||||
if v {
|
||||
dst[i] = 1
|
||||
} else {
|
||||
dst[i] = 0
|
||||
}
|
||||
}
|
||||
default:
|
||||
return s.payloadRefusal()
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// payloadRefusal names the datatype an encoder cannot serialise. The
|
||||
// plan produces none of them, so reaching one is a writer defect, and
|
||||
// it fails loudly rather than emitting a payload of silent zeros.
|
||||
func (s *hdf5OutSet) payloadRefusal() error {
|
||||
return base.Errf("dataset %q: the writer cannot serialise datatype class %d of %d bytes per element", s.path, s.class, s.width)
|
||||
}
|
||||
|
||||
// hdf5Attr is one attribute of the plan: its name and the text its
|
||||
// value is written from.
|
||||
type hdf5Attr struct {
|
||||
name string
|
||||
text string
|
||||
}
|
||||
|
||||
// hdf5PlanAttrs validates and sorts the attributes of one object: a
|
||||
// name must be non-empty and free of NUL bytes, the same constraint
|
||||
// the reader's rendering can round-trip under.
|
||||
func hdf5PlanAttrs(name, path string, attrs map[string]string) ([]hdf5Attr, error) {
|
||||
if len(attrs) > hdf5MaxAttrs {
|
||||
return nil, base.Errf("%s: %q carries %d attributes, past the %d the writer stores in one object header", name, path, len(attrs), hdf5MaxAttrs)
|
||||
}
|
||||
keys := slices.Sorted(maps.Keys(attrs))
|
||||
out := make([]hdf5Attr, 0, len(keys))
|
||||
for _, k := range keys {
|
||||
if k == "" {
|
||||
return nil, base.Errf("%s: %q carries an attribute with an empty name", name, path)
|
||||
}
|
||||
if len(k)+1 > 0xffff {
|
||||
return nil, base.Errf("%s: %q carries the attribute %q whose name is past the %d bytes the attribute message counts", name, path, k, 0xffff)
|
||||
}
|
||||
if strings.IndexByte(k, 0) >= 0 || strings.IndexByte(attrs[k], 0) >= 0 {
|
||||
return nil, base.Errf("%s: %q carries the attribute %q with a NUL byte in its name or value", name, path, k)
|
||||
}
|
||||
out = append(out, hdf5Attr{name: k, text: attrs[k]})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// hdf5Writer builds the file image: every address is an offset into
|
||||
// buf, so structures written later can be referenced by structures
|
||||
// written earlier through the patch at the end. The chunk staging
|
||||
// fields are reused across every chunk of one write: the gather and
|
||||
// shuffle buffers and the index vectors grow to the largest chunk the
|
||||
// write lays out, and the deflater carries one compressor and one
|
||||
// output buffer for the whole file.
|
||||
type hdf5Writer struct {
|
||||
buf []byte
|
||||
latest bool
|
||||
opts HDF5WriteOptions
|
||||
sbAddr uint64
|
||||
|
||||
gatherScratch []byte
|
||||
shuffleScratch []byte
|
||||
chunkIdx []int
|
||||
comp *zlib.Writer
|
||||
compBuf bytes.Buffer
|
||||
compLevel int
|
||||
}
|
||||
|
||||
// write lays the file out the way the reference library builds it: a
|
||||
// placeholder superblock first (its fields name the end of the file
|
||||
// and the root group, which are known only once everything is
|
||||
// written), then the root group, whose header is allocated before its
|
||||
// subtree and filled once the subtree has addresses, and the
|
||||
// superblock itself last.
|
||||
func (w *hdf5Writer) write(root *hdf5OutNode) error {
|
||||
// The image is built into one buffer, so it is allocated once from
|
||||
// the plan's own size: growing it as the structures are laid out
|
||||
// would copy the whole file at every step.
|
||||
w.buf = make([]byte, 0, hdf5ImageEstimate(root, w.opts))
|
||||
if w.latest {
|
||||
w.sbAddr = w.alloc(hdf5Superblock3Size)
|
||||
} else {
|
||||
w.sbAddr = w.alloc(hdf5Superblock0Size)
|
||||
}
|
||||
if err := w.writeGroup(root); err != nil {
|
||||
return err
|
||||
}
|
||||
w.finishSuperblock(root)
|
||||
return nil
|
||||
}
|
||||
|
||||
// writeGroup dispatches to the group writer of the file's layout.
|
||||
func (w *hdf5Writer) writeGroup(g *hdf5OutNode) error {
|
||||
if w.latest {
|
||||
return w.writeNewGroup(g)
|
||||
}
|
||||
return w.writeClassicGroup(g)
|
||||
}
|
||||
|
||||
// Sizes of the fixed parts the writer places first. Every address and
|
||||
// length is eight bytes, as in the fixtures.
|
||||
const (
|
||||
hdf5Superblock0Size = 24 + 4*8 + 8 + 8 + 4 + 4 + 16 // 96
|
||||
hdf5Superblock3Size = 12 + 4*8 + 4 // 48
|
||||
)
|
||||
|
||||
// finishSuperblock fills the placeholder: the classic superblock
|
||||
// names the root group through a symbol table entry whose cache holds
|
||||
// the group's B-tree and local heap, the latest one names its object
|
||||
// header directly and checksums the whole block with lookup3.
|
||||
func (w *hdf5Writer) finishSuperblock(root *hdf5OutNode) {
|
||||
copy(w.buf[w.sbAddr:], hdf5Magic)
|
||||
if w.latest {
|
||||
w.buf[w.sbAddr+8] = 3
|
||||
w.buf[w.sbAddr+9] = 8
|
||||
w.buf[w.sbAddr+10] = 8
|
||||
w.buf[w.sbAddr+11] = 0 // file consistency flags
|
||||
w.set64(w.sbAddr+12, 0)
|
||||
w.set64(w.sbAddr+20, math.MaxUint64)
|
||||
w.set64(w.sbAddr+28, uint64(len(w.buf)))
|
||||
w.set64(w.sbAddr+36, root.addr)
|
||||
w.set32(w.sbAddr+44, hdf5Lookup3(w.buf[w.sbAddr:w.sbAddr+44]))
|
||||
return
|
||||
}
|
||||
w.buf[w.sbAddr+8] = 0 // superblock version
|
||||
w.buf[w.sbAddr+9] = 0 // free space storage version
|
||||
w.buf[w.sbAddr+10] = 0 // root group symbol table entry version
|
||||
w.buf[w.sbAddr+11] = 0 // reserved
|
||||
w.buf[w.sbAddr+12] = 0 // shared header message format version
|
||||
w.buf[w.sbAddr+13] = 8 // size of offsets
|
||||
w.buf[w.sbAddr+14] = 8 // size of lengths
|
||||
w.buf[w.sbAddr+15] = 0 // reserved
|
||||
binary.LittleEndian.PutUint16(w.buf[w.sbAddr+16:], hdf5GroupLeafK)
|
||||
binary.LittleEndian.PutUint16(w.buf[w.sbAddr+18:], hdf5GroupInnerK)
|
||||
// The file consistency flags at +20 stay zero.
|
||||
w.set64(w.sbAddr+24, 0) // base address
|
||||
w.set64(w.sbAddr+32, math.MaxUint64) // free space information
|
||||
w.set64(w.sbAddr+40, uint64(len(w.buf))) // end of file
|
||||
w.set64(w.sbAddr+48, math.MaxUint64) // driver information
|
||||
w.set64(w.sbAddr+56, 0) // root entry: link name offset
|
||||
w.set64(w.sbAddr+64, root.addr) // root entry: object header address
|
||||
w.set32(w.sbAddr+72, 1) // root entry: symbol table cache
|
||||
w.set64(w.sbAddr+80, root.btree) // cache: B-tree address
|
||||
w.set64(w.sbAddr+88, root.heap) // cache: local heap address
|
||||
}
|
||||
|
||||
func (w *hdf5Writer) alloc(n int) uint64 {
|
||||
addr := uint64(len(w.buf))
|
||||
w.buf = append(w.buf, make([]byte, n)...)
|
||||
return addr
|
||||
}
|
||||
|
||||
// reserve extends the image by n bytes without writing them and
|
||||
// returns the address they start at; the caller fills the whole span
|
||||
// in the same breath, so the payload passes through the writer once.
|
||||
// The capacity beyond len(buf) always holds the zeros the buffer's
|
||||
// allocations left there, which every fixed structure and string slot
|
||||
// is padded from, and the encode that follows a reservation writes
|
||||
// every byte the payload itself does not.
|
||||
func (w *hdf5Writer) reserve(n int) uint64 {
|
||||
addr := uint64(len(w.buf))
|
||||
if cap(w.buf)-len(w.buf) < n {
|
||||
w.buf = append(w.buf, make([]byte, n)...)
|
||||
return addr
|
||||
}
|
||||
w.buf = w.buf[:len(w.buf)+n]
|
||||
return addr
|
||||
}
|
||||
|
||||
func (w *hdf5Writer) bytes(b []byte) uint64 {
|
||||
addr := uint64(len(w.buf))
|
||||
w.buf = append(w.buf, b...)
|
||||
return addr
|
||||
}
|
||||
|
||||
// pad8 aligns the image to the eight-byte boundary the format inserts
|
||||
// between the structures of the classic layout.
|
||||
func (w *hdf5Writer) pad8() {
|
||||
if r := len(w.buf) % 8; r != 0 {
|
||||
w.buf = append(w.buf, make([]byte, 8-r)...)
|
||||
}
|
||||
}
|
||||
|
||||
func (w *hdf5Writer) set32(at uint64, v uint32) {
|
||||
binary.LittleEndian.PutUint32(w.buf[at:], v)
|
||||
}
|
||||
|
||||
func (w *hdf5Writer) set64(at uint64, v uint64) {
|
||||
binary.LittleEndian.PutUint64(w.buf[at:], v)
|
||||
}
|
||||
Reference in New Issue
Block a user