Files
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

787 lines
28 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package io
// HDF5 writing, the mirror of the reader in hdf5.go. The reader's
// verified decoders are the specification: every structure here is
// written in the shape the reader accepts and in the shape the
// reference library writes, as the fixtures under testdata/h5 pin it.
//
// Written: superblock version 0 (the classic layout, the default) and
// version 3 (the "latest" layout, whose superblock and object headers
// carry lookup3 checksums), object headers versions 1 and 2, groups
// stored as symbol tables (local heap, version 1 group B-tree, symbol
// table nodes) in the classic layout or as link messages in the latest
// one, datasets stored contiguously or in chunks through a version 1
// chunk B-tree, the deflate and shuffle filters, fixed-point and
// floating-point datatypes of the usual widths, the boolean
// enumeration convention, fixed-length string datatypes, and
// attributes in the object header.
//
// Every address and length is eight bytes, as in the fixtures. The
// output is deterministic: children are written in sorted name order
// and nothing depends on map iteration.
import (
"bytes"
"compress/zlib"
"encoding/binary"
"maps"
"math"
"os"
"slices"
"strings"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
// HDF5WriteOptions tunes SaveHDF5 and SaveHDF5Text. The zero value
// writes the classic file layout with contiguous datasets, which every
// reader of the format understands.
type HDF5WriteOptions struct {
// Latest writes superblock version 3 with version 2 object
// headers: groups become link messages and every structure the
// format checksums carries a lookup3 sum. Latest files hold
// contiguous datasets only: the reference library stores filtered
// chunks of a latest file in a version 2 B-tree, which LoadHDF5
// does not read, so combining Latest with a filter is refused.
Latest bool
// Gzip applies the deflate filter to every numeric dataset at the
// given level: 0 (the default) disables it, -1 means the default
// level and 1 to 9 are the levels of the format. A filtered
// dataset is stored in chunks.
Gzip int
// Shuffle applies the shuffle filter before deflate, which
// regroups the bytes of each element so compression sees the
// high-order bytes together. Shuffle alone also forces chunks.
Shuffle bool
// ChunkBytes is the target size of one chunk in bytes for
// filtered datasets; 0 selects a default of 64 KiB. Datasets
// smaller than the target stay in one chunk.
ChunkBytes int
}
// SaveHDF5 writes the datasets as an HDF5 file: the mirror of
// LoadHDF5. The paths build the group tree (the dataset "/g/f32" sits
// in the group "/g"), so the file reads back with the same paths,
// shapes, dtypes and values. Each dataset's Attrs are written on the
// dataset itself; the attributes of the root and of the groups come
// from groupAttrs, keyed by group path with the root keyed "/".
//
// Every dtype the writer stores lands the same dtype through LoadHDF5:
// bool through the HDF5 boolean enumeration convention, the narrow
// integers at their stored width and signedness, float32, float64 and
// int64 directly. Float16 and complex arrays are refused: LoadHDF5
// decodes neither a two-byte floating-point nor a complex datatype, so
// the writer refuses them rather than write a file this package cannot
// read back. Attribute values are parsed back into typed attributes: a
// whole number becomes an int64 attribute, a decimal a float64 one, a
// bracketed list an int64 or float64 array, and anything else a
// fixed-length string, so a file written from LoadHDF5's own output
// reads back with the same attribute text. When several option values
// are passed the last one wins.
func SaveHDF5(path string, datasets []HDF5Dataset, groupAttrs map[string]map[string]string, opts ...HDF5WriteOptions) error {
const name = "SaveHDF5"
options := HDF5WriteOptions{}
for _, o := range opts {
options = o
}
if err := hdf5CheckOptions(name, options); err != nil {
return err
}
root, err := hdf5BuildPlan(name, datasets, nil, groupAttrs)
if err != nil {
return err
}
w := &hdf5Writer{latest: options.Latest, opts: options}
if err := w.write(root); err != nil {
return err
}
if err := os.WriteFile(path, w.buf, 0o644); err != nil {
return base.Errf("%s: %w", name, err)
}
return nil
}
// HDF5TextDataset is one fixed-length string dataset for SaveHDF5Text:
// Text holds the elements in row-major order, each padded to the
// longest element in the file. The reader of this package refuses
// string datasets (it reads numeric arrays only), so these files are
// for other readers of the format.
type HDF5TextDataset struct {
Path string
Shape []int
Text []string
}
// SaveHDF5Text writes fixed-length string datasets, the string side of
// the datatype message the reader refuses for data but accepts for
// attributes. The datasets are stored contiguously; the deflate and
// shuffle filters, which are chunked-storage filters, are refused
// here by name rather than silently dropped. When several option
// values are passed the last one wins.
func SaveHDF5Text(path string, texts []HDF5TextDataset, opts ...HDF5WriteOptions) error {
const name = "SaveHDF5Text"
options := HDF5WriteOptions{}
for _, o := range opts {
options = o
}
if options.Gzip != 0 || options.Shuffle {
return base.Errf("%s: the deflate and shuffle filters apply to numeric datasets; string datasets are written contiguously", name)
}
if err := hdf5CheckOptions(name, options); err != nil {
return err
}
root, err := hdf5BuildPlan(name, nil, texts, nil)
if err != nil {
return err
}
w := &hdf5Writer{latest: options.Latest, opts: options}
if err := w.write(root); err != nil {
return err
}
if err := os.WriteFile(path, w.buf, 0o644); err != nil {
return base.Errf("%s: %w", name, err)
}
return nil
}
// hdf5CheckOptions rejects the option combinations the writer cannot
// honour, naming each one.
func hdf5CheckOptions(name string, opts HDF5WriteOptions) error {
switch opts.Gzip {
case 0, -1, 1, 2, 3, 4, 5, 6, 7, 8, 9:
default:
return base.Errf("%s: gzip level %d: use 0 to disable the filter, -1 for the default level or 1 to 9", name, opts.Gzip)
}
if opts.ChunkBytes < 0 {
return base.Errf("%s: a chunk target of %d bytes is negative", name, opts.ChunkBytes)
}
if opts.Latest && (opts.Gzip != 0 || opts.Shuffle) {
return base.Errf("%s: the latest format stores filtered chunks through a version 2 B-tree, which LoadHDF5 does not read; write filtered datasets to a classic file", name)
}
return nil
}
// The B-tree and symbol node fanouts the classic layout declares in
// its superblock: a node of the format holds twice the K of its kind,
// so a symbol table node holds eight entries, a group B-tree node
// thirty-two children and a chunk B-tree node (whose K the format
// fixes at thirty-two for superblock version 0) sixty-four chunks.
const (
hdf5GroupLeafK = 4
hdf5GroupInnerK = 16
hdf5IStoreK = 32
hdf5ChunkTarget = 64 << 10
hdf5MaxChunks = 4 << 20
hdf5MaxAttrs = 4096
hdf5MaxRank = 32
hdf5MaxNameBytes = 4096
)
// hdf5OutSet is one dataset of the plan: the source of its payload,
// its shape and its datatype class (0 fixed-point, 1 floating-point,
// 3 string, 8 the boolean enumeration).
type hdf5OutSet struct {
path string
shape []int
class byte
width int
// signed marks a fixed-point payload as two's complement: the class
// bit field's bit 0x08, the bit the datatype message carries and
// the reader keys its landing on.
signed bool
// nbytes is the payload's byte size, which the plan has bounded.
nbytes int
// The source lives in exactly one of the payload fields below, the
// one its class, width and signedness name: a contiguous write
// encodes the values straight into the image and a chunked write
// stages them chunk by chunk, so no encoded copy of the payload is
// built. Every numeric payload is serialised little-endian at its
// own stored width, bool as one zero-or-one byte per element.
bools []bool
ints []int64
i8s []int8
u8s []uint8
i16s []int16
u16s []uint16
i32s []int32
u32s []uint32
f32s []float32
f64s []float64
texts []string
attrs []hdf5Attr
// written state
addr uint64
}
// hdf5OutNode is one group of the plan: the root is the node whose
// path is "/". Groups sort their children by name before writing so
// the heap offsets and the B-tree order agree.
type hdf5OutNode struct {
path string
name string
attrs []hdf5Attr
groups []*hdf5OutNode
sets []*hdf5OutSet
// written state: the object header address and, in the classic
// layout, the group's B-tree and local heap.
addr uint64
btree uint64
heap uint64
}
// hdf5BuildPlan validates the paths, the dtypes and the attribute
// texts and lays the file out as a tree: the datasets under their
// groups, every attribute sorted by name. Anything the writer would
// refuse it refuses here, before a byte is written.
func hdf5BuildPlan(name string, datasets []HDF5Dataset, texts []HDF5TextDataset, groupAttrs map[string]map[string]string) (*hdf5OutNode, error) {
root := &hdf5OutNode{path: "/"}
groups := map[string]*hdf5OutNode{"/": root}
used := map[string]bool{"/": true}
var ensure func(path string) (*hdf5OutNode, error)
ensure = func(path string) (*hdf5OutNode, error) {
if g, ok := groups[path]; ok {
return g, nil
}
segs, err := hdf5CheckPath(name, path, "group")
if err != nil {
return nil, err
}
parentPath := "/"
if len(segs) > 1 {
parentPath = "/" + strings.Join(segs[:len(segs)-1], "/")
}
parent, err := ensure(parentPath)
if err != nil {
return nil, err
}
if used[path] {
return nil, base.Errf("%s: %q names both a dataset and a group", name, path)
}
g := &hdf5OutNode{path: path, name: segs[len(segs)-1]}
parent.groups = append(parent.groups, g)
groups[path] = g
used[path] = true
return g, nil
}
place := func(path string) (*hdf5OutNode, error) {
segs, err := hdf5CheckPath(name, path, "dataset")
if err != nil {
return nil, err
}
if used[path] {
return nil, base.Errf("%s: the path %q is written twice", name, path)
}
parentPath := "/"
if len(segs) > 1 {
parentPath = "/" + strings.Join(segs[:len(segs)-1], "/")
}
parent, err := ensure(parentPath)
if err != nil {
return nil, err
}
used[path] = true
return parent, nil
}
for i := range datasets {
d := &datasets[i]
parent, err := place(d.Path)
if err != nil {
return nil, err
}
s, err := hdf5PlanSet(name, d)
if err != nil {
return nil, err
}
parent.sets = append(parent.sets, s)
}
for i := range texts {
tx := &texts[i]
parent, err := place(tx.Path)
if err != nil {
return nil, err
}
s, err := hdf5PlanText(name, tx)
if err != nil {
return nil, err
}
parent.sets = append(parent.sets, s)
}
keys := slices.Sorted(maps.Keys(groupAttrs))
for _, k := range keys {
g, ok := groups[k]
if !ok {
return nil, base.Errf("%s: the attribute path %q does not name a group of this file", name, k)
}
attrs, err := hdf5PlanAttrs(name, k, groupAttrs[k])
if err != nil {
return nil, err
}
g.attrs = attrs
}
hdf5SortNode(root)
return root, nil
}
// hdf5CheckPath splits an absolute object path into its segments and
// refuses what no file should carry: a relative path, the root path
// where an object is wanted, an empty segment, a segment holding a
// NUL byte or a path longer than the bound a sane file keeps.
func hdf5CheckPath(name, path, kind string) ([]string, error) {
if path == "" || path[0] != '/' {
return nil, base.Errf("%s: %s path %q is not absolute", name, kind, path)
}
if path == "/" {
return nil, base.Errf("%s: the root path does not name a %s", name, kind)
}
if len(path) > hdf5MaxNameBytes {
return nil, base.Errf("%s: %s path %q is longer than %d bytes", name, kind, path, hdf5MaxNameBytes)
}
segs := strings.Split(path[1:], "/")
for _, s := range segs {
if s == "" {
return nil, base.Errf("%s: %s path %q has an empty segment", name, kind, path)
}
if strings.IndexByte(s, 0) >= 0 {
return nil, base.Errf("%s: %s path %q holds a NUL byte", name, kind, path)
}
}
return segs, nil
}
// hdf5SortNode orders the children of a group by name, recursively, so
// the written file is independent of the order the caller supplied.
func hdf5SortNode(g *hdf5OutNode) {
slices.SortFunc(g.groups, func(a, b *hdf5OutNode) int { return strings.Compare(a.name, b.name) })
slices.SortFunc(g.sets, func(a, b *hdf5OutSet) int { return strings.Compare(a.path, b.path) })
for _, sub := range g.groups {
hdf5SortNode(sub)
}
}
// hdf5ImageEstimate bounds the byte size of the image the plan writes,
// the number the buffer is allocated from. The bound is loose on
// purpose: every structure the format wraps around a payload is
// charged a fixed frame, a deflated chunk cannot grow past its own
// bytes, and a chunked dataset is charged one chunk of padding plus a
// node per sixty-four chunks. Over-estimating costs the memory the
// write frees again; under-estimating costs one reallocation.
func hdf5ImageEstimate(g *hdf5OutNode, opts HDF5WriteOptions) int {
// The sum is kept in int64 and clamped to what an int holds, so no
// pile of frames can wrap it into a negative capacity.
return int(min(hdf5NodeEstimate(g, opts), maxInt))
}
// maxInt is the largest value an int holds on this platform.
const maxInt = int64(^uint(0) >> 1)
// hdf5NodeEstimate sums one group's own frame, its attributes and the
// datasets and subgroups beneath it.
func hdf5NodeEstimate(g *hdf5OutNode, opts HDF5WriteOptions) int64 {
total := int64(4096+128*(len(g.groups)+len(g.sets))) + hdf5AttrsEstimate(g.attrs)
for _, s := range g.sets {
total += hdf5SetEstimate(s, opts)
}
for _, sub := range g.groups {
total += hdf5NodeEstimate(sub, opts)
}
return total
}
// hdf5SetEstimate bounds the image bytes one dataset occupies: its
// stored payload, its object header and, when it is chunked, the
// padding, the chunk B-tree nodes and the filter that shrinks it.
func hdf5SetEstimate(s *hdf5OutSet, opts HDF5WriteOptions) int64 {
total := int64(s.nbytes+s.nbytes/64) + 4096
if len(s.shape) == 0 || s.nbytes == 0 || s.class == 3 || (opts.Gzip == 0 && !opts.Shuffle) {
return total
}
chunkTarget := hdf5ChunkTargetOf(opts)
// The node count is a starting hint the write grows past by append,
// never a bound it must honour, so it is capped at what a 512-byte
// target would charge: a pathological chunk target of one or two
// bytes would otherwise charge a node, and its kilobyte of image,
// to every single data byte up front.
nodes := 1 + min(s.nbytes/max(chunkTarget/2, 1), s.nbytes/512+1, 1<<20)
return total + int64(chunkTarget) + int64(nodes)*int64(hdf5ChunkNodeSize(len(s.shape)))
}
// hdf5AttrsEstimate bounds the attribute messages of one object: every
// value's bytes are at most four times the text they are parsed from,
// and the fields around them are charged a fixed frame.
func hdf5AttrsEstimate(attrs []hdf5Attr) int64 {
var total int64
for _, a := range attrs {
total += int64(4*(len(a.name)+len(a.text)) + 256)
}
return total
}
// hdf5PlanSet validates one numeric dataset and keeps its values for
// the write, which encodes them little-endian, the byte order the
// datatype message declares.
func hdf5PlanSet(name string, d *HDF5Dataset) (*hdf5OutSet, error) {
if d.Values == nil {
return nil, base.Errf("%s: dataset %q has no values", name, d.Path)
}
var class byte
var width int
var signed bool
var bools []bool
var ints []int64
var i8s []int8
var u8s []uint8
var i16s []int16
var u16s []uint16
var i32s []int32
var u32s []uint32
var f32s []float32
var f64s []float64
switch d.Values.Dtype() {
case core.Bool:
class, width, bools = 8, 1, d.Values.RawBools()
case core.Int8:
class, width, signed, i8s = 0, 1, true, d.Values.RawInt8s()
case core.Uint8:
class, width, u8s = 0, 1, d.Values.RawUint8s()
case core.Int16:
class, width, signed, i16s = 0, 2, true, d.Values.RawInt16s()
case core.Uint16:
class, width, u16s = 0, 2, d.Values.RawUint16s()
case core.Int32:
class, width, signed, i32s = 0, 4, true, d.Values.RawInt32s()
case core.Uint32:
class, width, u32s = 0, 4, d.Values.RawUint32s()
case core.Int:
class, width, signed, ints = 0, 8, true, d.Values.RawInts()
case core.Float32:
class, width, f32s = 1, 4, d.Values.RawFloat32s()
case core.Float:
class, width, f64s = 1, 8, d.Values.RawFloats()
default:
return nil, base.Errf("%s: dataset %q: dtype %s is not supported; the writer stores bool, int8, uint8, int16, uint16, int32, uint32, int64, float32 and float64", name, d.Path, d.Values.Dtype())
}
shape := d.Values.Shape()
if d.Shape != nil && !slices.Equal(d.Shape, shape) {
return nil, base.Errf("%s: dataset %q declares a shape of %v for values shaped %v", name, d.Path, d.Shape, shape)
}
if len(shape) > hdf5MaxRank {
return nil, base.Errf("%s: dataset %q has %d dimensions, the format allows %d", name, d.Path, len(shape), hdf5MaxRank)
}
// The payload's byte size is bounded here, in the plan, so the write
// can reserve it whole: this check is what stands between a wrapped
// shape and a reservation past the budget.
n, err := hdf5ByteExtent(shape, width, hdf5MaxDatasetBytes)
if err != nil {
return nil, base.Errf("%s: dataset %q: %w", name, d.Path, err)
}
attrs, err := hdf5PlanAttrs(name, d.Path, d.Attrs)
if err != nil {
return nil, err
}
return &hdf5OutSet{
path: d.Path, shape: shape, class: class, width: width, signed: signed,
nbytes: int(n), bools: bools, ints: ints, i8s: i8s, u8s: u8s,
i16s: i16s, u16s: u16s, i32s: i32s, u32s: u32s,
f32s: f32s, f64s: f64s, attrs: attrs,
}, nil
}
// hdf5PlanText validates one string dataset and pads its elements to
// the longest one, the fixed length the datatype message declares.
func hdf5PlanText(name string, tx *HDF5TextDataset) (*hdf5OutSet, error) {
if len(tx.Shape) > hdf5MaxRank {
return nil, base.Errf("%s: dataset %q has %d dimensions, the format allows %d", name, tx.Path, len(tx.Shape), hdf5MaxRank)
}
n := 1
for _, d := range tx.Shape {
if d < 0 {
return nil, base.Errf("%s: dataset %q has the negative extent %d", name, tx.Path, d)
}
n *= d
}
if len(tx.Text) != n {
return nil, base.Errf("%s: dataset %q holds %d strings for a shape of %d elements", name, tx.Path, len(tx.Text), n)
}
width := 1
for _, s := range tx.Text {
if strings.IndexByte(s, 0) >= 0 {
return nil, base.Errf("%s: dataset %q holds a string with a NUL byte, which a fixed-length element cannot carry", name, tx.Path)
}
width = max(width, len(s))
}
// The same byte budget the numeric plan answers to: a shape whose
// extents wrap the element count onto len(nil) would otherwise pass
// the length check and write a header declaring data it does not
// store.
if _, err := hdf5ByteExtent(tx.Shape, width, hdf5MaxDatasetBytes); err != nil {
return nil, base.Errf("%s: dataset %q: %w", name, tx.Path, err)
}
// Every element occupies the fixed width; the write lays the
// strings into the image itself, where the padding behind each is
// the zero the buffer already holds.
return &hdf5OutSet{path: tx.Path, shape: tx.Shape, class: 3, width: width, nbytes: n * width, texts: tx.Text}, nil
}
// encode lays the dataset's payload into dst, which holds exactly the
// nbytes the plan bounded: every numeric value little-endian at its own
// stored width, bool as one zero-or-one byte per element, the strings
// each into the fixed-width slot its index names. The zeros dst arrives
// with are the padding behind every string, so the encoder writes only
// the bytes the values themselves fill. A class or width the plan never
// produces is a loud error, never a silent zero payload.
func (s *hdf5OutSet) encode(dst []byte) error {
switch s.class {
case 0:
switch {
case s.width == 1 && s.signed:
for i, v := range s.i8s {
dst[i] = byte(v)
}
case s.width == 1:
for i, v := range s.u8s {
dst[i] = v
}
case s.width == 2 && s.signed:
for i, v := range s.i16s {
binary.LittleEndian.PutUint16(dst[i*2:], uint16(v))
}
case s.width == 2:
for i, v := range s.u16s {
binary.LittleEndian.PutUint16(dst[i*2:], v)
}
case s.width == 4 && s.signed:
for i, v := range s.i32s {
binary.LittleEndian.PutUint32(dst[i*4:], uint32(v))
}
case s.width == 4:
for i, v := range s.u32s {
binary.LittleEndian.PutUint32(dst[i*4:], v)
}
case s.width == 8:
for i, v := range s.ints {
binary.LittleEndian.PutUint64(dst[i*8:], uint64(v))
}
default:
return s.payloadRefusal()
}
case 1:
switch s.width {
case 4:
for i, v := range s.f32s {
binary.LittleEndian.PutUint32(dst[i*4:], math.Float32bits(v))
}
case 8:
for i, v := range s.f64s {
binary.LittleEndian.PutUint64(dst[i*8:], math.Float64bits(v))
}
default:
return s.payloadRefusal()
}
case 3:
for i, t := range s.texts {
copy(dst[i*s.width:], t)
}
case 8:
// The enumeration's member values, one byte per element.
for i, v := range s.bools {
if v {
dst[i] = 1
} else {
dst[i] = 0
}
}
default:
return s.payloadRefusal()
}
return nil
}
// payloadRefusal names the datatype an encoder cannot serialise. The
// plan produces none of them, so reaching one is a writer defect, and
// it fails loudly rather than emitting a payload of silent zeros.
func (s *hdf5OutSet) payloadRefusal() error {
return base.Errf("dataset %q: the writer cannot serialise datatype class %d of %d bytes per element", s.path, s.class, s.width)
}
// hdf5Attr is one attribute of the plan: its name and the text its
// value is written from.
type hdf5Attr struct {
name string
text string
}
// hdf5PlanAttrs validates and sorts the attributes of one object: a
// name must be non-empty and free of NUL bytes, the same constraint
// the reader's rendering can round-trip under.
func hdf5PlanAttrs(name, path string, attrs map[string]string) ([]hdf5Attr, error) {
if len(attrs) > hdf5MaxAttrs {
return nil, base.Errf("%s: %q carries %d attributes, past the %d the writer stores in one object header", name, path, len(attrs), hdf5MaxAttrs)
}
keys := slices.Sorted(maps.Keys(attrs))
out := make([]hdf5Attr, 0, len(keys))
for _, k := range keys {
if k == "" {
return nil, base.Errf("%s: %q carries an attribute with an empty name", name, path)
}
if len(k)+1 > 0xffff {
return nil, base.Errf("%s: %q carries the attribute %q whose name is past the %d bytes the attribute message counts", name, path, k, 0xffff)
}
if strings.IndexByte(k, 0) >= 0 || strings.IndexByte(attrs[k], 0) >= 0 {
return nil, base.Errf("%s: %q carries the attribute %q with a NUL byte in its name or value", name, path, k)
}
out = append(out, hdf5Attr{name: k, text: attrs[k]})
}
return out, nil
}
// hdf5Writer builds the file image: every address is an offset into
// buf, so structures written later can be referenced by structures
// written earlier through the patch at the end. The chunk staging
// fields are reused across every chunk of one write: the gather and
// shuffle buffers and the index vectors grow to the largest chunk the
// write lays out, and the deflater carries one compressor and one
// output buffer for the whole file.
type hdf5Writer struct {
buf []byte
latest bool
opts HDF5WriteOptions
sbAddr uint64
gatherScratch []byte
shuffleScratch []byte
chunkIdx []int
comp *zlib.Writer
compBuf bytes.Buffer
compLevel int
}
// write lays the file out the way the reference library builds it: a
// placeholder superblock first (its fields name the end of the file
// and the root group, which are known only once everything is
// written), then the root group, whose header is allocated before its
// subtree and filled once the subtree has addresses, and the
// superblock itself last.
func (w *hdf5Writer) write(root *hdf5OutNode) error {
// The image is built into one buffer, so it is allocated once from
// the plan's own size: growing it as the structures are laid out
// would copy the whole file at every step.
w.buf = make([]byte, 0, hdf5ImageEstimate(root, w.opts))
if w.latest {
w.sbAddr = w.alloc(hdf5Superblock3Size)
} else {
w.sbAddr = w.alloc(hdf5Superblock0Size)
}
if err := w.writeGroup(root); err != nil {
return err
}
w.finishSuperblock(root)
return nil
}
// writeGroup dispatches to the group writer of the file's layout.
func (w *hdf5Writer) writeGroup(g *hdf5OutNode) error {
if w.latest {
return w.writeNewGroup(g)
}
return w.writeClassicGroup(g)
}
// Sizes of the fixed parts the writer places first. Every address and
// length is eight bytes, as in the fixtures.
const (
hdf5Superblock0Size = 24 + 4*8 + 8 + 8 + 4 + 4 + 16 // 96
hdf5Superblock3Size = 12 + 4*8 + 4 // 48
)
// finishSuperblock fills the placeholder: the classic superblock
// names the root group through a symbol table entry whose cache holds
// the group's B-tree and local heap, the latest one names its object
// header directly and checksums the whole block with lookup3.
func (w *hdf5Writer) finishSuperblock(root *hdf5OutNode) {
copy(w.buf[w.sbAddr:], hdf5Magic)
if w.latest {
w.buf[w.sbAddr+8] = 3
w.buf[w.sbAddr+9] = 8
w.buf[w.sbAddr+10] = 8
w.buf[w.sbAddr+11] = 0 // file consistency flags
w.set64(w.sbAddr+12, 0)
w.set64(w.sbAddr+20, math.MaxUint64)
w.set64(w.sbAddr+28, uint64(len(w.buf)))
w.set64(w.sbAddr+36, root.addr)
w.set32(w.sbAddr+44, hdf5Lookup3(w.buf[w.sbAddr:w.sbAddr+44]))
return
}
w.buf[w.sbAddr+8] = 0 // superblock version
w.buf[w.sbAddr+9] = 0 // free space storage version
w.buf[w.sbAddr+10] = 0 // root group symbol table entry version
w.buf[w.sbAddr+11] = 0 // reserved
w.buf[w.sbAddr+12] = 0 // shared header message format version
w.buf[w.sbAddr+13] = 8 // size of offsets
w.buf[w.sbAddr+14] = 8 // size of lengths
w.buf[w.sbAddr+15] = 0 // reserved
binary.LittleEndian.PutUint16(w.buf[w.sbAddr+16:], hdf5GroupLeafK)
binary.LittleEndian.PutUint16(w.buf[w.sbAddr+18:], hdf5GroupInnerK)
// The file consistency flags at +20 stay zero.
w.set64(w.sbAddr+24, 0) // base address
w.set64(w.sbAddr+32, math.MaxUint64) // free space information
w.set64(w.sbAddr+40, uint64(len(w.buf))) // end of file
w.set64(w.sbAddr+48, math.MaxUint64) // driver information
w.set64(w.sbAddr+56, 0) // root entry: link name offset
w.set64(w.sbAddr+64, root.addr) // root entry: object header address
w.set32(w.sbAddr+72, 1) // root entry: symbol table cache
w.set64(w.sbAddr+80, root.btree) // cache: B-tree address
w.set64(w.sbAddr+88, root.heap) // cache: local heap address
}
func (w *hdf5Writer) alloc(n int) uint64 {
addr := uint64(len(w.buf))
w.buf = append(w.buf, make([]byte, n)...)
return addr
}
// reserve extends the image by n bytes without writing them and
// returns the address they start at; the caller fills the whole span
// in the same breath, so the payload passes through the writer once.
// The capacity beyond len(buf) always holds the zeros the buffer's
// allocations left there, which every fixed structure and string slot
// is padded from, and the encode that follows a reservation writes
// every byte the payload itself does not.
func (w *hdf5Writer) reserve(n int) uint64 {
addr := uint64(len(w.buf))
if cap(w.buf)-len(w.buf) < n {
w.buf = append(w.buf, make([]byte, n)...)
return addr
}
w.buf = w.buf[:len(w.buf)+n]
return addr
}
func (w *hdf5Writer) bytes(b []byte) uint64 {
addr := uint64(len(w.buf))
w.buf = append(w.buf, b...)
return addr
}
// pad8 aligns the image to the eight-byte boundary the format inserts
// between the structures of the classic layout.
func (w *hdf5Writer) pad8() {
if r := len(w.buf) % 8; r != 0 {
w.buf = append(w.buf, make([]byte, 8-r)...)
}
}
func (w *hdf5Writer) set32(at uint64, v uint32) {
binary.LittleEndian.PutUint32(w.buf[at:], v)
}
func (w *hdf5Writer) set64(at uint64, v uint64) {
binary.LittleEndian.PutUint64(w.buf[at:], v)
}