718 lines
25 KiB
Go
718 lines
25 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
package io
|
|
|
|
// Group and chunked-storage writing for the HDF5 writer. Both group
|
|
// layouts the reader accepts are written: the classic symbol table (a
|
|
// local heap of names, symbol table nodes and a version 1 group
|
|
// B-tree) and the latest-style compact group (link messages in the
|
|
// object header). Chunked storage hangs its chunks off a version 1
|
|
// chunk B-tree, whose conventions the reference files pin: every node
|
|
// is allocated at the size the format's fanout implies, an interior
|
|
// node's key is the first key of the child's subtree, and a node's
|
|
// closing key is a sentinel ordered past its last chunk.
|
|
|
|
import (
|
|
"bytes"
|
|
"compress/zlib"
|
|
"encoding/binary"
|
|
"math"
|
|
"slices"
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
|
)
|
|
|
|
// hdf5GroupKid is one child pointer of a group B-tree node. The key of
|
|
// kid i is the heap offset of the last name child i-1 holds, so a
|
|
// node's entries are headed by the zero key (the heap's null name,
|
|
// below every real name) and each entry's key closes the range of the
|
|
// child before it, which is the convention the fixtures store.
|
|
type hdf5GroupKid struct {
|
|
key uint64
|
|
child uint64
|
|
}
|
|
|
|
// hdf5GroupChild is one child entry of a group: its link name, its
|
|
// object header address and, until the children are written, the
|
|
// subgroup or dataset whose address it carries.
|
|
type hdf5GroupChild struct {
|
|
name string
|
|
addr uint64
|
|
group *hdf5OutNode
|
|
set *hdf5OutSet
|
|
}
|
|
|
|
// hdf5Kids merges a group's subgroups and datasets into one list in
|
|
// name order, the order the classic symbol table requires: the local
|
|
// heap assigns its offsets in list order and the reference library
|
|
// binary-searches both the B-tree keys and the symbol node records on
|
|
// those offsets, so a groups-first order would hide every child whose
|
|
// name interleaves with a dataset's. The latest layout's link list
|
|
// keeps the same order.
|
|
func hdf5Kids(g *hdf5OutNode) []hdf5GroupChild {
|
|
kids := make([]hdf5GroupChild, 0, len(g.groups)+len(g.sets))
|
|
for _, sub := range g.groups {
|
|
kids = append(kids, hdf5GroupChild{name: sub.name, group: sub})
|
|
}
|
|
for _, s := range g.sets {
|
|
kids = append(kids, hdf5GroupChild{name: baseName(s.path), set: s})
|
|
}
|
|
slices.SortFunc(kids, func(a, b hdf5GroupChild) int { return strings.Compare(a.name, b.name) })
|
|
return kids
|
|
}
|
|
|
|
// kidAddr is a child's written object header address.
|
|
func kidAddr(k hdf5GroupChild) uint64 {
|
|
if k.group != nil {
|
|
return k.group.addr
|
|
}
|
|
return k.set.addr
|
|
}
|
|
|
|
// writeClassicGroup writes a group in the classic layout, in the
|
|
// construction order the reference library uses: the object header is
|
|
// allocated first (over a placeholder, since the symbol table message
|
|
// can only name the B-tree and heap once they exist), then the local
|
|
// heap, then the children, and the symbol table nodes and B-tree last.
|
|
// A child group's symbol entry carries the B-tree and heap addresses
|
|
// in its cache, as the reference writes; a dataset's carries none.
|
|
func (w *hdf5Writer) writeClassicGroup(g *hdf5OutNode) error {
|
|
kids := hdf5Kids(g)
|
|
// The header placeholder: sized and shaped exactly like the final
|
|
// image, with the B-tree and heap addresses still zero.
|
|
msgs, err := w.classicGroupMsgs(g, 0, 0)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
image, err := hdf5HeaderV1(msgs)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
hdrAddr := w.bytes(image)
|
|
// The heap assigns every name its offset; the children are in
|
|
// sorted order, so offset order and name order agree and the
|
|
// B-tree's byte-offset keys sort as the names do. The heap carries
|
|
// slack behind the names, written as the free block the reference
|
|
// library leaves there: a next offset of one is the sentinel that
|
|
// ends the free list.
|
|
heapData := make([]byte, 8) // offset 0 is the null name no entry points at
|
|
offsets := make([]uint64, len(kids))
|
|
for i, k := range kids {
|
|
offsets[i] = uint64(len(heapData))
|
|
heapData = append(heapData, k.name...)
|
|
heapData = hdf5AppendAlign(append(heapData, 0))
|
|
}
|
|
size := max(len(heapData)+16, 88)
|
|
free := len(heapData)
|
|
heapData = append(heapData, make([]byte, 16)...) // the free block: next, size
|
|
binary.LittleEndian.PutUint64(heapData[free:], 1)
|
|
binary.LittleEndian.PutUint64(heapData[free+8:], uint64(size-free))
|
|
heapData = append(heapData, make([]byte, size-len(heapData))...)
|
|
dataAddr := w.bytes(heapData)
|
|
w.pad8()
|
|
heap := w.writeLocalHeap(dataAddr, size, free)
|
|
// The children: datasets, then the subgroups, each of which lays
|
|
// out its own subtree in the children's name order.
|
|
for _, s := range g.sets {
|
|
addr, err := w.writeDataset(s)
|
|
if err != nil {
|
|
// writeDataset names the dataset itself; a second wrap
|
|
// here would prefix it twice.
|
|
return err
|
|
}
|
|
s.addr = addr
|
|
}
|
|
for _, sub := range g.groups {
|
|
if err := w.writeGroup(sub); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
for i := range kids {
|
|
kids[i].addr = kidAddr(kids[i])
|
|
}
|
|
// Symbol table nodes, eight entries each: the node occupies the
|
|
// size the format's leaf K of four implies, however few entries are
|
|
// used, and an empty group still holds one.
|
|
leaves := make([]hdf5GroupKid, 0, max((len(kids)+2*hdf5GroupLeafK-1)/(2*hdf5GroupLeafK), 1))
|
|
for start := 0; start < len(kids); start += 2 * hdf5GroupLeafK {
|
|
leaves = append(leaves, w.writeSymbolNode(kids[start:min(start+2*hdf5GroupLeafK, len(kids))], offsets[start:]))
|
|
}
|
|
if len(leaves) == 0 {
|
|
leaves = append(leaves, w.writeSymbolNode(nil, nil))
|
|
}
|
|
btree := w.writeGroupTree(leaves)
|
|
// The header over the placeholder, now that the B-tree and heap
|
|
// have their final addresses.
|
|
msgs, err = w.classicGroupMsgs(g, btree, heap)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
image, err = hdf5HeaderV1(msgs)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
w.headerAt(hdrAddr, image)
|
|
g.addr, g.btree, g.heap = hdrAddr, btree, heap
|
|
return nil
|
|
}
|
|
|
|
// classicGroupMsgs builds the message list of a classic group's object
|
|
// header: the symbol table message naming the B-tree and heap, then
|
|
// the attributes in name order.
|
|
func (w *hdf5Writer) classicGroupMsgs(g *hdf5OutNode, btree, heap uint64) ([]hdf5OutMsg, error) {
|
|
msgs := []hdf5OutMsg{{typ: hdf5MsgSymbolTable, data: w.appendAddrs(nil, btree, heap)}}
|
|
for _, a := range g.attrs {
|
|
m, err := hdf5AttrMessage(a)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
msgs = append(msgs, m)
|
|
}
|
|
return msgs, nil
|
|
}
|
|
|
|
// writeSymbolNode writes one symbol table node of the given entries,
|
|
// allocated at eight slots. A child group's cache holds its B-tree and
|
|
// heap addresses; a dataset's cache stays empty.
|
|
func (w *hdf5Writer) writeSymbolNode(batch []hdf5GroupChild, offsets []uint64) hdf5GroupKid {
|
|
node := make([]byte, 8+2*hdf5GroupLeafK*(8+8+24))
|
|
copy(node, hdf5SymbolNode)
|
|
node[4] = 1
|
|
binary.LittleEndian.PutUint16(node[6:], uint16(len(batch)))
|
|
p := 8
|
|
for i, k := range batch {
|
|
binary.LittleEndian.PutUint64(node[p:], offsets[i])
|
|
binary.LittleEndian.PutUint64(node[p+8:], k.addr)
|
|
if k.group != nil {
|
|
binary.LittleEndian.PutUint32(node[p+16:], 1)
|
|
binary.LittleEndian.PutUint64(node[p+24:], k.group.btree)
|
|
binary.LittleEndian.PutUint64(node[p+32:], k.group.heap)
|
|
}
|
|
p += 8 + 8 + 24
|
|
}
|
|
leaf := hdf5GroupKid{child: w.bytes(node)}
|
|
if len(batch) > 0 {
|
|
leaf.key = offsets[len(batch)-1]
|
|
}
|
|
return leaf
|
|
}
|
|
|
|
// baseName returns the last segment of an absolute path.
|
|
func baseName(path string) string {
|
|
for i := len(path) - 1; i >= 0; i-- {
|
|
if path[i] == '/' {
|
|
return path[i+1:]
|
|
}
|
|
}
|
|
return path
|
|
}
|
|
|
|
// writeNewGroup writes a group in the latest layout, the construction
|
|
// order the reference library uses: the object header is allocated
|
|
// first over a placeholder, then the children, and the header is
|
|
// written over the placeholder once every link's address is known.
|
|
// The link info and group info messages the reference writes head the
|
|
// message list; a group without links carries no link info message:
|
|
// beside no links the reader of this package takes that message for
|
|
// dense storage, which it refuses.
|
|
func (w *hdf5Writer) writeNewGroup(g *hdf5OutNode) error {
|
|
msgs, err := w.newGroupMsgs(g)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
image, err := hdf5HeaderV2(msgs)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
hdrAddr := w.bytes(image)
|
|
for _, s := range g.sets {
|
|
addr, err := w.writeDataset(s)
|
|
if err != nil {
|
|
// writeDataset names the dataset itself; a second wrap
|
|
// here would prefix it twice.
|
|
return err
|
|
}
|
|
s.addr = addr
|
|
}
|
|
for _, sub := range g.groups {
|
|
if err := w.writeGroup(sub); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
msgs, err = w.newGroupMsgs(g)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
image, err = hdf5HeaderV2(msgs)
|
|
if err != nil {
|
|
return base.Errf("SaveHDF5: group %q: %w", g.path, err)
|
|
}
|
|
w.headerAt(hdrAddr, image)
|
|
g.addr = hdrAddr
|
|
return nil
|
|
}
|
|
|
|
// newGroupMsgs builds the message list of a latest-style group's
|
|
// object header. Addresses unknown at placeholder time are zero; the
|
|
// message sizes are the same either way.
|
|
func (w *hdf5Writer) newGroupMsgs(g *hdf5OutNode) ([]hdf5OutMsg, error) {
|
|
var msgs []hdf5OutMsg
|
|
if len(g.groups)+len(g.sets) > 0 {
|
|
undef := bytes.Repeat([]byte{0xff}, 8)
|
|
info := append([]byte{0, 0}, undef...)
|
|
info = append(info, undef...)
|
|
msgs = append(msgs,
|
|
hdf5OutMsg{typ: hdf5MsgLinkInfo, data: info},
|
|
hdf5OutMsg{typ: hdf5MsgGroupInfo, data: []byte{0, 0}},
|
|
)
|
|
}
|
|
addLink := func(name string, addr uint64) {
|
|
body := []byte{1, 0} // version 1, no creation order, one-byte length
|
|
if len(name) >= 256 {
|
|
body[1] = 0x01 // the name length widens to two bytes
|
|
body = binary.LittleEndian.AppendUint16(body, uint16(len(name)))
|
|
} else {
|
|
body = append(body, byte(len(name)))
|
|
}
|
|
body = append(body, name...)
|
|
body = binary.LittleEndian.AppendUint64(body, addr)
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgLink, data: body})
|
|
}
|
|
for _, k := range hdf5Kids(g) {
|
|
addLink(k.name, kidAddr(k))
|
|
}
|
|
for _, a := range g.attrs {
|
|
m, err := hdf5AttrMessage(a)
|
|
if err != nil {
|
|
// Unwrapped: the caller carries the group context and
|
|
// prefixes the entry point, so wrapping here would double
|
|
// both.
|
|
return nil, err
|
|
}
|
|
msgs = append(msgs, m)
|
|
}
|
|
return msgs, nil
|
|
}
|
|
|
|
// writeLocalHeap writes a local heap header in front of the data
|
|
// segment written earlier: the data segment's size, the free list head
|
|
// and the segment address.
|
|
func (w *hdf5Writer) writeLocalHeap(dataAddr uint64, size, free int) uint64 {
|
|
head := append([]byte{}, hdf5LocalHeap...)
|
|
head = append(head, 0, 0, 0, 0) // version and reserved
|
|
head = binary.LittleEndian.AppendUint64(head, uint64(size))
|
|
head = binary.LittleEndian.AppendUint64(head, uint64(free))
|
|
head = binary.LittleEndian.AppendUint64(head, dataAddr)
|
|
return w.bytes(head)
|
|
}
|
|
|
|
// writeGroupTree packs the symbol table nodes into version 1 group
|
|
// B-tree nodes of thirty-two children and returns the root's address.
|
|
func (w *hdf5Writer) writeGroupTree(leaves []hdf5GroupKid) uint64 {
|
|
kids := leaves
|
|
level := byte(0)
|
|
for len(kids) > 2*hdf5GroupInnerK {
|
|
var next []hdf5GroupKid
|
|
for start := 0; start < len(kids); start += 2 * hdf5GroupInnerK {
|
|
batch := kids[start:min(start+2*hdf5GroupInnerK, len(kids))]
|
|
addr := w.writeGroupNode(level, batch)
|
|
next = append(next, hdf5GroupKid{key: batch[len(batch)-1].key, child: addr})
|
|
}
|
|
kids = next
|
|
level++
|
|
}
|
|
return w.writeGroupNode(level, kids)
|
|
}
|
|
|
|
// writeGroupNode writes one group B-tree node, allocated at the size
|
|
// the format's internal K of sixteen implies. The sibling addresses
|
|
// are the undefined value: a zero would be a defined address, and a
|
|
// reader would follow it as a sibling node.
|
|
func (w *hdf5Writer) writeGroupNode(level byte, kids []hdf5GroupKid) uint64 {
|
|
node := make([]byte, 8+2*8+2*hdf5GroupInnerK*(8+8)+8)
|
|
copy(node, hdf5Tree)
|
|
node[4] = 0 // a group B-tree
|
|
node[5] = level
|
|
binary.LittleEndian.PutUint16(node[6:], uint16(len(kids)))
|
|
binary.LittleEndian.PutUint64(node[8:], math.MaxUint64) // left sibling
|
|
binary.LittleEndian.PutUint64(node[16:], math.MaxUint64) // right sibling
|
|
p := 24
|
|
binary.LittleEndian.PutUint64(node[p:], 0) // key[0]: the null name
|
|
p += 8
|
|
for _, k := range kids {
|
|
binary.LittleEndian.PutUint64(node[p:], k.child)
|
|
p += 8
|
|
binary.LittleEndian.PutUint64(node[p:], k.key) // the key closing this child
|
|
p += 8
|
|
}
|
|
return w.bytes(node)
|
|
}
|
|
|
|
// appendAddrs appends the given addresses as the file's eight-byte
|
|
// address fields.
|
|
func (w *hdf5Writer) appendAddrs(b []byte, addrs ...uint64) []byte {
|
|
for _, a := range addrs {
|
|
b = binary.LittleEndian.AppendUint64(b, a)
|
|
}
|
|
return b
|
|
}
|
|
|
|
// hdf5ChunkKey is a version 1 chunk B-tree key: the chunk's stored
|
|
// size, its coordinates, one per dataset dimension plus the element
|
|
// size dimension the format keys last, and the address the chunk's
|
|
// bytes occupy.
|
|
type hdf5ChunkKey struct {
|
|
size uint32
|
|
coords []uint64
|
|
addr uint64
|
|
}
|
|
|
|
// hdf5ChunkKid is one child of a chunk B-tree node with the first and
|
|
// last keys of its subtree, which become the node's own keys.
|
|
type hdf5ChunkKid struct {
|
|
first, last hdf5ChunkKey
|
|
child uint64
|
|
}
|
|
|
|
// hdf5ChunkNodeSize is the size a chunk B-tree node occupies: the
|
|
// header, twice the storage K of thirty-two keys with their child
|
|
// pointers, and the closing key.
|
|
func hdf5ChunkNodeSize(rank int) int {
|
|
keySize := 8 + 8*(rank+1)
|
|
return 8 + 2*8 + 2*hdf5IStoreK*(keySize+8) + keySize
|
|
}
|
|
|
|
// hdf5ChunkSentinel is the key that closes a node: the last chunk's
|
|
// coordinates with the element size dimension set to the element size,
|
|
// which orders it past every chunk of the node, as the reference
|
|
// files store it. Reference tooling reads either shape of the
|
|
// sentinel: recent releases write the chunk shape itself with the
|
|
// element size appended, which orders identically; this writer keeps
|
|
// the fixture convention.
|
|
func hdf5ChunkSentinel(k hdf5ChunkKey, width int) hdf5ChunkKey {
|
|
coords := append([]uint64{}, k.coords...)
|
|
coords[len(coords)-1] = uint64(width)
|
|
return hdf5ChunkKey{size: 0, coords: coords}
|
|
}
|
|
|
|
// writeChunks chunks one dataset, applies the pipeline to every chunk
|
|
// and lays the chunks out through a chunk B-tree, returning the
|
|
// B-tree's address and the chunk shape. A chunk is stored complete:
|
|
// the edge chunks of a dataset whose extent is not a whole number of
|
|
// chunks are padded with the zero fill value, which is what chunked
|
|
// storage holds and what the reader requires. The staging buffers, the
|
|
// index vectors and the deflater are writer state reused across every
|
|
// chunk and every dataset of the write; each chunk fully overwrites
|
|
// what it stages, so no bytes travel between chunks.
|
|
func (w *hdf5Writer) writeChunks(s *hdf5OutSet) (uint64, []int, error) {
|
|
chunk := hdf5ChunkShape(s.shape, s.width, w.chunkTarget())
|
|
grid := make([]int, len(s.shape))
|
|
total := 1
|
|
for i := range grid {
|
|
grid[i] = (s.shape[i] + chunk[i] - 1) / chunk[i]
|
|
total *= grid[i]
|
|
}
|
|
if total > hdf5MaxChunks {
|
|
return 0, nil, base.Errf("chunking to a %d byte target needs %d chunks, past the %d the writer lays out; raise the chunk target", w.chunkTarget(), total, hdf5MaxChunks)
|
|
}
|
|
if _, err := hdf5ByteExtent(chunk, s.width, hdf5MaxDatasetBytes); err != nil {
|
|
return 0, nil, base.Errf("the chunk shape %v: %w", chunk, err)
|
|
}
|
|
level := w.opts.Gzip
|
|
if level == -1 {
|
|
level = 6
|
|
}
|
|
keys := make([]hdf5ChunkKey, 0, total)
|
|
coords := make([]int, len(grid))
|
|
// Every key aliases one range of a single coordinate slab, which
|
|
// the chunk B-tree reads before the write ends.
|
|
coordSlab := make([]uint64, total*(len(chunk)+1))
|
|
rank := len(s.shape)
|
|
chunkElems := 1
|
|
for _, c := range chunk {
|
|
chunkElems *= c
|
|
}
|
|
chunkBytes := chunkElems * s.width
|
|
if cap(w.gatherScratch) < chunkBytes {
|
|
w.gatherScratch = make([]byte, chunkBytes)
|
|
}
|
|
if w.opts.Shuffle && s.width > 1 && cap(w.shuffleScratch) < chunkBytes {
|
|
w.shuffleScratch = make([]byte, chunkBytes)
|
|
}
|
|
if cap(w.chunkIdx) < 4*rank {
|
|
w.chunkIdx = make([]int, 4*rank)
|
|
}
|
|
origin := w.chunkIdx[:rank]
|
|
count := w.chunkIdx[rank : 2*rank]
|
|
srcStride := w.chunkIdx[2*rank : 3*rank]
|
|
dstStride := w.chunkIdx[3*rank : 4*rank]
|
|
cs, ds := 1, 1
|
|
for i := rank - 1; i >= 0; i-- {
|
|
srcStride[i], dstStride[i] = cs, ds
|
|
cs *= s.shape[i]
|
|
ds *= chunk[i]
|
|
}
|
|
for ci := range total {
|
|
for d := range rank {
|
|
origin[d] = coords[d] * chunk[d]
|
|
}
|
|
data := w.gatherScratch[:chunkBytes]
|
|
if err := w.fillChunk(data, s, chunk, origin, count, srcStride, dstStride); err != nil {
|
|
// The dataset caller wraps with the dataset's own context;
|
|
// the bare encode error keeps it single.
|
|
return 0, nil, err
|
|
}
|
|
if w.opts.Shuffle && s.width > 1 {
|
|
hdf5ShuffleInto(w.shuffleScratch[:chunkBytes], data, s.width)
|
|
data = w.shuffleScratch[:chunkBytes]
|
|
}
|
|
if w.opts.Gzip != 0 {
|
|
compressed, err := w.deflateChunk(data, level)
|
|
if err != nil {
|
|
// The dataset caller wraps with the dataset's own
|
|
// context; the bare filter error keeps it single.
|
|
return 0, nil, err
|
|
}
|
|
data = compressed
|
|
}
|
|
addr := w.bytes(data)
|
|
w.pad8()
|
|
// The key's trailing coordinate is the element size dimension
|
|
// the format keys last, and the reference library carries 0 in
|
|
// it for every real chunk: only the closing sentinel holds the
|
|
// element size, which is what orders it past the node's chunks.
|
|
// A real key carrying the size would tie the sentinel and hide
|
|
// every chunk from a binary search.
|
|
key := hdf5ChunkKey{size: uint32(len(data)), coords: coordSlab[ci*(len(chunk)+1) : (ci+1)*(len(chunk)+1)], addr: addr}
|
|
for i := range chunk {
|
|
key.coords[i] = uint64(coords[i] * chunk[i])
|
|
}
|
|
keys = append(keys, key)
|
|
// The odometer walks the chunk grid in row-major order, which
|
|
// is the order the B-tree's keys must be sorted in.
|
|
for d := len(coords) - 1; d >= 0; d-- {
|
|
coords[d]++
|
|
if coords[d] < grid[d] {
|
|
break
|
|
}
|
|
coords[d] = 0
|
|
}
|
|
}
|
|
return w.writeChunkTree(keys, len(s.shape), s.width), chunk, nil
|
|
}
|
|
|
|
// fillChunk stages one chunk of the dataset into dst, which holds the
|
|
// chunk's full element extent: the cells inside the shape take the
|
|
// values encoded little-endian at their stored width, exactly the
|
|
// bytes encode produces from the same values, and the cells past any
|
|
// edge stay the zeros clear left. Every byte of dst is written on
|
|
// every call, so the reused gather buffer carries nothing between
|
|
// chunks. origin is the chunk's first cell in dataset coordinates,
|
|
// count walks the overlap of the chunk and the shape, and the strides
|
|
// are row-major element strides of the dataset and of the chunk. A
|
|
// class or width the plan never produces is the same loud refusal
|
|
// encode gives.
|
|
func (w *hdf5Writer) fillChunk(dst []byte, s *hdf5OutSet, chunk, origin, count, srcStride, dstStride []int) error {
|
|
clear(dst)
|
|
shape := s.shape
|
|
rank := len(shape)
|
|
width := s.width
|
|
for d := range rank {
|
|
count[d] = origin[d]
|
|
}
|
|
for {
|
|
src, loc := 0, 0
|
|
for d := range rank {
|
|
src += count[d] * srcStride[d]
|
|
loc += (count[d] - origin[d]) * dstStride[d]
|
|
}
|
|
p := loc * width
|
|
switch {
|
|
case s.class == 0 && width == 1 && s.signed:
|
|
dst[p] = byte(s.i8s[src])
|
|
case s.class == 0 && width == 1:
|
|
dst[p] = s.u8s[src]
|
|
case s.class == 0 && width == 2 && s.signed:
|
|
binary.LittleEndian.PutUint16(dst[p:], uint16(s.i16s[src]))
|
|
case s.class == 0 && width == 2:
|
|
binary.LittleEndian.PutUint16(dst[p:], s.u16s[src])
|
|
case s.class == 0 && width == 4 && s.signed:
|
|
binary.LittleEndian.PutUint32(dst[p:], uint32(s.i32s[src]))
|
|
case s.class == 0 && width == 4:
|
|
binary.LittleEndian.PutUint32(dst[p:], s.u32s[src])
|
|
case s.class == 0 && width == 8:
|
|
binary.LittleEndian.PutUint64(dst[p:], uint64(s.ints[src]))
|
|
case s.class == 8:
|
|
if s.bools[src] {
|
|
dst[p] = 1
|
|
} else {
|
|
dst[p] = 0
|
|
}
|
|
case s.class == 1 && width == 4:
|
|
binary.LittleEndian.PutUint32(dst[p:], math.Float32bits(s.f32s[src]))
|
|
case s.class == 1 && width == 8:
|
|
binary.LittleEndian.PutUint64(dst[p:], math.Float64bits(s.f64s[src]))
|
|
default:
|
|
return s.payloadRefusal()
|
|
}
|
|
d := rank - 1
|
|
for d >= 0 {
|
|
count[d]++
|
|
if count[d] < min(origin[d]+chunk[d], shape[d]) {
|
|
break
|
|
}
|
|
count[d] = origin[d]
|
|
d--
|
|
}
|
|
if d < 0 {
|
|
return nil
|
|
}
|
|
}
|
|
}
|
|
|
|
// chunkTarget is the configured chunk byte target, or the default.
|
|
func (w *hdf5Writer) chunkTarget() int { return hdf5ChunkTargetOf(w.opts) }
|
|
|
|
// hdf5ChunkTargetOf resolves the chunk target the options ask for.
|
|
func hdf5ChunkTargetOf(opts HDF5WriteOptions) int {
|
|
if opts.ChunkBytes == 0 {
|
|
return hdf5ChunkTarget
|
|
}
|
|
return opts.ChunkBytes
|
|
}
|
|
|
|
// hdf5ChunkShape picks a chunk shape that aims at the target bytes: a
|
|
// dataset that fits the target stays whole, otherwise the last axis is
|
|
// split first, which keeps the access the files of this package see
|
|
// (a row of a table, a trace of a signal) inside one chunk.
|
|
func hdf5ChunkShape(shape []int, width, target int) []int {
|
|
n := 1
|
|
for _, d := range shape {
|
|
n *= d
|
|
}
|
|
if n*width <= target {
|
|
return append([]int{}, shape...)
|
|
}
|
|
row := width
|
|
for _, d := range shape[:len(shape)-1] {
|
|
row *= d
|
|
}
|
|
last := max(target/row, 1)
|
|
out := append([]int{}, shape[:len(shape)-1]...)
|
|
return append(out, min(last, shape[len(shape)-1]))
|
|
}
|
|
|
|
// writeChunkTree packs the chunks into version 1 chunk B-tree nodes of
|
|
// sixty-four children and returns the root's address. An interior
|
|
// node's key for child i is the first key of child i's subtree, and a
|
|
// node's level counts its distance from the chunks.
|
|
func (w *hdf5Writer) writeChunkTree(keys []hdf5ChunkKey, rank, width int) uint64 {
|
|
kids := make([]hdf5ChunkKid, len(keys))
|
|
for i, k := range keys {
|
|
kids[i] = hdf5ChunkKid{first: k, last: k, child: k.addr}
|
|
}
|
|
level := byte(0)
|
|
for len(kids) > 2*hdf5IStoreK {
|
|
var next []hdf5ChunkKid
|
|
for start := 0; start < len(kids); start += 2 * hdf5IStoreK {
|
|
batch := kids[start:min(start+2*hdf5IStoreK, len(kids))]
|
|
addr := w.writeChunkNode(level, batch, rank, width)
|
|
next = append(next, hdf5ChunkKid{first: batch[0].first, last: batch[len(batch)-1].last, child: addr})
|
|
}
|
|
kids = next
|
|
level++
|
|
}
|
|
return w.writeChunkNode(level, kids, rank, width)
|
|
}
|
|
|
|
// writeChunkNode writes one chunk B-tree node: the first key of every
|
|
// child, a child pointer each, and the sentinel key that closes the
|
|
// node. The sibling addresses are the undefined value, as in the group
|
|
// B-tree.
|
|
func (w *hdf5Writer) writeChunkNode(level byte, kids []hdf5ChunkKid, rank, width int) uint64 {
|
|
node := make([]byte, hdf5ChunkNodeSize(rank))
|
|
copy(node, hdf5Tree)
|
|
node[4] = 1 // a chunk B-tree
|
|
node[5] = level
|
|
binary.LittleEndian.PutUint16(node[6:], uint16(len(kids)))
|
|
binary.LittleEndian.PutUint64(node[8:], math.MaxUint64) // left sibling
|
|
binary.LittleEndian.PutUint64(node[16:], math.MaxUint64) // right sibling
|
|
p := 24
|
|
writeKey := func(k hdf5ChunkKey) {
|
|
binary.LittleEndian.PutUint32(node[p:], k.size)
|
|
binary.LittleEndian.PutUint32(node[p+4:], 0) // the filter mask: every filter ran
|
|
for i, c := range k.coords {
|
|
binary.LittleEndian.PutUint64(node[p+8+8*i:], c)
|
|
}
|
|
p += 8 + 8*len(k.coords)
|
|
}
|
|
for _, k := range kids {
|
|
writeKey(k.first)
|
|
binary.LittleEndian.PutUint64(node[p:], k.child)
|
|
p += 8
|
|
}
|
|
writeKey(hdf5ChunkSentinel(kids[len(kids)-1].last, width))
|
|
return w.bytes(node)
|
|
}
|
|
|
|
// filterEntries lists the configured pipeline in application order for
|
|
// the filter pipeline message: shuffle before deflate, the order that
|
|
// lays each element's high-order bytes together for the compressor.
|
|
// The reference stores the element width as the shuffle's one client
|
|
// value and the level as deflate's.
|
|
func (w *hdf5Writer) filterEntries(width int) []hdf5OutFilter {
|
|
var out []hdf5OutFilter
|
|
if w.opts.Shuffle {
|
|
out = append(out, hdf5OutFilter{id: hdf5FilterShuffle, name: "shuffle", values: []uint32{uint32(width)}})
|
|
}
|
|
if w.opts.Gzip != 0 {
|
|
level := w.opts.Gzip
|
|
if level == -1 {
|
|
level = 6
|
|
}
|
|
out = append(out, hdf5OutFilter{id: hdf5FilterDeflate, name: "deflate", values: []uint32{uint32(level)}})
|
|
}
|
|
return out
|
|
}
|
|
|
|
// hdf5ShuffleInto transposes the element bytes of data into out, which
|
|
// must hold the same length: the first block takes every element's
|
|
// first byte, the next every second byte. Every byte of out is written.
|
|
func hdf5ShuffleInto(out, data []byte, width int) {
|
|
n := len(data) / width
|
|
for i := range width {
|
|
for j := range n {
|
|
out[i*n+j] = data[j*width+i]
|
|
}
|
|
}
|
|
}
|
|
|
|
// deflateChunk compresses one staged chunk into the zlib stream the
|
|
// deflate filter stores: a zlib header in front of the raw deflate
|
|
// stream, which is what the reader's inflate expects. The compressor
|
|
// and its output buffer are reused across the chunks of a write;
|
|
// Reset leaves the compressor in the state a fresh writer holds, so
|
|
// the stream carries the same bytes a per-chunk writer produced. The
|
|
// returned bytes are the buffer's, valid until the next call.
|
|
func (w *hdf5Writer) deflateChunk(data []byte, level int) ([]byte, error) {
|
|
if w.comp == nil || w.compLevel != level {
|
|
zw, err := zlib.NewWriterLevel(&w.compBuf, level)
|
|
if err != nil {
|
|
return nil, base.Errf("the deflate level %d is refused: %w", level, err)
|
|
}
|
|
w.comp, w.compLevel = zw, level
|
|
} else {
|
|
w.comp.Reset(&w.compBuf)
|
|
}
|
|
w.compBuf.Reset()
|
|
if _, err := w.comp.Write(data); err != nil {
|
|
return nil, base.Errf("deflate: %w", err)
|
|
}
|
|
if err := w.comp.Close(); err != nil {
|
|
return nil, base.Errf("deflate: %w", err)
|
|
}
|
|
return w.compBuf.Bytes(), nil
|
|
}
|