// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package io // Group and chunked-storage writing for the HDF5 writer. Both group // layouts the reader accepts are written: the classic symbol table (a // local heap of names, symbol table nodes and a version 1 group // B-tree) and the latest-style compact group (link messages in the // object header). Chunked storage hangs its chunks off a version 1 // chunk B-tree, whose conventions the reference files pin: every node // is allocated at the size the format's fanout implies, an interior // node's key is the first key of the child's subtree, and a node's // closing key is a sentinel ordered past its last chunk. import ( "bytes" "compress/zlib" "encoding/binary" "math" "slices" "strings" "sourcedock.dev/petrbalvin/tensor/internal/base" ) // hdf5GroupKid is one child pointer of a group B-tree node. The key of // kid i is the heap offset of the last name child i-1 holds, so a // node's entries are headed by the zero key (the heap's null name, // below every real name) and each entry's key closes the range of the // child before it, which is the convention the fixtures store. type hdf5GroupKid struct { key uint64 child uint64 } // hdf5GroupChild is one child entry of a group: its link name, its // object header address and, until the children are written, the // subgroup or dataset whose address it carries. type hdf5GroupChild struct { name string addr uint64 group *hdf5OutNode set *hdf5OutSet } // hdf5Kids merges a group's subgroups and datasets into one list in // name order, the order the classic symbol table requires: the local // heap assigns its offsets in list order and the reference library // binary-searches both the B-tree keys and the symbol node records on // those offsets, so a groups-first order would hide every child whose // name interleaves with a dataset's. The latest layout's link list // keeps the same order. func hdf5Kids(g *hdf5OutNode) []hdf5GroupChild { kids := make([]hdf5GroupChild, 0, len(g.groups)+len(g.sets)) for _, sub := range g.groups { kids = append(kids, hdf5GroupChild{name: sub.name, group: sub}) } for _, s := range g.sets { kids = append(kids, hdf5GroupChild{name: baseName(s.path), set: s}) } slices.SortFunc(kids, func(a, b hdf5GroupChild) int { return strings.Compare(a.name, b.name) }) return kids } // kidAddr is a child's written object header address. func kidAddr(k hdf5GroupChild) uint64 { if k.group != nil { return k.group.addr } return k.set.addr } // writeClassicGroup writes a group in the classic layout, in the // construction order the reference library uses: the object header is // allocated first (over a placeholder, since the symbol table message // can only name the B-tree and heap once they exist), then the local // heap, then the children, and the symbol table nodes and B-tree last. // A child group's symbol entry carries the B-tree and heap addresses // in its cache, as the reference writes; a dataset's carries none. func (w *hdf5Writer) writeClassicGroup(g *hdf5OutNode) error { kids := hdf5Kids(g) // The header placeholder: sized and shaped exactly like the final // image, with the B-tree and heap addresses still zero. msgs, err := w.classicGroupMsgs(g, 0, 0) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } image, err := hdf5HeaderV1(msgs) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } hdrAddr := w.bytes(image) // The heap assigns every name its offset; the children are in // sorted order, so offset order and name order agree and the // B-tree's byte-offset keys sort as the names do. The heap carries // slack behind the names, written as the free block the reference // library leaves there: a next offset of one is the sentinel that // ends the free list. heapData := make([]byte, 8) // offset 0 is the null name no entry points at offsets := make([]uint64, len(kids)) for i, k := range kids { offsets[i] = uint64(len(heapData)) heapData = append(heapData, k.name...) heapData = hdf5AppendAlign(append(heapData, 0)) } size := max(len(heapData)+16, 88) free := len(heapData) heapData = append(heapData, make([]byte, 16)...) // the free block: next, size binary.LittleEndian.PutUint64(heapData[free:], 1) binary.LittleEndian.PutUint64(heapData[free+8:], uint64(size-free)) heapData = append(heapData, make([]byte, size-len(heapData))...) dataAddr := w.bytes(heapData) w.pad8() heap := w.writeLocalHeap(dataAddr, size, free) // The children: datasets, then the subgroups, each of which lays // out its own subtree in the children's name order. for _, s := range g.sets { addr, err := w.writeDataset(s) if err != nil { // writeDataset names the dataset itself; a second wrap // here would prefix it twice. return err } s.addr = addr } for _, sub := range g.groups { if err := w.writeGroup(sub); err != nil { return err } } for i := range kids { kids[i].addr = kidAddr(kids[i]) } // Symbol table nodes, eight entries each: the node occupies the // size the format's leaf K of four implies, however few entries are // used, and an empty group still holds one. leaves := make([]hdf5GroupKid, 0, max((len(kids)+2*hdf5GroupLeafK-1)/(2*hdf5GroupLeafK), 1)) for start := 0; start < len(kids); start += 2 * hdf5GroupLeafK { leaves = append(leaves, w.writeSymbolNode(kids[start:min(start+2*hdf5GroupLeafK, len(kids))], offsets[start:])) } if len(leaves) == 0 { leaves = append(leaves, w.writeSymbolNode(nil, nil)) } btree := w.writeGroupTree(leaves) // The header over the placeholder, now that the B-tree and heap // have their final addresses. msgs, err = w.classicGroupMsgs(g, btree, heap) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } image, err = hdf5HeaderV1(msgs) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } w.headerAt(hdrAddr, image) g.addr, g.btree, g.heap = hdrAddr, btree, heap return nil } // classicGroupMsgs builds the message list of a classic group's object // header: the symbol table message naming the B-tree and heap, then // the attributes in name order. func (w *hdf5Writer) classicGroupMsgs(g *hdf5OutNode, btree, heap uint64) ([]hdf5OutMsg, error) { msgs := []hdf5OutMsg{{typ: hdf5MsgSymbolTable, data: w.appendAddrs(nil, btree, heap)}} for _, a := range g.attrs { m, err := hdf5AttrMessage(a) if err != nil { return nil, err } msgs = append(msgs, m) } return msgs, nil } // writeSymbolNode writes one symbol table node of the given entries, // allocated at eight slots. A child group's cache holds its B-tree and // heap addresses; a dataset's cache stays empty. func (w *hdf5Writer) writeSymbolNode(batch []hdf5GroupChild, offsets []uint64) hdf5GroupKid { node := make([]byte, 8+2*hdf5GroupLeafK*(8+8+24)) copy(node, hdf5SymbolNode) node[4] = 1 binary.LittleEndian.PutUint16(node[6:], uint16(len(batch))) p := 8 for i, k := range batch { binary.LittleEndian.PutUint64(node[p:], offsets[i]) binary.LittleEndian.PutUint64(node[p+8:], k.addr) if k.group != nil { binary.LittleEndian.PutUint32(node[p+16:], 1) binary.LittleEndian.PutUint64(node[p+24:], k.group.btree) binary.LittleEndian.PutUint64(node[p+32:], k.group.heap) } p += 8 + 8 + 24 } leaf := hdf5GroupKid{child: w.bytes(node)} if len(batch) > 0 { leaf.key = offsets[len(batch)-1] } return leaf } // baseName returns the last segment of an absolute path. func baseName(path string) string { for i := len(path) - 1; i >= 0; i-- { if path[i] == '/' { return path[i+1:] } } return path } // writeNewGroup writes a group in the latest layout, the construction // order the reference library uses: the object header is allocated // first over a placeholder, then the children, and the header is // written over the placeholder once every link's address is known. // The link info and group info messages the reference writes head the // message list; a group without links carries no link info message: // beside no links the reader of this package takes that message for // dense storage, which it refuses. func (w *hdf5Writer) writeNewGroup(g *hdf5OutNode) error { msgs, err := w.newGroupMsgs(g) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } image, err := hdf5HeaderV2(msgs) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } hdrAddr := w.bytes(image) for _, s := range g.sets { addr, err := w.writeDataset(s) if err != nil { // writeDataset names the dataset itself; a second wrap // here would prefix it twice. return err } s.addr = addr } for _, sub := range g.groups { if err := w.writeGroup(sub); err != nil { return err } } msgs, err = w.newGroupMsgs(g) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } image, err = hdf5HeaderV2(msgs) if err != nil { return base.Errf("SaveHDF5: group %q: %w", g.path, err) } w.headerAt(hdrAddr, image) g.addr = hdrAddr return nil } // newGroupMsgs builds the message list of a latest-style group's // object header. Addresses unknown at placeholder time are zero; the // message sizes are the same either way. func (w *hdf5Writer) newGroupMsgs(g *hdf5OutNode) ([]hdf5OutMsg, error) { var msgs []hdf5OutMsg if len(g.groups)+len(g.sets) > 0 { undef := bytes.Repeat([]byte{0xff}, 8) info := append([]byte{0, 0}, undef...) info = append(info, undef...) msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgLinkInfo, data: info}, hdf5OutMsg{typ: hdf5MsgGroupInfo, data: []byte{0, 0}}, ) } addLink := func(name string, addr uint64) { body := []byte{1, 0} // version 1, no creation order, one-byte length if len(name) >= 256 { body[1] = 0x01 // the name length widens to two bytes body = binary.LittleEndian.AppendUint16(body, uint16(len(name))) } else { body = append(body, byte(len(name))) } body = append(body, name...) body = binary.LittleEndian.AppendUint64(body, addr) msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgLink, data: body}) } for _, k := range hdf5Kids(g) { addLink(k.name, kidAddr(k)) } for _, a := range g.attrs { m, err := hdf5AttrMessage(a) if err != nil { // Unwrapped: the caller carries the group context and // prefixes the entry point, so wrapping here would double // both. return nil, err } msgs = append(msgs, m) } return msgs, nil } // writeLocalHeap writes a local heap header in front of the data // segment written earlier: the data segment's size, the free list head // and the segment address. func (w *hdf5Writer) writeLocalHeap(dataAddr uint64, size, free int) uint64 { head := append([]byte{}, hdf5LocalHeap...) head = append(head, 0, 0, 0, 0) // version and reserved head = binary.LittleEndian.AppendUint64(head, uint64(size)) head = binary.LittleEndian.AppendUint64(head, uint64(free)) head = binary.LittleEndian.AppendUint64(head, dataAddr) return w.bytes(head) } // writeGroupTree packs the symbol table nodes into version 1 group // B-tree nodes of thirty-two children and returns the root's address. func (w *hdf5Writer) writeGroupTree(leaves []hdf5GroupKid) uint64 { kids := leaves level := byte(0) for len(kids) > 2*hdf5GroupInnerK { var next []hdf5GroupKid for start := 0; start < len(kids); start += 2 * hdf5GroupInnerK { batch := kids[start:min(start+2*hdf5GroupInnerK, len(kids))] addr := w.writeGroupNode(level, batch) next = append(next, hdf5GroupKid{key: batch[len(batch)-1].key, child: addr}) } kids = next level++ } return w.writeGroupNode(level, kids) } // writeGroupNode writes one group B-tree node, allocated at the size // the format's internal K of sixteen implies. The sibling addresses // are the undefined value: a zero would be a defined address, and a // reader would follow it as a sibling node. func (w *hdf5Writer) writeGroupNode(level byte, kids []hdf5GroupKid) uint64 { node := make([]byte, 8+2*8+2*hdf5GroupInnerK*(8+8)+8) copy(node, hdf5Tree) node[4] = 0 // a group B-tree node[5] = level binary.LittleEndian.PutUint16(node[6:], uint16(len(kids))) binary.LittleEndian.PutUint64(node[8:], math.MaxUint64) // left sibling binary.LittleEndian.PutUint64(node[16:], math.MaxUint64) // right sibling p := 24 binary.LittleEndian.PutUint64(node[p:], 0) // key[0]: the null name p += 8 for _, k := range kids { binary.LittleEndian.PutUint64(node[p:], k.child) p += 8 binary.LittleEndian.PutUint64(node[p:], k.key) // the key closing this child p += 8 } return w.bytes(node) } // appendAddrs appends the given addresses as the file's eight-byte // address fields. func (w *hdf5Writer) appendAddrs(b []byte, addrs ...uint64) []byte { for _, a := range addrs { b = binary.LittleEndian.AppendUint64(b, a) } return b } // hdf5ChunkKey is a version 1 chunk B-tree key: the chunk's stored // size, its coordinates, one per dataset dimension plus the element // size dimension the format keys last, and the address the chunk's // bytes occupy. type hdf5ChunkKey struct { size uint32 coords []uint64 addr uint64 } // hdf5ChunkKid is one child of a chunk B-tree node with the first and // last keys of its subtree, which become the node's own keys. type hdf5ChunkKid struct { first, last hdf5ChunkKey child uint64 } // hdf5ChunkNodeSize is the size a chunk B-tree node occupies: the // header, twice the storage K of thirty-two keys with their child // pointers, and the closing key. func hdf5ChunkNodeSize(rank int) int { keySize := 8 + 8*(rank+1) return 8 + 2*8 + 2*hdf5IStoreK*(keySize+8) + keySize } // hdf5ChunkSentinel is the key that closes a node: the last chunk's // coordinates with the element size dimension set to the element size, // which orders it past every chunk of the node, as the reference // files store it. Reference tooling reads either shape of the // sentinel: recent releases write the chunk shape itself with the // element size appended, which orders identically; this writer keeps // the fixture convention. func hdf5ChunkSentinel(k hdf5ChunkKey, width int) hdf5ChunkKey { coords := append([]uint64{}, k.coords...) coords[len(coords)-1] = uint64(width) return hdf5ChunkKey{size: 0, coords: coords} } // writeChunks chunks one dataset, applies the pipeline to every chunk // and lays the chunks out through a chunk B-tree, returning the // B-tree's address and the chunk shape. A chunk is stored complete: // the edge chunks of a dataset whose extent is not a whole number of // chunks are padded with the zero fill value, which is what chunked // storage holds and what the reader requires. The staging buffers, the // index vectors and the deflater are writer state reused across every // chunk and every dataset of the write; each chunk fully overwrites // what it stages, so no bytes travel between chunks. func (w *hdf5Writer) writeChunks(s *hdf5OutSet) (uint64, []int, error) { chunk := hdf5ChunkShape(s.shape, s.width, w.chunkTarget()) grid := make([]int, len(s.shape)) total := 1 for i := range grid { grid[i] = (s.shape[i] + chunk[i] - 1) / chunk[i] total *= grid[i] } if total > hdf5MaxChunks { return 0, nil, base.Errf("chunking to a %d byte target needs %d chunks, past the %d the writer lays out; raise the chunk target", w.chunkTarget(), total, hdf5MaxChunks) } if _, err := hdf5ByteExtent(chunk, s.width, hdf5MaxDatasetBytes); err != nil { return 0, nil, base.Errf("the chunk shape %v: %w", chunk, err) } level := w.opts.Gzip if level == -1 { level = 6 } keys := make([]hdf5ChunkKey, 0, total) coords := make([]int, len(grid)) // Every key aliases one range of a single coordinate slab, which // the chunk B-tree reads before the write ends. coordSlab := make([]uint64, total*(len(chunk)+1)) rank := len(s.shape) chunkElems := 1 for _, c := range chunk { chunkElems *= c } chunkBytes := chunkElems * s.width if cap(w.gatherScratch) < chunkBytes { w.gatherScratch = make([]byte, chunkBytes) } if w.opts.Shuffle && s.width > 1 && cap(w.shuffleScratch) < chunkBytes { w.shuffleScratch = make([]byte, chunkBytes) } if cap(w.chunkIdx) < 4*rank { w.chunkIdx = make([]int, 4*rank) } origin := w.chunkIdx[:rank] count := w.chunkIdx[rank : 2*rank] srcStride := w.chunkIdx[2*rank : 3*rank] dstStride := w.chunkIdx[3*rank : 4*rank] cs, ds := 1, 1 for i := rank - 1; i >= 0; i-- { srcStride[i], dstStride[i] = cs, ds cs *= s.shape[i] ds *= chunk[i] } for ci := range total { for d := range rank { origin[d] = coords[d] * chunk[d] } data := w.gatherScratch[:chunkBytes] if err := w.fillChunk(data, s, chunk, origin, count, srcStride, dstStride); err != nil { // The dataset caller wraps with the dataset's own context; // the bare encode error keeps it single. return 0, nil, err } if w.opts.Shuffle && s.width > 1 { hdf5ShuffleInto(w.shuffleScratch[:chunkBytes], data, s.width) data = w.shuffleScratch[:chunkBytes] } if w.opts.Gzip != 0 { compressed, err := w.deflateChunk(data, level) if err != nil { // The dataset caller wraps with the dataset's own // context; the bare filter error keeps it single. return 0, nil, err } data = compressed } addr := w.bytes(data) w.pad8() // The key's trailing coordinate is the element size dimension // the format keys last, and the reference library carries 0 in // it for every real chunk: only the closing sentinel holds the // element size, which is what orders it past the node's chunks. // A real key carrying the size would tie the sentinel and hide // every chunk from a binary search. key := hdf5ChunkKey{size: uint32(len(data)), coords: coordSlab[ci*(len(chunk)+1) : (ci+1)*(len(chunk)+1)], addr: addr} for i := range chunk { key.coords[i] = uint64(coords[i] * chunk[i]) } keys = append(keys, key) // The odometer walks the chunk grid in row-major order, which // is the order the B-tree's keys must be sorted in. for d := len(coords) - 1; d >= 0; d-- { coords[d]++ if coords[d] < grid[d] { break } coords[d] = 0 } } return w.writeChunkTree(keys, len(s.shape), s.width), chunk, nil } // fillChunk stages one chunk of the dataset into dst, which holds the // chunk's full element extent: the cells inside the shape take the // values encoded little-endian at their stored width, exactly the // bytes encode produces from the same values, and the cells past any // edge stay the zeros clear left. Every byte of dst is written on // every call, so the reused gather buffer carries nothing between // chunks. origin is the chunk's first cell in dataset coordinates, // count walks the overlap of the chunk and the shape, and the strides // are row-major element strides of the dataset and of the chunk. A // class or width the plan never produces is the same loud refusal // encode gives. func (w *hdf5Writer) fillChunk(dst []byte, s *hdf5OutSet, chunk, origin, count, srcStride, dstStride []int) error { clear(dst) shape := s.shape rank := len(shape) width := s.width for d := range rank { count[d] = origin[d] } for { src, loc := 0, 0 for d := range rank { src += count[d] * srcStride[d] loc += (count[d] - origin[d]) * dstStride[d] } p := loc * width switch { case s.class == 0 && width == 1 && s.signed: dst[p] = byte(s.i8s[src]) case s.class == 0 && width == 1: dst[p] = s.u8s[src] case s.class == 0 && width == 2 && s.signed: binary.LittleEndian.PutUint16(dst[p:], uint16(s.i16s[src])) case s.class == 0 && width == 2: binary.LittleEndian.PutUint16(dst[p:], s.u16s[src]) case s.class == 0 && width == 4 && s.signed: binary.LittleEndian.PutUint32(dst[p:], uint32(s.i32s[src])) case s.class == 0 && width == 4: binary.LittleEndian.PutUint32(dst[p:], s.u32s[src]) case s.class == 0 && width == 8: binary.LittleEndian.PutUint64(dst[p:], uint64(s.ints[src])) case s.class == 8: if s.bools[src] { dst[p] = 1 } else { dst[p] = 0 } case s.class == 1 && width == 4: binary.LittleEndian.PutUint32(dst[p:], math.Float32bits(s.f32s[src])) case s.class == 1 && width == 8: binary.LittleEndian.PutUint64(dst[p:], math.Float64bits(s.f64s[src])) default: return s.payloadRefusal() } d := rank - 1 for d >= 0 { count[d]++ if count[d] < min(origin[d]+chunk[d], shape[d]) { break } count[d] = origin[d] d-- } if d < 0 { return nil } } } // chunkTarget is the configured chunk byte target, or the default. func (w *hdf5Writer) chunkTarget() int { return hdf5ChunkTargetOf(w.opts) } // hdf5ChunkTargetOf resolves the chunk target the options ask for. func hdf5ChunkTargetOf(opts HDF5WriteOptions) int { if opts.ChunkBytes == 0 { return hdf5ChunkTarget } return opts.ChunkBytes } // hdf5ChunkShape picks a chunk shape that aims at the target bytes: a // dataset that fits the target stays whole, otherwise the last axis is // split first, which keeps the access the files of this package see // (a row of a table, a trace of a signal) inside one chunk. func hdf5ChunkShape(shape []int, width, target int) []int { n := 1 for _, d := range shape { n *= d } if n*width <= target { return append([]int{}, shape...) } row := width for _, d := range shape[:len(shape)-1] { row *= d } last := max(target/row, 1) out := append([]int{}, shape[:len(shape)-1]...) return append(out, min(last, shape[len(shape)-1])) } // writeChunkTree packs the chunks into version 1 chunk B-tree nodes of // sixty-four children and returns the root's address. An interior // node's key for child i is the first key of child i's subtree, and a // node's level counts its distance from the chunks. func (w *hdf5Writer) writeChunkTree(keys []hdf5ChunkKey, rank, width int) uint64 { kids := make([]hdf5ChunkKid, len(keys)) for i, k := range keys { kids[i] = hdf5ChunkKid{first: k, last: k, child: k.addr} } level := byte(0) for len(kids) > 2*hdf5IStoreK { var next []hdf5ChunkKid for start := 0; start < len(kids); start += 2 * hdf5IStoreK { batch := kids[start:min(start+2*hdf5IStoreK, len(kids))] addr := w.writeChunkNode(level, batch, rank, width) next = append(next, hdf5ChunkKid{first: batch[0].first, last: batch[len(batch)-1].last, child: addr}) } kids = next level++ } return w.writeChunkNode(level, kids, rank, width) } // writeChunkNode writes one chunk B-tree node: the first key of every // child, a child pointer each, and the sentinel key that closes the // node. The sibling addresses are the undefined value, as in the group // B-tree. func (w *hdf5Writer) writeChunkNode(level byte, kids []hdf5ChunkKid, rank, width int) uint64 { node := make([]byte, hdf5ChunkNodeSize(rank)) copy(node, hdf5Tree) node[4] = 1 // a chunk B-tree node[5] = level binary.LittleEndian.PutUint16(node[6:], uint16(len(kids))) binary.LittleEndian.PutUint64(node[8:], math.MaxUint64) // left sibling binary.LittleEndian.PutUint64(node[16:], math.MaxUint64) // right sibling p := 24 writeKey := func(k hdf5ChunkKey) { binary.LittleEndian.PutUint32(node[p:], k.size) binary.LittleEndian.PutUint32(node[p+4:], 0) // the filter mask: every filter ran for i, c := range k.coords { binary.LittleEndian.PutUint64(node[p+8+8*i:], c) } p += 8 + 8*len(k.coords) } for _, k := range kids { writeKey(k.first) binary.LittleEndian.PutUint64(node[p:], k.child) p += 8 } writeKey(hdf5ChunkSentinel(kids[len(kids)-1].last, width)) return w.bytes(node) } // filterEntries lists the configured pipeline in application order for // the filter pipeline message: shuffle before deflate, the order that // lays each element's high-order bytes together for the compressor. // The reference stores the element width as the shuffle's one client // value and the level as deflate's. func (w *hdf5Writer) filterEntries(width int) []hdf5OutFilter { var out []hdf5OutFilter if w.opts.Shuffle { out = append(out, hdf5OutFilter{id: hdf5FilterShuffle, name: "shuffle", values: []uint32{uint32(width)}}) } if w.opts.Gzip != 0 { level := w.opts.Gzip if level == -1 { level = 6 } out = append(out, hdf5OutFilter{id: hdf5FilterDeflate, name: "deflate", values: []uint32{uint32(level)}}) } return out } // hdf5ShuffleInto transposes the element bytes of data into out, which // must hold the same length: the first block takes every element's // first byte, the next every second byte. Every byte of out is written. func hdf5ShuffleInto(out, data []byte, width int) { n := len(data) / width for i := range width { for j := range n { out[i*n+j] = data[j*width+i] } } } // deflateChunk compresses one staged chunk into the zlib stream the // deflate filter stores: a zlib header in front of the raw deflate // stream, which is what the reader's inflate expects. The compressor // and its output buffer are reused across the chunks of a write; // Reset leaves the compressor in the state a fresh writer holds, so // the stream carries the same bytes a per-chunk writer produced. The // returned bytes are the buffer's, valid until the next call. func (w *hdf5Writer) deflateChunk(data []byte, level int) ([]byte, error) { if w.comp == nil || w.compLevel != level { zw, err := zlib.NewWriterLevel(&w.compBuf, level) if err != nil { return nil, base.Errf("the deflate level %d is refused: %w", level, err) } w.comp, w.compLevel = zw, level } else { w.comp.Reset(&w.compBuf) } w.compBuf.Reset() if _, err := w.comp.Write(data); err != nil { return nil, base.Errf("deflate: %w", err) } if err := w.comp.Close(); err != nil { return nil, base.Errf("deflate: %w", err) } return w.compBuf.Bytes(), nil }