Files
tensor/io/hdf5.go
T
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

2304 lines
81 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package io
import (
"bytes"
"compress/zlib"
"encoding/binary"
"io"
"maps"
"math"
"math/bits"
"os"
"slices"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
// HDF5 (HDF5 1.8/1.10 file format), read-only, for the numeric arrays
// scientific files carry: every dataset of a file comes back as an
// Array with its path, shape and attributes.
//
// Supported: superblocks 0 to 3 (the classic layout and the "latest"
// library version, whose checksums are the lookup3 sum and are
// verified), object headers versions 1 and 2, groups stored as symbol
// tables or as link messages, datasets stored contiguously, compactly
// or in chunks through a version 1 B-tree, the deflate, shuffle and
// fletcher32 filters, fixed-point and floating-point datatypes of the
// usual widths, the boolean enumeration convention, and attributes in
// the object header. Fixed-point datasets land the core dtype their
// stored width and signedness declare; floating-point datasets land
// float64 at width 8 and float32 at width 4.
//
// Refused with an error naming what is missing, never guessed at: the
// superblock extension, the fractal-heap link storage of dense groups,
// the version 2 B-tree a latest-version file uses to index chunks,
// variable-length, compound, array and every non-boolean enumeration
// datatype, bit fields, string datasets, and the szip, nbit and
// scale-offset filters.
// Superblock and object header signatures.
var (
hdf5Magic = []byte{0x89, 'H', 'D', 'F', '\r', '\n', 0x1a, '\n'}
hdf5Tree = []byte{'T', 'R', 'E', 'E'}
hdf5SymbolNode = []byte{'S', 'N', 'O', 'D'}
hdf5LocalHeap = []byte{'H', 'E', 'A', 'P'}
hdf5ObjHdr2 = []byte{'O', 'H', 'D', 'R'}
hdf5Chunk2 = []byte{'O', 'C', 'H', 'K'}
)
// hdf5MaxDatasetBytes bounds the raw byte extent of one dataset and of
// one chunk. The reader materialises whole datasets, so a declared size
// beyond this budget is a hostile header, not data: without the cap a
// 16-byte file could order a terabyte-scale allocation through the
// dataspace, the fill-value path or a single chunk declaration. Honest
// datasets above the budget belong behind mmap, not behind an eager
// read.
const hdf5MaxDatasetBytes = 2 << 30 // 2 GiB
// hdf5ByteExtent multiplies declared extents by an element width into a
// byte count, bounding every factor against budget *before* it is
// multiplied. Multiplying first and comparing the product afterwards is
// the defect the review found: a dataspace of 2^32 by 2^32 float64
// elements is 2^64 bytes, which wraps to zero, and a cap that runs after
// the multiplication therefore accepts a declaration the file cannot
// back, while a chunk of 2^32 by 2^29 elements looks empty the same way.
// Every partial product here stays inside budget, so no caller can be
// handed a wrapped count either. A zero extent is legal and yields zero
// bytes (an empty dataset); a negative one is a hostile header, as is a
// non-positive element width.
func hdf5ByteExtent(dims []int, width int, budget uint64) (uint64, error) {
total := uint64(1)
empty := false
for _, d := range dims {
if d < 0 {
return 0, base.Errf("a declared extent of %d", d)
}
if d == 0 {
empty = true
continue
}
if uint64(d) > budget/total {
return 0, base.Errf("the declared extents multiply past the %d byte budget", budget)
}
total *= uint64(d)
}
if empty {
return 0, nil
}
if width <= 0 {
return 0, base.Errf("a declared element width of %d bytes", width)
}
if total > budget/uint64(width) {
return 0, base.Errf("the declared extents multiply past the %d byte budget", budget)
}
return total * uint64(width), nil
}
// HDF5 dataset object header message types.
const (
hdf5MsgDataspace = 1
hdf5MsgLinkInfo = 2
hdf5MsgDatatype = 3
hdf5MsgFillValueOld = 4
hdf5MsgFillValue = 5
hdf5MsgLink = 6
hdf5MsgDataLayout = 8
hdf5MsgGroupInfo = 10
hdf5MsgFilterPipeline = 11
hdf5MsgAttribute = 12
hdf5MsgContinuation = 16
hdf5MsgSymbolTable = 17
)
// HDF5 filter identifiers.
const (
hdf5FilterDeflate = 1
hdf5FilterShuffle = 2
hdf5FilterFletcher32 = 3
)
// HDF5Dataset is one dataset of a file: its path from the root, its
// shape (row-major, as the file stores it), its values and its
// attributes. The attribute map also carries the attributes of the
// groups the dataset sits in, the nearest one winning, because that is
// where files usually put the units and titles that apply to a whole
// group.
type HDF5Dataset struct {
Path string
Shape []int
Values *core.Array
Attrs map[string]string
}
// LoadHDF5 reads every dataset of an HDF5 file, in path order, as
// numeric arrays. The values keep the file's own dtype where the core
// has one: fixed-point data lands by stored width and signedness
// (int8, uint8, int16, uint16, int32, uint32, and int64 as Int), the
// boolean enumeration convention lands Bool, and floating-point data
// lands float64 at width 8 and float32 at width 4. Unsigned 64-bit
// data is refused, because no core dtype holds every value of it.
//
// An object hard-linked under several names is read once: its datasets
// appear under the first path the traversal reaches, the links of a
// group being visited in their file order, and never under the later
// ones. A link that closes a cycle along the path to it is an error,
// because no linear listing of the file exists.
func LoadHDF5(path string) ([]HDF5Dataset, error) {
data, err := os.ReadFile(path)
if err != nil {
return nil, base.Errf("LoadHDF5: %w", err)
}
f, err := newHDF5File(data)
if err != nil {
return nil, err
}
out := []HDF5Dataset{}
if err := f.walk(f.rootAddress, newHDF5WalkState(), &out); err != nil {
return nil, err
}
slices.SortFunc(out, func(x, y HDF5Dataset) int { return strings.Compare(x.Path, y.Path) })
return out, nil
}
// hdf5File is a parsed superblock plus the raw bytes of the file every
// offset is resolved against.
type hdf5File struct {
data []byte
superblock byte
offSize int
lenSize int
rootAddress uint64
groupLeafK int
groupInnerK int
}
func newHDF5File(data []byte) (*hdf5File, error) {
const name = "LoadHDF5"
if len(data) < 96 || !bytes.Equal(data[:8], hdf5Magic) {
return nil, base.Errf("%s: not an HDF5 file, the signature is % x", name, data[:min(8, len(data))])
}
version := data[8]
f := &hdf5File{data: data, superblock: version}
switch version {
case 0, 1:
// The classic layout: the sizes and the root group entry follow
// the fixed header.
case 2, 3:
// The modern layout: the sizes follow the version byte, the
// four addresses follow without the group K values, and the
// root group is named by its object header address. The whole
// superblock carries a lookup3 checksum.
f.offSize = int(data[9])
f.lenSize = int(data[10])
switch f.offSize {
case 4, 8:
default:
return nil, base.Errf("%s: %d-byte addresses are not supported", name, f.offSize)
}
if f.lenSize != f.offSize && f.lenSize != 8 && f.lenSize != 4 {
return nil, base.Errf("%s: %d-byte lengths are not supported", name, f.lenSize)
}
stored := 12 + 4*f.offSize + 4
if len(data) < stored {
return nil, base.Errf("%s: superblock version %d is truncated", name, version)
}
// Every address the format carries is absolute, so a non-zero
// base would need base-relative resolution, which is not
// implemented; the extension holds the file space strategy and
// driver settings, which the reader does not interpret. Both
// are refused by name rather than read wrong.
if baseAddr := f.u64At(12, f.offSize); baseAddr != 0 {
return nil, base.Errf("%s: a non-zero base address of %d is not supported", name, baseAddr)
}
if ext := f.u64At(12+f.offSize, f.offSize); ext != math.MaxUint64 {
return nil, base.Errf("%s: the superblock extension at %d is not supported; rewrite the file with the default library version", name, ext)
}
if got, want := binary.LittleEndian.Uint32(data[12+4*f.offSize:]), hdf5Lookup3(data[:12+4*f.offSize]); got != want {
return nil, base.Errf("%s: the superblock fails its checksum (%#08x, want %#08x)", name, got, want)
}
f.rootAddress = f.u64At(12+3*f.offSize, f.offSize)
if f.rootAddress == math.MaxUint64 {
return nil, base.Errf("%s: the root group has no object header", name)
}
return f, nil
default:
return nil, base.Errf("%s: superblock version %d is not supported; rewrite the file with the default version", name, version)
}
f.offSize = int(data[13])
f.lenSize = int(data[14])
switch f.offSize {
case 4, 8:
default:
return nil, base.Errf("%s: %d-byte addresses are not supported", name, f.offSize)
}
if f.lenSize != f.offSize && f.lenSize != 8 && f.lenSize != 4 {
return nil, base.Errf("%s: %d-byte lengths are not supported", name, f.lenSize)
}
f.groupLeafK = int(binary.LittleEndian.Uint16(data[16:]))
f.groupInnerK = int(binary.LittleEndian.Uint16(data[18:]))
pos := 24
if version == 1 {
// Version 1 carries the indexed-storage K before the addresses.
pos = 28
}
// base address (every file this reader accepts carries 0 there:
// a non-zero base would require base-relative offset resolution,
// which is not implemented), free space information, end of file,
// driver information: four addresses, then the root group's symbol
// table entry, whose first field is the link name offset and is
// sized by the length size, not the address size.
pos += 4 * f.offSize
f.rootAddress = f.u64At(pos+f.lenSize, f.offSize)
if f.rootAddress == math.MaxUint64 {
return nil, base.Errf("%s: the root group has no object header", name)
}
return f, nil
}
// u64At reads an address or length of size bytes at an absolute file
// offset; every offset the reader resolves is absolute.
func (f *hdf5File) u64At(off int, size int) uint64 {
if off < 0 || off+size > len(f.data) {
return math.MaxUint64
}
switch size {
case 4:
return uint64(binary.LittleEndian.Uint32(f.data[off:]))
case 8:
return binary.LittleEndian.Uint64(f.data[off:])
}
return math.MaxUint64
}
// at reads an address field of the file's address size at off.
func (f *hdf5File) at(off int) uint64 { return f.u64At(off, f.offSize) }
// len reads a length field of the file's length size at off.
func (f *hdf5File) length(off int) uint64 { return f.u64At(off, f.lenSize) }
// bytes returns the file bytes of the given region, or nil when the
// region lies outside the file.
func (f *hdf5File) bytes(addr uint64, n uint64) []byte {
if addr == math.MaxUint64 || n > uint64(len(f.data)) {
return nil
}
start := addr
if start > uint64(len(f.data)) || n > uint64(len(f.data))-start {
return nil
}
return f.data[start : start+n]
}
// hdf5Message is one header message with its raw payload.
type hdf5Message struct {
typ uint16
data []byte
}
// messages reads an object header's messages, following continuation
// blocks. Versions 1 and 2 are supported: version 1 is what the
// library writes by default, version 2 what the "latest" library
// version writes.
func (f *hdf5File) messages(addr uint64) ([]hdf5Message, error) {
const name = "LoadHDF5"
head := f.bytes(addr, 4)
if head == nil {
return nil, base.Errf("%s: object header at %d lies outside the file", name, addr)
}
if bytes.Equal(head, hdf5ObjHdr2) {
return f.messagesV2(addr)
}
start := f.bytes(addr, 16)
if start == nil {
return nil, base.Errf("%s: object header at %d lies outside the file", name, addr)
}
if start[0] != 1 {
return nil, base.Errf("%s: object header version %d is not supported; rewrite the file with the default library version", name, start[0])
}
// version(1), reserved(1), messages(2), references(4), data size(4),
// padding(4), then the message data itself.
nmsg := int(binary.LittleEndian.Uint16(start[2:]))
dataSize := int(binary.LittleEndian.Uint32(start[8:]))
pos := addr + 16
region := f.bytes(pos, uint64(dataSize))
if region == nil {
return nil, base.Errf("%s: object header at %d is truncated", name, addr)
}
// The declared count sizes the preallocation, so it is capped by the
// region that would have to hold the messages: a header claiming
// 65535 messages costs 16 bytes in the file, and every message is at
// least eight bytes, so the region bounds the count that can follow.
out := make([]hdf5Message, 0, min(nmsg, len(region)/8+1))
p := 0
for i := range nmsg {
if p+8 > len(region) {
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
}
typ := binary.LittleEndian.Uint16(region[p:])
size := int(binary.LittleEndian.Uint16(region[p+2:]))
if p+8+size > len(region) {
return nil, base.Errf("%s: object header at %d ends inside message %d", name, addr, i)
}
body := region[p+8 : p+8+size]
if typ == hdf5MsgContinuation {
// offset, length: another block of messages. The declared
// count includes the messages the chain carries, so the
// link is the last message physically present in this
// block; every further link is followed inside messagesIn,
// whose visited set keeps a hostile self-referencing block
// from looping. The link body must hold an address and a
// length outright: reading the fields from whatever follows
// a short message would guess at bytes the message never
// carried.
if size < f.offSize+f.lenSize {
return nil, base.Errf("%s: object header at %d carries a continuation message shorter than an offset and a length", name, addr)
}
blockAddr := f.at(int(pos) + p + 8)
next := f.bytes(blockAddr, f.length(int(pos)+p+8+f.offSize))
if next == nil {
return nil, base.Errf("%s: object header continuation at %d lies outside the file", name, addr)
}
cont, err := f.messagesIn(next, blockAddr, map[uint64]bool{addr: true}, 0)
if err != nil {
return nil, err
}
out = append(out, cont...)
break
}
out = append(out, hdf5Message{typ: typ, data: body})
p += 8 + size
p = alignUp(p, 8)
}
return out, nil
}
// messagesV2 reads a version 2 object header: the OHDR signature, a
// version and a flag byte, the flag-selected prefix fields, the size
// of the first chunk, then the messages and a lookup3 checksum over
// everything from the signature to the end of the chunk. The format
// inserts no alignment into the stream, so the walk reads the fields
// packed as they lie.
func (f *hdf5File) messagesV2(addr uint64) ([]hdf5Message, error) {
const name = "LoadHDF5"
head := f.bytes(addr, 6)
if head == nil {
return nil, base.Errf("%s: object header at %d lies outside the file", name, addr)
}
if head[4] != 2 {
return nil, base.Errf("%s: object header version %d is not supported; rewrite the file with the default library version", name, head[4])
}
flags := head[5]
if flags&^byte(0x3f) != 0 {
return nil, base.Errf("%s: the object header at %d carries unknown status flags %#02x", name, addr, flags)
}
p := addr + 6
if flags&0x20 != 0 {
p += 16 // access, modification, change and birth times
}
if flags&0x10 != 0 {
p += 4 // the max compact and min dense attribute counts
}
width := 1 << (flags & 0x03)
// The checksum of four bytes follows the chunk directly, so the
// size field, the chunk and the checksum must all lie inside the
// file; the bound is settled before the size is read, so a hostile
// header cannot point the read past the end.
if p+uint64(width)+4 > uint64(len(f.data)) {
return nil, base.Errf("%s: object header at %d is truncated", name, addr)
}
chunk0 := leN(f.data[p:], width)
if chunk0 > uint64(len(f.data))-p-uint64(width)-4 {
return nil, base.Errf("%s: object header at %d is truncated", name, addr)
}
start := p + uint64(width)
end := start + chunk0
if got, want := binary.LittleEndian.Uint32(f.data[end:]), hdf5Lookup3(f.data[addr:end]); got != want {
return nil, base.Errf("%s: the object header at %d fails its checksum (%#08x, want %#08x)", name, addr, got, want)
}
visited := map[uint64]bool{addr: true}
return f.messagesV2Stream(f.data[start:end], addr, flags&0x04 != 0, visited, 0)
}
// messagesV2Stream walks one version 2 message region: each message is
// one type byte, a two-byte size and a flag byte, widened by a
// two-byte creation order when the header tracks creation order. A
// continuation message hands the walk to a further OCHK block, whose
// checksum covers signature and messages alike; visited refuses a
// block reached twice and depth an unbounded chain. A writer may leave
// a gap of up to three bytes before the region's checksum; anything
// wider would be a message the region cannot hold.
func (f *hdf5File) messagesV2Stream(region []byte, addr uint64, ordered bool, visited map[uint64]bool, depth int) ([]hdf5Message, error) {
const name = "LoadHDF5"
if depth > hdf5MaxHeaderBlocks {
return nil, base.Errf("%s: the object header at %d chains through more than %d continuation blocks", name, addr, hdf5MaxHeaderBlocks)
}
out := []hdf5Message{}
p := 0
for p+4 <= len(region) {
typ := region[p]
size := int(binary.LittleEndian.Uint16(region[p+1:]))
p += 4
if ordered {
if p+2 > len(region) {
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
}
p += 2
}
if p+size > len(region) {
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
}
if typ == hdf5MsgContinuation {
// The link body is an offset of the address size followed by
// a length of the length size; a block that ends inside the
// link has neither, so the read is refused instead of taken
// from bytes past the block.
if p+f.offSize+f.lenSize > len(region) {
return nil, base.Errf("%s: object header at %d carries a continuation message shorter than an offset and a length", name, addr)
}
blockAddr := leN(region[p:], f.offSize)
block := f.bytes(blockAddr, leN(region[p+f.offSize:], f.lenSize))
if block == nil || len(block) < 8 || !bytes.Equal(block[:4], hdf5Chunk2) {
return nil, base.Errf("%s: no version 2 continuation block at %d", name, blockAddr)
}
if got, want := binary.LittleEndian.Uint32(block[len(block)-4:]), hdf5Lookup3(block[:len(block)-4]); got != want {
return nil, base.Errf("%s: the continuation block at %d fails its checksum (%#08x, want %#08x)", name, blockAddr, got, want)
}
if visited[blockAddr] {
return nil, base.Errf("%s: object header continuation block at %d is reachable twice", name, blockAddr)
}
visited[blockAddr] = true
cont, err := f.messagesV2Stream(block[4:len(block)-4], blockAddr, ordered, visited, depth+1)
if err != nil {
return nil, err
}
out = append(out, cont...)
} else if typ != 0 {
// A null message is the residue of a deletion and names no
// content; the live messages carry on behind it.
out = append(out, hdf5Message{typ: uint16(typ), data: region[p : p+size]})
}
p += size
}
if gap := len(region) - p; gap >= 4 {
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
}
return out, nil
}
// messagesIn reads the messages of one continuation block, whose size
// is the block itself. A continuation message inside the block chains
// to a further block; the visited set holds every block address already
// read, so a cycle is an error instead of an infinite walk, and depth
// bounds the chain the same way the version 2 walk bounds its own.
func (f *hdf5File) messagesIn(block []byte, addr uint64, visited map[uint64]bool, depth int) ([]hdf5Message, error) {
const name = "LoadHDF5"
if visited[addr] {
return nil, base.Errf("%s: object header continuation block at %d is reachable twice", name, addr)
}
visited[addr] = true
if depth > hdf5MaxHeaderBlocks {
return nil, base.Errf("%s: the object header at %d chains through more than %d continuation blocks", name, addr, hdf5MaxHeaderBlocks)
}
out := []hdf5Message{}
p := 0
for p+8 <= len(block) {
typ := binary.LittleEndian.Uint16(block[p:])
size := int(binary.LittleEndian.Uint16(block[p+2:]))
if p+8+size > len(block) {
return nil, base.Errf("%s: object header continuation block ends inside a message", name)
}
if typ == hdf5MsgContinuation {
// The link body is an offset of the address size followed
// by a length of the length size; a block that ends inside
// the link has neither, so the read is refused instead of
// taken from bytes past the block.
if p+8+f.offSize+f.lenSize > len(block) {
return nil, base.Errf("%s: object header continuation block at %d carries a continuation message shorter than an offset and a length", name, addr)
}
blockAddr := leN(block[p+8:], f.offSize)
next := f.bytes(blockAddr, leN(block[p+8+f.offSize:], f.lenSize))
if next == nil {
return nil, base.Errf("%s: object header continuation block at %d lies outside the file", name, blockAddr)
}
cont, err := f.messagesIn(next, blockAddr, visited, depth+1)
if err != nil {
return nil, err
}
out = append(out, cont...)
p = alignUp(p+8+size, 8)
continue
}
out = append(out, hdf5Message{typ: typ, data: block[p+8 : p+8+size]})
p = alignUp(p+8+size, 8)
}
return out, nil
}
// leN reads a little-endian integer of n bytes from the head of b.
func leN(b []byte, n int) uint64 {
var v uint64
for i := 0; i < n && i < len(b); i++ {
v |= uint64(b[i]) << (8 * i)
}
return v
}
func alignUp(n, to int) int { return (n + to - 1) / to * to }
// hdf5MaxGroupDepth bounds how deeply groups may nest. It is a stack
// guard for the walk, not a memory guard: the traversal carries one path
// buffer and one attribute map for the whole file, so the memory a deep
// file costs grows with its depth rather than with its square. Real
// files nest a handful of levels deep; the cap only refuses the extreme.
const hdf5MaxGroupDepth = 512
// hdf5MaxHeaderBlocks bounds how many continuation blocks one object
// header may chain through. It is a stack guard for the version 2
// message walk, whose recursion descends one frame per block: real
// headers split over a handful of blocks, the cap only refuses the
// chain a hostile file builds to exhaust the stack.
const hdf5MaxHeaderBlocks = 512
// hdf5DatasetEnvelope is the fixed per-dataset charge the walk deducts
// from the read budget beyond the value bytes themselves: the result's
// struct, shape slice, path string and attribute map cost roughly the
// same for every dataset, however small its values are, and a file of
// millions of empty datasets must pay for the envelopes it orders.
// The envelope is a conservative upper estimate of that structure, not
// a measurement of it, and it is charged together with the dataset's
// path length, which the result copies out of the walk's buffer.
const hdf5DatasetEnvelope = 512
// hdf5WalkState is the traversal state shared by the whole walk: the
// object headers currently on the path, so a hard-link cycle is refused
// instead of recursed, the object headers already walked from any path,
// so a diamond of hard links is read once instead of once per path, the
// inherited attribute map, which every level adds to and then undoes,
// and the path buffer, which every level extends and then truncates.
// All are shared rather than copied per level: a copy per level makes
// the live memory grow with the square of the nesting, which is the
// hostile-file blow-up the review found.
type hdf5WalkState struct {
onPath map[uint64]bool
seen map[uint64]bool
attrs map[string]string
path []byte
depth int
// budget is the byte extent the walk may still hand out, the
// whole-file aggregate behind the per-dataset cap: a file whose
// datasets each fit the cap can still declare thousands of them,
// and the sum must refuse, not the machine. Every dataset pays its
// value bytes plus hdf5DatasetEnvelope and its path.
budget int64
}
func newHDF5WalkState() *hdf5WalkState {
return &hdf5WalkState{onPath: map[uint64]bool{}, seen: map[uint64]bool{}, attrs: map[string]string{}, path: []byte{'/'}, budget: hdf5MaxDatasetBytes}
}
// hdf5AttrUndo records one attribute this level overwrote, so leaving
// the level restores the outer value instead of dropping it.
type hdf5AttrUndo struct {
key string
prev string
had bool
}
// walk visits the group whose path is already in st.path and everything
// under it, appending the datasets. The attributes of a group are
// inherited by everything below it, the nearest group winning, so a
// dataset carries the units and titles its file hangs on the enclosing
// groups.
//
// An object reachable through several hard links (a diamond) is walked
// once, under the first path the traversal reaches it from: st.seen
// records every object header already walked and a second arrival
// skips. The first path wins deterministically because the links of a
// group are visited in the order their file stores them. An object that
// closes a cycle along the current path is different: it is refused
// with an error, because no order of visits can read a cycle at all.
func (f *hdf5File) walk(addr uint64, st *hdf5WalkState, out *[]HDF5Dataset) error {
const name = "LoadHDF5"
if st.onPath[addr] {
return base.Errf("%s: the group at object header %d is linked into itself along %q; a hard-link cycle cannot be walked", name, addr, string(st.path))
}
if st.seen[addr] {
// Already walked from another path: its datasets are in out
// under that, earlier, path and the file's content is read.
return nil
}
if st.depth >= hdf5MaxGroupDepth {
return base.Errf("%s: the groups nest deeper than %d levels", name, hdf5MaxGroupDepth)
}
st.onPath[addr] = true
st.seen[addr] = true
st.depth++
defer func() {
delete(st.onPath, addr)
st.depth--
}()
msgs, err := f.messages(addr)
if err != nil {
return err
}
undo := make([]hdf5AttrUndo, 0, 4)
for k, v := range f.attributesOf(msgs) {
prev, had := st.attrs[k]
undo = append(undo, hdf5AttrUndo{key: k, prev: prev, had: had})
st.attrs[k] = v
}
defer func() {
for _, u := range slices.Backward(undo) {
if u.had {
st.attrs[u.key] = u.prev
continue
}
delete(st.attrs, u.key)
}
}()
links, err := f.links(msgs)
if err != nil {
return err
}
var datasetMsgs []hdf5Message
for _, m := range msgs {
switch m.typ {
case hdf5MsgDataspace, hdf5MsgDatatype, hdf5MsgDataLayout, hdf5MsgFilterPipeline, hdf5MsgFillValue:
datasetMsgs = append(datasetMsgs, m)
}
}
// A group has no dataspace; a dataset does.
hasSpace := false
for _, m := range datasetMsgs {
if m.typ == hdf5MsgDataspace {
hasSpace = true
}
}
if hasSpace {
ds, err := f.dataset(string(st.path), datasetMsgs, st)
if err != nil {
return err
}
// The dataset owns its attributes: a clone, because the walk's
// map keeps changing for the levels below this one.
ds.Attrs = maps.Clone(st.attrs)
*out = append(*out, ds)
return nil
}
for _, l := range links {
mark := len(st.path)
if st.path[mark-1] != '/' {
st.path = append(st.path, '/')
}
st.path = append(st.path, l.name...)
err := f.walk(l.address, st, out)
st.path = st.path[:mark]
if err != nil {
return err
}
}
return nil
}
// hdf5Link is one named hard link of a group.
type hdf5Link struct {
name string
address uint64
}
// links collects a group's hard links from its link messages and, when
// present, its symbol table.
func (f *hdf5File) links(msgs []hdf5Message) ([]hdf5Link, error) {
out := []hdf5Link{}
for _, m := range msgs {
switch m.typ {
case hdf5MsgLink:
l, err := f.decodeLink(m.data)
if err != nil {
return nil, err
}
if l.address != math.MaxUint64 {
out = append(out, l)
}
case hdf5MsgSymbolTable:
st, err := f.symbolTableLinks(m.data)
if err != nil {
return nil, err
}
out = append(out, st...)
case hdf5MsgLinkInfo:
// Dense storage would need the fractal heap; a group whose
// links all live in the header carries the info message
// without using the dense structures.
}
}
if len(out) == 0 {
for _, m := range msgs {
if m.typ == hdf5MsgLinkInfo {
const name = "LoadHDF5"
return nil, base.Errf("%s: this group stores its links in a dense (fractal heap) structure, which is not supported", name)
}
}
}
return out, nil
}
// decodeLink parses one link message, ignoring soft and external links
// (they name no object in this file).
func (f *hdf5File) decodeLink(m []byte) (hdf5Link, error) {
const name = "LoadHDF5"
if len(m) < 2 {
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
}
if m[0] != 1 {
return hdf5Link{}, base.Errf("%s: link message version %d is not supported", name, m[0])
}
flags := m[1]
p := 2
linkType := uint8(0)
if flags&0x08 != 0 {
if p >= len(m) {
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
}
linkType = m[p]
p++
}
if flags&0x04 != 0 {
if p+8 > len(m) {
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
}
p += 8 // creation order
}
nameLenSize := 1 << (flags & 0x03) // 1, 2, 4 or 8 bytes
nameLen, ok := readUintLE(m[p:], nameLenSize)
if !ok {
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
}
p += nameLenSize
if uint64(p)+nameLen > uint64(len(m)) {
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
}
linkName := string(m[p : p+int(nameLen)])
p += int(nameLen)
if linkType != 0 {
// A soft or external link: skip it, it names no object here.
return hdf5Link{name: linkName, address: math.MaxUint64}, nil
}
addr := f.at2(m[p:])
return hdf5Link{name: linkName, address: addr}, nil
}
// at2 reads an address of the file's address size from a message
// payload, or returns the undefined address.
func (f *hdf5File) at2(b []byte) uint64 {
if len(b) < f.offSize {
return math.MaxUint64
}
switch f.offSize {
case 4:
return uint64(binary.LittleEndian.Uint32(b))
case 8:
return binary.LittleEndian.Uint64(b)
}
return math.MaxUint64
}
func readUintLE(b []byte, size int) (uint64, bool) {
if len(b) < size {
return 0, false
}
switch size {
case 1:
return uint64(b[0]), true
case 2:
return uint64(binary.LittleEndian.Uint16(b)), true
case 4:
return uint64(binary.LittleEndian.Uint32(b)), true
case 8:
return binary.LittleEndian.Uint64(b), true
}
return 0, false
}
// symbolTableLinks walks the group's version 1 B-tree of symbol table
// nodes and reads the names out of the local heap.
func (f *hdf5File) symbolTableLinks(m []byte) ([]hdf5Link, error) {
const name = "LoadHDF5"
if len(m) < 2*f.offSize {
return nil, base.Errf("%s: symbol table message is truncated", name)
}
treeAddr := f.at2(m)
heapAddr := f.at2(m[f.offSize:])
heap, err := f.localHeap(heapAddr)
if err != nil {
return nil, err
}
out := []hdf5Link{}
set := map[uint64]bool{}
if err := f.treeLinks(treeAddr, heap, set, &out, 0); err != nil {
return nil, err
}
return out, nil
}
// treeLinks descends a group B-tree, reading a symbol table node at
// every leaf.
func (f *hdf5File) treeLinks(addr uint64, heap []byte, seen map[uint64]bool, out *[]hdf5Link, depth int) error {
const name = "LoadHDF5"
if depth > 32 {
return base.Errf("%s: the group B-tree is more than 32 levels deep", name)
}
if addr == math.MaxUint64 || seen[addr] {
return nil
}
seen[addr] = true
// The version 1 B-tree header is the signature, the type, the level
// and the entry count (eight bytes) plus a left and a right sibling
// address, so it is 8+2*offSize wide, not the 24 an 8-byte-address
// file happens to make it. Keying the entries off a pinned 24 read
// a 4/4 file's first key as a sibling address and lost the node.
node := f.bytes(addr, uint64(8+2*f.offSize))
if node == nil || !bytes.Equal(node[:4], hdf5Tree) {
return base.Errf("%s: no B-tree node at %d", name, addr)
}
nodeType := node[4]
level := node[5]
entries := int(binary.LittleEndian.Uint16(node[6:]))
if nodeType != 0 {
return base.Errf("%s: B-tree node type %d is not a group", name, nodeType)
}
// Each entry is a key of one address size followed by a child
// address; the node ends with one trailing key.
entrySize := f.lenSize + f.offSize
base := int(addr) + 8 + 2*f.offSize
for i := range entries {
child := f.at(base + i*entrySize + f.lenSize)
if level == 0 {
if err := f.symbolNodeLinks(child, heap, seen, out); err != nil {
return err
}
continue
}
if err := f.treeLinks(child, heap, seen, out, depth+1); err != nil {
return err
}
}
return nil
}
// symbolNodeLinks reads one symbol table node's entries.
func (f *hdf5File) symbolNodeLinks(addr uint64, heap []byte, seen map[uint64]bool, out *[]hdf5Link) error {
const name = "LoadHDF5"
if addr == math.MaxUint64 || seen[addr] {
return nil
}
seen[addr] = true
node := f.bytes(addr, 8)
if node == nil || !bytes.Equal(node[:4], hdf5SymbolNode) {
return base.Errf("%s: no symbol table node at %d", name, addr)
}
count := int(binary.LittleEndian.Uint16(node[6:]))
// An entry is the heap offset (of the length size), the object
// address (of the address size), the cache type, a reserved word
// and the 16-byte scratch pad: lenSize+offSize+24, not the 40 an
// 8/8 file happens to make it. A pinned 40 walked a 4/4 node's
// entries into the middle of the first one and read its second
// link's name from the wrong heap offset.
entrySize := f.lenSize + f.offSize + 24
start := int(addr) + 8
for i := range count {
off := start + i*entrySize
nameOff := f.length(off)
objAddr := f.at(off + f.lenSize)
linkName, err := heapString(heap, nameOff)
if err != nil {
return err
}
*out = append(*out, hdf5Link{name: linkName, address: objAddr})
}
return nil
}
// localHeap reads a local heap's data segment, the pool of
// null-terminated names the symbol table entries point into.
func (f *hdf5File) localHeap(addr uint64) ([]byte, error) {
const name = "LoadHDF5"
// The header is the signature and version, the data segment size
// and the free-list head (both of the length size), then the data
// segment address of the address size. Sizing the header with the
// address size instead (8+3*offSize) sliced the address out of a
// header that a 4/8 file stores in 20 bytes at [24:], which
// panicked; every field here is sized by its own kind.
head := f.bytes(addr, uint64(8+2*f.lenSize+f.offSize))
if head == nil || !bytes.Equal(head[:4], hdf5LocalHeap) {
return nil, base.Errf("%s: no local heap at %d", name, addr)
}
dataAddr := f.at2(head[8+2*f.lenSize:])
size := f.length(int(addr) + 8) // data segment size, then the free list head
seg := f.bytes(dataAddr, size)
if seg == nil {
return nil, base.Errf("%s: the local heap at %d lies outside the file", name, addr)
}
return seg, nil
}
// heapString reads the null-terminated string at an offset in the heap.
func heapString(heap []byte, off uint64) (string, error) {
if off >= uint64(len(heap)) {
return "", base.Errf("LoadHDF5: a symbol table entry names a name at heap offset %d, past the %d-byte heap",
off, len(heap))
}
end := bytes.IndexByte(heap[off:], 0)
if end < 0 {
return "", base.Errf("LoadHDF5: the name at heap offset %d is not terminated", off)
}
return string(heap[off : off+uint64(end)]), nil
}
// attributesOf collects the attributes carried in an object header.
func (f *hdf5File) attributesOf(msgs []hdf5Message) map[string]string {
out := map[string]string{}
for _, m := range msgs {
if m.typ != hdf5MsgAttribute {
continue
}
linkName, value, ok := f.decodeAttribute(m.data)
if ok {
out[linkName] = value
}
}
return out
}
// hdf5Type is the parsed datatype of a dataset or attribute: enough of
// it to read the values.
type hdf5Type struct {
class byte
size int
signed bool
width int // fixed-point: bytes
isFloat bool
isBool bool // a boolean enumeration: lands core.Bool
bitOrder byte
}
// hdf5ClassName names a datatype class for a refusal message: the
// number alone tells whoever wrote the file little about what the
// reader missed.
func hdf5ClassName(class byte) string {
switch class {
case 2:
return "date and time"
case 4:
return "bit field"
case 5:
return "opaque"
case 6:
return "compound"
case 7:
return "reference"
case 8:
return "enumeration"
case 9:
return "variable-length"
}
return "unknown"
}
// decodeType parses a datatype message (version 1, which is what the
// default library version writes). Strings are refused for datasets and
// accepted for attributes, where a string is exactly what is wanted.
// An enumeration is accepted only in the boolean convention HDF5
// writers carry booleans in, and lands core.Bool; every other
// enumeration and every bit field is refused by name.
func decodeType(m []byte, allowStrings bool) (hdf5Type, error) {
const name = "LoadHDF5"
if len(m) < 8 {
return hdf5Type{}, base.Errf("%s: datatype message is truncated", name)
}
// The first byte carries the version in its high nibble and the
// datatype class in its low one.
classAndVersion := m[0]
version := classAndVersion >> 4
class := classAndVersion & 0x0f
if version != 1 {
return hdf5Type{}, base.Errf("%s: datatype message version %d is not supported", name, version)
}
size := int(binary.LittleEndian.Uint32(m[4:]))
t := hdf5Type{class: class, size: size, bitOrder: m[1] >> 0}
// The class bit field's bit 0 is the byte order: 0 little-endian,
// anything else big-endian. Every element below is decoded as
// little-endian, so a big-endian datatype, on a dataset or on an
// attribute alike, would come back as byte-swapped noise with no
// error; it is refused by name instead.
if (class == 0 || class == 1) && m[1]&0x01 != 0 {
return hdf5Type{}, base.Errf("%s: big-endian datatypes are not supported; rewrite the file in the little-endian order", name)
}
switch class {
case 0: // fixed-point
if size != 1 && size != 2 && size != 4 && size != 8 {
return hdf5Type{}, base.Errf("%s: %d-byte fixed-point values are not supported", name, size)
}
t.signed = m[1]&0x08 != 0
t.width = size
case 1: // floating-point
if len(m) < 20 {
return hdf5Type{}, base.Errf("%s: floating-point datatype message is truncated", name)
}
switch size {
case 4, 8:
default:
return hdf5Type{}, base.Errf("%s: %d-byte floating-point values are not supported", name, size)
}
t.isFloat = true
t.width = size
case 3: // string
if !allowStrings {
return hdf5Type{}, base.Errf("%s: string datasets are not supported (only numeric arrays are read)", name)
}
t.width = size
if t.width <= 0 {
return hdf5Type{}, base.Errf("%s: a string attribute declares %d bytes", name, size)
}
case 8: // enumeration: only the boolean convention is read
if err := hdf5EnumBool(m); err != nil {
return hdf5Type{}, err
}
t.isBool = true
t.width = size
case 9: // variable-length: the text lives in a global heap collection
if !allowStrings {
return hdf5Type{}, base.Errf("%s: variable-length datasets are not supported (only numeric arrays are read)", name)
}
t.class = 9
t.width = size // the descriptor: length, heap address, object index
default:
return hdf5Type{}, base.Errf("%s: datatype class %d (%s) is not supported", name, class, hdf5ClassName(class))
}
return t, nil
}
// hdf5EnumBool verifies that an enumeration datatype message carries
// the boolean convention HDF5 writers store booleans in: a one-byte
// unsigned base type whose member values are a subset of {0, 1}. The
// member names are irrelevant to the values, which is what the landing
// keys on, and any other enumeration is refused by name.
//
// The message layout comes from the HDF5 file format specification's
// datatype message, enumeration class: the class bit field carries the
// member count in its low sixteen bits, and the properties hold the
// base type as a complete datatype message, then the member names
// (each a NUL-terminated string stored in a multiple of eight bytes,
// padded from its own field's start) and the packed member values. The
// base type here is a complete version 1 fixed-point message: its own
// eight-byte header plus the bit offset and bit precision the
// specification's fixed-point property table defines, twelve bytes in
// total, the same shape the reader already requires of the twenty-byte
// version 1 floating-point message.
func hdf5EnumBool(m []byte) error {
const name = "LoadHDF5"
members := int(binary.LittleEndian.Uint16(m[1:]))
if m[3] != 0 {
return base.Errf("%s: enumeration datatype class 8 carries unknown bit field bits %#02x", name, m[3])
}
if members < 1 {
return base.Errf("%s: enumeration datatype class 8 declares %d members", name, members)
}
if size := binary.LittleEndian.Uint32(m[4:]); size != 1 {
return base.Errf("%s: enumeration datatype class 8 with %d-byte values is not supported; only the one-byte unsigned boolean convention is read", name, size)
}
// The base type: a complete version 1 fixed-point message.
if len(m) < 8+12 {
return base.Errf("%s: enumeration datatype message is truncated", name)
}
b := m[8:]
if bclass := b[0] & 0x0f; b[0]>>4 != 1 || bclass != 0 {
return base.Errf("%s: enumeration datatype class 8 with a base type of class %d version %d is not supported; only the one-byte unsigned boolean convention is read",
name, bclass, b[0]>>4)
}
if b[1]&0x01 != 0 {
return base.Errf("%s: big-endian datatypes are not supported; rewrite the file in the little-endian order", name)
}
if b[1]&0x08 != 0 {
return base.Errf("%s: enumeration datatype class 8 with a signed base type is not supported; only the one-byte unsigned boolean convention is read", name)
}
if bsize := binary.LittleEndian.Uint32(b[4:]); bsize != 1 {
return base.Errf("%s: enumeration datatype class 8 with a base type of %d bytes is not supported; only the one-byte unsigned boolean convention is read", name, bsize)
}
// The names: each field is a NUL-terminated string padded, from its
// own start, to a multiple of eight bytes. The walk only needs the
// fields' extent; the values follow them.
p := 8 + 12
for i := range members {
end := -1
for j := p; j < len(m); j++ {
if m[j] == 0 {
end = j
break
}
}
if end < 0 {
return base.Errf("%s: enumeration datatype message ends inside member name %d", name, i+1)
}
p += alignUp(end-p+1, 8)
if p > len(m) {
return base.Errf("%s: enumeration datatype message ends inside the member names", name)
}
}
// The values: packed, one base-width byte each.
if members > len(m)-p {
return base.Errf("%s: enumeration datatype message holds %d member values, fewer than its %d members", name, len(m)-p, members)
}
for i := range members {
if v := m[p+i]; v > 1 {
return base.Errf("%s: enumeration datatype class 8 declares member value %d, outside the boolean convention {0, 1}", name, v)
}
}
return nil
}
// decodeDataspace parses a dataspace message into the dataset's shape.
// Version 1 is what the default library version writes, version 2 what
// the "latest" one writes: it drops the reserved bytes to one and
// always stores the dimensions in eight bytes.
func decodeDataspace(m []byte, lenSize int) ([]int, error) {
const name = "LoadHDF5"
if len(m) < 8 {
return nil, base.Errf("%s: dataspace message is truncated", name)
}
version := m[0]
rank := int(m[1])
flags := m[2]
if version != 1 && version != 2 {
return nil, base.Errf("%s: dataspace message version %d is not supported", name, version)
}
if rank == 0 {
return []int{1}, nil // a scalar: one element, no dimensions
}
pos, dimSize := 8, lenSize
if version == 2 {
pos, dimSize = 4, 8
}
shape := make([]int, rank)
for i := range rank {
if pos+dimSize > len(m) {
return nil, base.Errf("%s: dataspace message is truncated", name)
}
shape[i] = int(uint64At(m[pos:], dimSize))
pos += dimSize
}
if flags&0x02 != 0 {
return nil, base.Errf("%s: datasets with a dimension permutation are not supported", name)
}
if flags&0x01 != 0 {
// Maximum dimensions are not the shape; nothing to do with them.
pos += rank * dimSize
}
return shape, nil
}
func uint64At(b []byte, size int) uint64 {
v, _ := readUintLE(b, size)
return v
}
// hdf5Layout is a dataset's storage description. For chunked storage
// the message records one more dimension than the dataset has: the
// dimensions are the chunk's shape followed by the element size in
// bytes, and the chunk B-tree keys carry the same extra offset.
type hdf5Layout struct {
class byte
addr uint64
size uint64
compact []byte
dims []int
}
func decodeLayout(m []byte, offSize int) (hdf5Layout, error) {
const name = "LoadHDF5"
if len(m) < 2 {
return hdf5Layout{}, base.Errf("%s: data layout message is truncated", name)
}
version := m[0]
class := m[1]
switch version {
case 3, 4:
default:
return hdf5Layout{}, base.Errf("%s: data layout message version %d is not supported", name, version)
}
l := hdf5Layout{class: class}
switch class {
case 0: // compact: the data lives in the message
if len(m) < 4 {
return hdf5Layout{}, base.Errf("%s: compact layout message is truncated", name)
}
size := int(binary.LittleEndian.Uint16(m[2:]))
if 4+size > len(m) {
return hdf5Layout{}, base.Errf("%s: compact data runs past the message", name)
}
l.compact = m[4 : 4+size]
case 1: // contiguous
if len(m) < 2+offSize+8 {
return hdf5Layout{}, base.Errf("%s: contiguous layout message is truncated", name)
}
l.addr = uint64At(m[2:], offSize)
l.size = uint64At(m[2+offSize:], 8)
case 2: // chunked
if len(m) < 3+offSize {
return hdf5Layout{}, base.Errf("%s: chunked layout message is truncated", name)
}
rank := int(m[2])
p := 3
l.addr = uint64At(m[p:], offSize)
p += offSize
// Version 4 stores one flag byte before the chunk dimensions
// and widens them to 8 bytes when the flag asks for it.
dimSize := 4
if version == 4 {
if p >= len(m) {
return hdf5Layout{}, base.Errf("%s: chunked layout message is truncated", name)
}
if m[p]&0x01 != 0 {
dimSize = 8
}
p++
}
l.dims = make([]int, rank)
for i := range rank {
if p+dimSize > len(m) {
return hdf5Layout{}, base.Errf("%s: chunk dimensions run past the message", name)
}
l.dims[i] = int(uint64At(m[p:], dimSize))
p += dimSize
}
default:
return hdf5Layout{}, base.Errf("%s: storage class %d is not supported", name, class)
}
return l, nil
}
// hdf5Filter is one entry of a filter pipeline.
type hdf5Filter struct {
id uint16
values []uint32
}
func decodeFilters(m []byte) ([]hdf5Filter, error) {
const name = "LoadHDF5"
if len(m) < 2 {
return nil, base.Errf("%s: filter pipeline message is truncated", name)
}
version := m[0]
if version != 1 {
// Only version 1's layout is known; parsing another version on
// this layout would fail far away with an unrelated truncation
// error instead of naming the defect.
return nil, base.Errf("%s: unsupported filter pipeline message version %d", name, version)
}
count := int(m[1])
p := 2
p += 6 // reserved
out := make([]hdf5Filter, 0, count)
for range count {
if p+8 > len(m) {
return nil, base.Errf("%s: filter pipeline message is truncated", name)
}
filter := hdf5Filter{id: binary.LittleEndian.Uint16(m[p:])}
nameLen := int(binary.LittleEndian.Uint16(m[p+2:]))
nValues := int(binary.LittleEndian.Uint16(m[p+6:]))
p += 8
if p+nameLen > len(m) {
return nil, base.Errf("%s: filter name runs past the message", name)
}
p += nameLen
p = alignUp(p, 8)
for range nValues {
if p+4 > len(m) {
return nil, base.Errf("%s: filter values run past the message", name)
}
filter.values = append(filter.values, binary.LittleEndian.Uint32(m[p:]))
p += 4
}
p = alignUp(p, 8)
switch filter.id {
case hdf5FilterDeflate, hdf5FilterShuffle, hdf5FilterFletcher32:
default:
return nil, base.Errf("%s: filter %d is not supported", name, filter.id)
}
out = append(out, filter)
}
return out, nil
}
// dataset reads one dataset's values.
func (f *hdf5File) dataset(path string, msgs []hdf5Message, st *hdf5WalkState) (HDF5Dataset, error) {
const name = "LoadHDF5"
var (
shape []int
dtype hdf5Type
layout hdf5Layout
filters []hdf5Filter
haveType bool
haveSpace bool
haveLay bool
)
for _, m := range msgs {
switch m.typ {
case hdf5MsgDataspace:
s, err := decodeDataspace(m.data, f.lenSize)
if err != nil {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
}
shape, haveSpace = s, true
case hdf5MsgDatatype:
t, err := decodeType(m.data, false)
if err != nil {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
}
dtype, haveType = t, true
case hdf5MsgDataLayout:
l, err := decodeLayout(m.data, f.offSize)
if err != nil {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
}
layout, haveLay = l, true
case hdf5MsgFilterPipeline:
fs, err := decodeFilters(m.data)
if err != nil {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
}
filters = fs
}
}
if !haveSpace || !haveType || !haveLay {
return HDF5Dataset{}, base.Errf("%s: dataset %q is missing its %s", name, path,
missingOf(haveSpace, haveType, haveLay))
}
if layout.class == 2 {
// A latest-version file indexes its chunks with the version 2
// B-tree, which this reader does not walk; a classic file's
// chunks hang off the version 1 tree below.
if f.superblock >= 2 {
return HDF5Dataset{}, base.Errf("%s: dataset %q is chunked through a version 2 B-tree, which is not supported; rewrite the file with the default library version", name, path)
}
// The chunked message's last dimension is the element size, not
// a chunk dimension: the rank must be one less, and the value
// must agree with the datatype, which is a strong check that
// both were read correctly.
if len(layout.dims) != len(shape)+1 {
return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d chunk dimensions for rank %d",
name, path, len(layout.dims), len(shape))
}
if got := layout.dims[len(layout.dims)-1]; got != dtype.width {
return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d-byte chunk elements against a %d-byte datatype",
name, path, got, dtype.width)
}
layout.dims = layout.dims[:len(shape)]
}
// Fixed-point data lands by its stored width and signedness; an
// unsigned 64-bit value has no exact core dtype and is refused
// before any storage is read, with the same message the value
// decode reported it with.
if !dtype.isFloat && !dtype.signed && dtype.width == 8 {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path,
base.Errf("unsigned 64-bit integers have no exact core dtype"))
}
// The size arithmetic bounds every extent against the budget before
// it is multiplied: a hostile dataspace must fail the cap, not wrap
// the product into a small want that passes it. The validated byte
// count is what every storage class below works from.
want, err := hdf5ByteExtent(shape, dtype.width, hdf5MaxDatasetBytes)
if err != nil {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w; larger datasets need mmap", name, path, err)
}
n := int(want)
// The per-dataset cap bounds one dataset; the walk's aggregate
// bounds their sum, so a file declaring thousands of cap-fitting
// datasets refuses here instead of allocating the sum. The sum is
// charged past the value bytes: every accepted dataset also costs
// its result structure and its copied path, which the envelope and
// the path length stand for, so empty datasets pay too.
charge := int64(n) + int64(len(st.path)) + hdf5DatasetEnvelope
if charge > st.budget {
return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d bytes, the file's datasets are past the %d byte read budget; larger files need mmap",
name, path, n, hdf5MaxDatasetBytes)
}
st.budget -= charge
if layout.class == 2 {
values, cerr := f.chunkedArray(path, shape, dtype, layout, filters, n)
if cerr != nil {
return HDF5Dataset{}, cerr
}
return HDF5Dataset{Path: path, Shape: shape, Values: values}, nil
}
raw, rerr := f.rawData(path, shape, dtype, layout, n)
if rerr != nil {
return HDF5Dataset{}, rerr
}
values, aerr := arrayFromRaw(raw, dtype, shape)
if aerr != nil {
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, aerr)
}
return HDF5Dataset{Path: path, Shape: shape, Values: values}, nil
}
func missingOf(haveSpace, haveType, haveLay bool) string {
switch {
case !haveSpace:
return "dataspace"
case !haveType:
return "datatype"
case !haveLay:
return "storage layout"
}
return "messages"
}
// rawData gathers a dataset's bytes for the storage classes that hold
// them contiguously: compact storage is the message itself and
// contiguous storage is a span of the file. Chunked storage decodes
// into the values array directly, in chunkedArray. n is the dataset's
// byte size, validated against the read budget by the caller.
func (f *hdf5File) rawData(path string, shape []int, dtype hdf5Type, layout hdf5Layout, n int) ([]byte, error) {
const name = "LoadHDF5"
switch layout.class {
case 0:
if len(layout.compact) < n {
return nil, base.Errf("%s: dataset %q holds %d compact bytes, %d are needed",
name, path, len(layout.compact), n)
}
return layout.compact[:n], nil
case 1:
if layout.addr == math.MaxUint64 {
// The undefined address means storage was never allocated:
// for an empty dataset that is the whole answer (the fill
// value is the data, as the chunked branch already models),
// and for a non-empty one the bytes simply do not exist.
if n == 0 {
return []byte{}, nil
}
return nil, base.Errf("%s: dataset %q declares storage that was never allocated", name, path)
}
raw := f.bytes(layout.addr, uint64(n))
if raw == nil {
return nil, base.Errf("%s: dataset %q lies outside the file", name, path)
}
return raw, nil
}
return nil, base.Errf("%s: dataset %q has an unsupported storage class", name, path)
}
// hdf5ChunkStage carries what one dataset's chunk decode reuses across
// its chunks: the inflate output, the unshuffle staging, the inflate
// reader over the file's bytes, its limited view, and the index
// vectors the placement walk uses. A decode is serial, so the stage
// belongs to one call and nothing here is shared between concurrent
// loads.
type hdf5ChunkStage struct {
inflate []byte
unshuf []byte
zr io.ReadCloser
zrReset zlib.Resetter
br bytes.Reader
lim io.LimitedReader
offsets []uint64
vec []int
seen map[uint64]bool
}
// chunkedArray reassembles a chunked dataset through its version 1
// B-tree, running each chunk back through the filter pipeline and
// decoding each cell into the values array as it lands. bytes is the
// dataset's size, already validated against the read budget by the
// caller, so this function never re-derives it in int. The staging
// buffers, the inflate reader and the index vectors are one stage,
// reused across every chunk of the dataset.
func (f *hdf5File) chunkedArray(path string, shape []int, dtype hdf5Type, layout hdf5Layout, filters []hdf5Filter, bytes int) (*core.Array, error) {
const name = "LoadHDF5"
rank := len(shape)
if len(layout.dims) != rank {
return nil, base.Errf("%s: dataset %q declares %d chunk dimensions for rank %d", name, path, len(layout.dims), rank)
}
for _, c := range layout.dims {
// A chunk holds at least one element of the datatype; a zero or
// negative declared dimension is a hostile header, and both the
// strides below and the placement's overlap box degenerate on it.
if c <= 0 {
return nil, base.Errf("%s: dataset %q declares a %d-element chunk dimension", name, path, c)
}
}
// One chunk may not exceed the reader budget on its own, because the
// filters inflate into a buffer of exactly this size. The product is
// bounded extent by extent, so a hostile chunk dimension fails the
// cap instead of wrapping it.
chunkBytes, err := hdf5ByteExtent(layout.dims, dtype.width, hdf5MaxDatasetBytes)
if err != nil {
return nil, base.Errf("%s: dataset %q declares a chunk: %w", name, path, err)
}
cb := int(chunkBytes)
// The destination array and the per-cell decode carry the value
// mapping arrayFromRaw applies to contiguous bytes; the cells land
// in the payload as the walk places them, so no assembled copy of
// the dataset's bytes exists. Fixed-point cells land by stored width
// and signedness, exactly as the contiguous path lands them, and a
// boolean cell outside the enumeration's {0, 1} values is refused
// rather than coerced.
nElems := 0
if dtype.width > 0 {
nElems = bytes / dtype.width
}
var put func(data []byte, ci, oi int) error
var arr *core.Array
switch {
case dtype.isFloat && dtype.width == 8:
vals := make([]float64, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = math.Float64frombits(binary.LittleEndian.Uint64(data[ci*8:]))
return nil
}
arr, err = core.FloatsFromArray(vals, shape...)
case dtype.isFloat && dtype.width == 4:
vals := make([]float32, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = math.Float32frombits(binary.LittleEndian.Uint32(data[ci*4:]))
return nil
}
// The aliasing constructor: the decode fills the array's own
// payload through vals, which a copying constructor would have
// left behind as a separate slice.
arr, err = core.FromFloat32Slice(vals, shape...)
case dtype.isBool:
vals := make([]bool, nElems)
put = func(data []byte, ci, oi int) error {
b := data[ci]
if b > 1 {
return base.Errf("a boolean value of %d is outside the members 0 and 1", b)
}
vals[oi] = b != 0
return nil
}
arr, err = core.BoolsFromArray(vals, shape...)
case !dtype.signed && dtype.width == 8:
// Unreachable through LoadHDF5, which refuses the datatype
// first; the decode carries the same refusal for direct callers.
return nil, base.Errf("%s: dataset %q: %w", name, path,
base.Errf("unsigned 64-bit integers have no exact core dtype"))
case dtype.width == 1 && dtype.signed:
vals := make([]int8, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = int8(data[ci])
return nil
}
arr, err = core.Int8sFromArray(vals, shape...)
case dtype.width == 1:
vals := make([]uint8, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = data[ci]
return nil
}
arr, err = core.Uint8sFromArray(vals, shape...)
case dtype.width == 2 && dtype.signed:
vals := make([]int16, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = int16(binary.LittleEndian.Uint16(data[ci*2:]))
return nil
}
arr, err = core.Int16sFromArray(vals, shape...)
case dtype.width == 2:
vals := make([]uint16, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = binary.LittleEndian.Uint16(data[ci*2:])
return nil
}
arr, err = core.Uint16sFromArray(vals, shape...)
case dtype.width == 4 && dtype.signed:
vals := make([]int32, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = int32(binary.LittleEndian.Uint32(data[ci*4:]))
return nil
}
arr, err = core.Int32sFromArray(vals, shape...)
case dtype.width == 4:
vals := make([]uint32, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = binary.LittleEndian.Uint32(data[ci*4:])
return nil
}
arr, err = core.Uint32sFromArray(vals, shape...)
default:
// Signed 64-bit, the only fixed-point width left.
vals := make([]int64, nElems)
put = func(data []byte, ci, oi int) error {
vals[oi] = int64(binary.LittleEndian.Uint64(data[ci*8:]))
return nil
}
arr, err = core.IntsFromArray(vals, shape...)
}
if err != nil {
return nil, base.Errf("%s: dataset %q: %w", name, path, err)
}
if layout.addr == math.MaxUint64 {
return arr, nil // no chunks written: the fill value is the data
}
stage := &hdf5ChunkStage{
seen: map[uint64]bool{},
offsets: make([]uint64, rank),
vec: make([]int, 5*rank),
}
seenNodes := map[uint64]bool{}
return arr, f.chunkTreeWalk(layout.addr, layout, rank, seenNodes, func(addr uint64, offsets []uint64, mask uint32, size int) error {
if stage.seen[addr] {
return base.Errf("%s: dataset %q has two chunks at the same address", name, path)
}
stage.seen[addr] = true
stored := f.bytes(addr, uint64(size))
if stored == nil {
return base.Errf("%s: dataset %q has a chunk outside the file", name, path)
}
data, ferr := runFilters(stored, filters, mask, dtype, cb, stage)
if ferr != nil {
return base.Errf("%s: dataset %q: %w", name, path, ferr)
}
return placeCells(shape, layout.dims, offsets, data, dtype.width, cb, nElems, stage.vec, put)
}, 0, stage.offsets)
}
// placeCells walks the overlap of one chunk and the dataset, calling
// put for every cell with the chunk-local cell index and the output
// element index; the decode writes the value as the cell lands, and a
// cell put refuses stops the walk with that error.
// chunkBytes is the byte size of one complete chunk, computed by the
// caller with every dimension bounded against the reader budget, so
// the strides cannot wrap and a chunk is known to hold whole elements.
// The loop walks the overlap box in output coordinates, so the work is
// proportional to the placed elements: a chunk hanging past the
// shape's edge is padding the file may hold but the array has no room
// for, and one starting past an axis's end places nothing at all.
// Walking the whole declared chunk instead would let a hostile chunk
// dimension pin the reader for hours without touching a byte. vec
// carries the box bounds, the row-major strides of chunk and output,
// and the counter: five vectors of the rank, reused across chunks.
func placeCells(shape, chunkDim []int, offsets []uint64, data []byte, width, chunkBytes, maxOut int, vec []int, put func(data []byte, ci, oi int) error) error {
const name = "LoadHDF5"
if len(data) == 0 {
return nil
}
rank := len(shape)
if len(offsets) != rank {
return base.Errf("%s: a chunk carries %d offsets for rank %d", name, len(offsets), rank)
}
// The chunk must arrive complete: a truncated deflate stream would
// otherwise read past its data below.
if len(data) < chunkBytes {
return base.Errf("%s: a chunk holds %d bytes, a full chunk of %d is needed", name, len(data), chunkBytes)
}
lo := vec[:rank]
hi := vec[rank : 2*rank]
chunkStride := vec[2*rank : 3*rank]
outStride := vec[3*rank : 4*rank]
count := vec[4*rank : 5*rank]
for d := range rank {
if offsets[d] >= uint64(shape[d]) {
return nil
}
lo[d] = int(offsets[d])
hi[d] = min(lo[d]+chunkDim[d], shape[d])
if lo[d] == hi[d] {
return nil
}
}
// Row-major strides of the chunk and of the output. Both products
// are bounded: the chunk's by chunkBytes and the array's by the
// dataset's own validated byte count.
cs, os := 1, 1
for i := rank - 1; i >= 0; i-- {
chunkStride[i], outStride[i] = cs, os
cs *= chunkDim[i]
os *= shape[i]
}
// The element walk indexes both sides with a computed offset. The
// chunk's data may be a view of the file buffer, whose capacity
// runs past the chunk, so a stride that reached outside the chunk
// would read the file's other bytes silently instead of failing:
// the two limits are therefore checked explicitly, once, in element
// units.
maxChunk := len(data) / width
copy(count, lo)
for {
// The element's output coordinate and chunk-local coordinate.
oi, ci := 0, 0
for d := range rank {
oi += count[d] * outStride[d]
ci += (count[d] - lo[d]) * chunkStride[d]
}
if oi < 0 || oi >= maxOut || ci < 0 || ci >= maxChunk {
return base.Errf("%s: a chunk element lands at element %d of the %d-element array and %d of the %d-element chunk",
name, oi, maxOut, ci, maxChunk)
}
if err := put(data, ci, oi); err != nil {
return err
}
// Advance the odometer over the overlap box.
d := rank - 1
for d >= 0 {
count[d]++
if count[d] < hi[d] {
break
}
count[d] = lo[d]
d--
}
if d < 0 {
return nil
}
}
}
// chunkTree walks the chunk B-tree, calling place for every chunk,
// with a fresh per-entry offset vector. The decode reuses one vector
// across the whole walk through chunkTreeWalk instead.
func (f *hdf5File) chunkTree(addr uint64, layout hdf5Layout, rank int, nodes map[uint64]bool, place func(uint64, []uint64, uint32, int) error, depth int) error {
return f.chunkTreeWalk(addr, layout, rank, nodes, place, depth, make([]uint64, rank))
}
// chunkTreeWalk is the walk itself, with the per-entry offset vector
// supplied by the caller, so a decode allocates one for the dataset
// rather than one per chunk entry; a place callback consumes the
// offsets before it returns. nodes carries the addresses of the
// interior nodes already visited: a legal B-tree never revisits one,
// and without the set a hostile file whose node lists itself among its
// children would multiply the walk into entries^depth visits before
// the depth guard ever fires.
func (f *hdf5File) chunkTreeWalk(addr uint64, layout hdf5Layout, rank int, nodes map[uint64]bool, place func(uint64, []uint64, uint32, int) error, depth int, offsets []uint64) error {
const name = "LoadHDF5"
if depth > 32 {
return base.Errf("%s: the chunk B-tree is more than 32 levels deep", name)
}
if nodes[addr] {
return base.Errf("%s: the chunk B-tree revisits node %d", name, addr)
}
nodes[addr] = true
// The version 1 B-tree header is 8+2*offSize wide, the signature,
// type, level and entry count plus the two sibling addresses.
node := f.bytes(addr, uint64(8+2*f.offSize))
if node == nil || !bytes.Equal(node[:4], hdf5Tree) {
return base.Errf("%s: no chunk B-tree node at %d", name, addr)
}
if node[4] != 1 {
return base.Errf("%s: B-tree node type %d is not a chunk tree", name, node[4])
}
level := node[5]
entries := int(binary.LittleEndian.Uint16(node[6:]))
// The node header is 8+2*offSize wide, as in the group B-tree; the
// first key starts behind it.
nodeHeader := 8 + 2*f.offSize
// A v1 chunk key is the chunk size, the filter mask and one offset
// per dimension of the chunk *plus* the element size, which is why
// the key has one more offset than the dataset has axes; each entry
// is a key plus a child address.
keySize := 8 + 8*(rank+1)
entrySize := keySize + f.offSize
start := int(addr) + nodeHeader
for i := range entries {
p := start + i*entrySize
size := int(f.u64At(p, 4))
mask := uint32(f.u64At(p+4, 4))
for d := range rank {
offsets[d] = f.length(p + 8 + d*8)
}
child := f.at(p + keySize)
if level == 0 {
if err := place(child, offsets, mask, size); err != nil {
return err
}
continue
}
if err := f.chunkTreeWalk(child, layout, rank, nodes, place, depth+1, offsets); err != nil {
return err
}
}
return nil
}
// runFilters undoes the filter pipeline for one chunk, in reverse
// order; a filter whose bit is set in the mask did not run on it. The
// returned bytes are either the stored chunk untouched or one of the
// stage's buffers, valid until the stage handles its next chunk.
func runFilters(data []byte, filters []hdf5Filter, mask uint32, dtype hdf5Type, chunkBytes int, stage *hdf5ChunkStage) ([]byte, error) {
const name = "LoadHDF5"
out := data
for i, filter := range slices.Backward(filters) {
if mask&(1<<uint(i)) != 0 {
continue
}
switch filter.id {
case hdf5FilterFletcher32:
if len(out) < 4 {
return nil, base.Errf("%s: a fletcher32 chunk is shorter than its checksum", name)
}
if err := checkFletcher32(out); err != nil {
return nil, err
}
out = out[:len(out)-4]
case hdf5FilterShuffle:
if dtype.width <= 1 {
// One byte per element: the transpose is the identity.
continue
}
if cap(stage.unshuf) < len(out) {
stage.unshuf = make([]byte, len(out))
}
dst := stage.unshuf[:len(out)]
unshuffleInto(dst, out, dtype.width)
out = dst
case hdf5FilterDeflate:
// The filter's value is the compression level the writer
// used, which inflate does not need.
inflated, err := inflate(out, chunkBytes, stage)
if err != nil {
return nil, err
}
out = inflated
}
}
return out, nil
}
// inflate decompresses a deflate chunk into the stage's reused output
// buffer, refusing to grow past the chunk's own size. The filter
// stores a zlib stream, not a raw deflate one: the two-byte zlib
// header is present in the file. The reader is created once per
// dataset and reset over each next chunk's bytes; Reset re-initialises
// it to the state a fresh reader holds.
func inflate(data []byte, want int, stage *hdf5ChunkStage) ([]byte, error) {
stage.br.Reset(data)
if stage.zr == nil {
zr, zerr := zlib.NewReader(&stage.br)
if zerr != nil {
return nil, base.Errf("the deflate stream is corrupt: %w", zerr)
}
stage.zr, stage.zrReset = zr, zr.(zlib.Resetter)
} else if rerr := stage.zrReset.Reset(&stage.br, nil); rerr != nil {
return nil, base.Errf("the deflate stream is corrupt: %w", rerr)
}
// The chunk's size is known, so the output is sized once instead of
// grown from a small first guess. One byte past the chunk is what
// separates a stream that fills it from one that overflows it.
if cap(stage.inflate) < want+1 {
stage.inflate = make([]byte, 0, want+1)
}
out := stage.inflate[:0]
stage.lim.R = stage.zr
stage.lim.N = int64(want) + 1
for len(out) < cap(out) {
n, rerr := stage.lim.Read(out[len(out):cap(out)])
out = out[:len(out)+n]
if rerr != nil {
if rerr != io.EOF {
return nil, base.Errf("the deflate stream is corrupt: %w", rerr)
}
break
}
}
if len(out) > want {
return nil, base.Errf("the deflate stream inflates past the chunk size")
}
return out, nil
}
// unshuffleInto undoes the shuffle filter into out, which holds the
// same length as data: the filter transposes the bytes of the
// elements, so the first block holds every element's first byte. The
// transpose runs over tiles of rows: a tile is a short contiguous run
// of out, and the reads that fill it walk one row of the source
// sequentially, so neither stream strides across the whole chunk per
// column. Every byte still lands where the flat transpose put it.
func unshuffleInto(out, data []byte, width int) {
n := len(data) / width
const tileRows = 64
for j0 := 0; j0 < n; j0 += tileRows {
j1 := min(j0+tileRows, n)
for i := range width {
src := data[i*n+j0 : i*n+j1]
dst := out[j0*width+i:]
for k, b := range src {
dst[k*width] = b
}
}
}
}
// checkFletcher32 verifies the fletcher32 checksum HDF5 appends to a
// filtered chunk.
func checkFletcher32(data []byte) error {
body, sum := data[:len(data)-4], binary.LittleEndian.Uint32(data[len(data)-4:])
if got := fletcher32(body); got != sum {
return base.Errf("the fletcher32 checksum does not match: %08x, want %08x", got, sum)
}
return nil
}
// arrayFromRaw builds the array for a dataset from its bytes. The
// decoded buffer is exactly one element per value, so the array takes
// it over instead of copying it a second time. Fixed-point data lands
// the core dtype its stored width and signedness declare, a boolean
// enumeration lands Bool, and a boolean cell outside the members 0 and
// 1 is refused rather than coerced.
func arrayFromRaw(raw []byte, dtype hdf5Type, shape []int) (*core.Array, error) {
switch {
case dtype.isFloat && dtype.width == 8:
vals := make([]float64, len(raw)/8)
for i := range vals {
vals[i] = math.Float64frombits(binary.LittleEndian.Uint64(raw[i*8:]))
}
return core.FloatsFromArray(vals, shape...)
case dtype.isFloat && dtype.width == 4:
vals := make([]float32, len(raw)/4)
for i := range vals {
vals[i] = math.Float32frombits(binary.LittleEndian.Uint32(raw[i*4:]))
}
return core.FromFloat32s(vals, shape...)
case dtype.isBool:
vals := make([]bool, len(raw))
for i, b := range raw {
if b > 1 {
return nil, base.Errf("a boolean value of %d is outside the members 0 and 1", b)
}
vals[i] = b != 0
}
return core.BoolsFromArray(vals, shape...)
case !dtype.signed && dtype.width == 8:
return nil, base.Errf("unsigned 64-bit integers have no exact core dtype")
}
switch {
case dtype.width == 1 && dtype.signed:
vals := make([]int8, len(raw))
for i, b := range raw {
vals[i] = int8(b)
}
return core.Int8sFromArray(vals, shape...)
case dtype.width == 1:
// raw may be a view of the file buffer; the payload copies out
// of it before the array takes the copy over.
vals := make([]uint8, len(raw))
copy(vals, raw)
return core.Uint8sFromArray(vals, shape...)
case dtype.width == 2 && dtype.signed:
vals := make([]int16, len(raw)/2)
for i := range vals {
vals[i] = int16(binary.LittleEndian.Uint16(raw[i*2:]))
}
return core.Int16sFromArray(vals, shape...)
case dtype.width == 2:
vals := make([]uint16, len(raw)/2)
for i := range vals {
vals[i] = binary.LittleEndian.Uint16(raw[i*2:])
}
return core.Uint16sFromArray(vals, shape...)
case dtype.width == 4 && dtype.signed:
vals := make([]int32, len(raw)/4)
for i := range vals {
vals[i] = int32(binary.LittleEndian.Uint32(raw[i*4:]))
}
return core.Int32sFromArray(vals, shape...)
case dtype.width == 4:
vals := make([]uint32, len(raw)/4)
for i := range vals {
vals[i] = binary.LittleEndian.Uint32(raw[i*4:])
}
return core.Uint32sFromArray(vals, shape...)
case dtype.width == 8:
// Signed 64-bit: unsigned was refused above, and decodeType
// admits no other fixed-point width.
vals := make([]int64, len(raw)/8)
for i := range vals {
vals[i] = int64(binary.LittleEndian.Uint64(raw[i*8:]))
}
return core.IntsFromArray(vals, shape...)
}
return nil, base.Errf("%d-byte fixed-point values have no core dtype", dtype.width)
}
// decodeAttribute reads one attribute message: its name and a textual
// rendering of its value, the shape the package's attribute maps use.
func (f *hdf5File) decodeAttribute(m []byte) (string, string, bool) {
if len(m) < 8 || m[0] != 1 {
return "", "", false
}
nameSize := int(binary.LittleEndian.Uint16(m[2:]))
typeSize := int(binary.LittleEndian.Uint16(m[4:]))
spaceSize := int(binary.LittleEndian.Uint16(m[6:]))
p := 8
if p+nameSize+typeSize+spaceSize > len(m) {
return "", "", false
}
name := strings.TrimRight(string(m[p:p+nameSize]), "\x00")
// The name field is padded so the datatype starts on an eight-byte
// boundary. The padding is inside the message, so the datatype is
// checked against the body's own length: the slice underneath is a
// view of the file buffer, whose capacity runs past the body.
p = alignUp(p+nameSize, 8)
if p+typeSize > len(m) {
return "", "", false
}
dtype, err := decodeType(m[p:p+typeSize], true)
if err != nil {
return "", "", false
}
// The datatype message is padded to an eight-byte boundary before
// the dataspace begins.
p = alignUp(p+typeSize, 8)
if p+spaceSize > len(m) {
return "", "", false
}
// The dataspace's dimension fields are sized by the file's own
// length size, exactly as the dataset path's are.
shape, err := decodeDataspace(m[p:p+spaceSize], f.lenSize)
if err != nil {
return "", "", false
}
p += spaceSize
// The value's byte count is bounded extent by extent against what is
// left of the message: a hostile dataspace must fail here, not wrap
// the product into a negative that would slip past a check made
// afterwards and then size an allocation.
size, err := hdf5ByteExtent(shape, dtype.width, uint64(len(m)-p))
if err != nil {
return "", "", false
}
raw := m[p : p+int(size)]
if dtype.class == 3 {
// A fixed-length string: the bytes are the text.
return name, strings.TrimRight(string(raw), "\x00"), true
}
if dtype.class == 9 {
// A variable-length string: the attribute holds a descriptor
// naming a global heap object, and that object holds the text.
// The descriptor is four bytes of length, the collection
// address, then the object index.
if len(raw) < 8+f.offSize {
return "", "", false
}
length := int(binary.LittleEndian.Uint32(raw))
heapAddr := uint64At(raw[4:], f.offSize)
index := binary.LittleEndian.Uint32(raw[4+f.offSize:])
text, err := f.heapString(heapAddr, index, length)
if err != nil {
return "", "", false
}
return name, text, true
}
// The element count the numeric rendering below walks, derived from
// the validated byte count rather than from a second multiplication.
count := 0
if dtype.width > 0 {
count = len(raw) / dtype.width
}
vals := make([]string, 0, count)
for i := range count {
cell := raw[i*dtype.width : (i+1)*dtype.width]
switch {
case dtype.isFloat && dtype.width == 8:
vals = append(vals, strconv.FormatFloat(math.Float64frombits(binary.LittleEndian.Uint64(cell)), 'g', -1, 64))
case dtype.isFloat && dtype.width == 4:
vals = append(vals, strconv.FormatFloat(float64(math.Float32frombits(binary.LittleEndian.Uint32(cell))), 'g', -1, 32))
case dtype.isBool:
// The boolean enumeration renders as its own values; a cell
// outside them is a file that contradicts its datatype, and
// the attribute is dropped rather than guessed at.
if cell[0] > 1 {
return "", "", false
}
vals = append(vals, strconv.FormatInt(int64(cell[0]), 10))
case dtype.signed:
// The datatype's own signed bit decides how its stored bits
// read: an int8 attribute of -1 renders "-1", not 255.
u := uint64At(cell, dtype.width)
if width := uint(dtype.width); width < 64 {
mask := uint64(1)<<(8*width) - 1
if u&(uint64(1)<<(8*width-1)) != 0 {
u |= ^mask // sign-extend into the 64-bit read
}
}
vals = append(vals, strconv.FormatInt(int64(u), 10))
default:
vals = append(vals, strconv.FormatUint(uint64At(cell, dtype.width), 10))
}
}
if len(vals) == 1 {
return name, vals[0], true
}
return name, "[" + strings.Join(vals, ", ") + "]", true
}
// heapString reads a variable-length string out of a global heap
// collection: the descriptor names the collection and an object index,
// and the collection lists (index, reference count, size, bytes)
// records, each padded to eight bytes.
func (f *hdf5File) heapString(addr uint64, index uint32, length int) (string, error) {
const name = "LoadHDF5"
head := f.bytes(addr, 16)
if head == nil || !bytes.Equal(head[:4], []byte("GCOL")) {
return "", base.Errf("%s: no global heap collection at %d", name, addr)
}
size := f.length(int(addr) + 8) // the collection size, header included
collection := f.bytes(addr, size)
if collection == nil {
return "", base.Errf("%s: the global heap collection at %d lies outside the file", name, addr)
}
p := 8 + f.lenSize
for p+8+f.lenSize <= len(collection) {
objIndex := uint32(binary.LittleEndian.Uint16(collection[p:]))
objSize := uint64At(collection[p+8:], f.lenSize)
body := p + 8 + f.lenSize
// The bound compares in uint64: an int addition would wrap a
// hostile object size past the guard into a negative value and
// the slice below would panic.
if objSize > uint64(len(collection)-body) {
break
}
size := int(objSize)
if objIndex == index {
text := collection[body : body+size]
if length > 0 && length < len(text) {
text = text[:length]
}
return strings.TrimRight(string(text), "\x00"), nil
}
p = alignUp(body+size, 8)
}
return "", base.Errf("%s: the global heap object %d is not in the collection at %d", name, index, addr)
}
// fletcher32 is the checksum HDF5's fletcher32 filter appends: a
// 32-bit Fletcher sum over 16-bit words, ones-complement folded.
func fletcher32(data []byte) uint32 {
sum1, sum2 := uint32(0xffff), uint32(0xffff)
// The filter runs in blocks of 359 words (718 bytes).
for len(data) > 0 {
n := min(len(data), 718)
block := data[:n]
if n%2 != 0 {
block = append(append([]byte{}, block...), 0)
}
for i := 0; i+1 < len(block); i += 2 {
sum1 += uint32(binary.BigEndian.Uint16(block[i:]))
sum2 += sum1
}
sum1 = (sum1 & 0xffff) + (sum1 >> 16)
sum2 = (sum2 & 0xffff) + (sum2 >> 16)
data = data[n:]
}
sum1 = (sum1 & 0xffff) + (sum1 >> 16)
sum2 = (sum2 & 0xffff) + (sum2 >> 16)
return sum2<<16 | sum1
}
// hdf5Lookup3 is the Jenkins lookup3 hash in the little-endian,
// byte-wise form HDF5 stores as the checksum of superblock versions 2
// and 3 and of every chunk of an object header version 2, with the
// zero initial value the library uses. The main loop mixes whole
// 12-byte blocks and the switch folds the tail, whose bytes fall into
// the three words highest first.
func hdf5Lookup3(key []byte) uint32 {
a := uint32(0xdeadbeef) + uint32(len(key))
b := a
c := a
p := 0
for ; len(key)-p > 12; p += 12 {
a += binary.LittleEndian.Uint32(key[p:])
b += binary.LittleEndian.Uint32(key[p+4:])
c += binary.LittleEndian.Uint32(key[p+8:])
a, b, c = hdf5Lookup3Mix(a, b, c)
}
switch r := key[p:]; len(r) {
case 12:
c += uint32(r[11]) << 24
fallthrough
case 11:
c += uint32(r[10]) << 16
fallthrough
case 10:
c += uint32(r[9]) << 8
fallthrough
case 9:
c += uint32(r[8])
fallthrough
case 8:
b += uint32(r[7]) << 24
fallthrough
case 7:
b += uint32(r[6]) << 16
fallthrough
case 6:
b += uint32(r[5]) << 8
fallthrough
case 5:
b += uint32(r[4])
fallthrough
case 4:
a += uint32(r[3]) << 24
fallthrough
case 3:
a += uint32(r[2]) << 16
fallthrough
case 2:
a += uint32(r[1]) << 8
fallthrough
case 1:
a += uint32(r[0])
case 0:
return c
}
a, b, c = hdf5Lookup3Final(a, b, c)
return c
}
// hdf5Lookup3Mix is lookup3's inner round.
func hdf5Lookup3Mix(a, b, c uint32) (uint32, uint32, uint32) {
a -= c
a ^= bits.RotateLeft32(c, 4)
c += b
b -= a
b ^= bits.RotateLeft32(a, 6)
a += c
c -= b
c ^= bits.RotateLeft32(b, 8)
b += a
a -= c
a ^= bits.RotateLeft32(c, 16)
c += b
b -= a
b ^= bits.RotateLeft32(a, 19)
a += c
c -= b
c ^= bits.RotateLeft32(b, 4)
b += a
return a, b, c
}
// hdf5Lookup3Final is lookup3's closing avalanche.
func hdf5Lookup3Final(a, b, c uint32) (uint32, uint32, uint32) {
c ^= b
c -= bits.RotateLeft32(b, 14)
a ^= c
a -= bits.RotateLeft32(c, 11)
b ^= a
b -= bits.RotateLeft32(a, 25)
c ^= b
c -= bits.RotateLeft32(b, 16)
a ^= c
a -= bits.RotateLeft32(c, 4)
b ^= a
b -= bits.RotateLeft32(a, 14)
c ^= b
c -= bits.RotateLeft32(b, 24)
return a, b, c
}