2304 lines
81 KiB
Go
2304 lines
81 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
package io
|
|
|
|
import (
|
|
"bytes"
|
|
"compress/zlib"
|
|
"encoding/binary"
|
|
"io"
|
|
"maps"
|
|
"math"
|
|
"math/bits"
|
|
"os"
|
|
"slices"
|
|
"strconv"
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
|
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
|
)
|
|
|
|
// HDF5 (HDF5 1.8/1.10 file format), read-only, for the numeric arrays
|
|
// scientific files carry: every dataset of a file comes back as an
|
|
// Array with its path, shape and attributes.
|
|
//
|
|
// Supported: superblocks 0 to 3 (the classic layout and the "latest"
|
|
// library version, whose checksums are the lookup3 sum and are
|
|
// verified), object headers versions 1 and 2, groups stored as symbol
|
|
// tables or as link messages, datasets stored contiguously, compactly
|
|
// or in chunks through a version 1 B-tree, the deflate, shuffle and
|
|
// fletcher32 filters, fixed-point and floating-point datatypes of the
|
|
// usual widths, the boolean enumeration convention, and attributes in
|
|
// the object header. Fixed-point datasets land the core dtype their
|
|
// stored width and signedness declare; floating-point datasets land
|
|
// float64 at width 8 and float32 at width 4.
|
|
//
|
|
// Refused with an error naming what is missing, never guessed at: the
|
|
// superblock extension, the fractal-heap link storage of dense groups,
|
|
// the version 2 B-tree a latest-version file uses to index chunks,
|
|
// variable-length, compound, array and every non-boolean enumeration
|
|
// datatype, bit fields, string datasets, and the szip, nbit and
|
|
// scale-offset filters.
|
|
|
|
// Superblock and object header signatures.
|
|
var (
|
|
hdf5Magic = []byte{0x89, 'H', 'D', 'F', '\r', '\n', 0x1a, '\n'}
|
|
hdf5Tree = []byte{'T', 'R', 'E', 'E'}
|
|
hdf5SymbolNode = []byte{'S', 'N', 'O', 'D'}
|
|
hdf5LocalHeap = []byte{'H', 'E', 'A', 'P'}
|
|
hdf5ObjHdr2 = []byte{'O', 'H', 'D', 'R'}
|
|
hdf5Chunk2 = []byte{'O', 'C', 'H', 'K'}
|
|
)
|
|
|
|
// hdf5MaxDatasetBytes bounds the raw byte extent of one dataset and of
|
|
// one chunk. The reader materialises whole datasets, so a declared size
|
|
// beyond this budget is a hostile header, not data: without the cap a
|
|
// 16-byte file could order a terabyte-scale allocation through the
|
|
// dataspace, the fill-value path or a single chunk declaration. Honest
|
|
// datasets above the budget belong behind mmap, not behind an eager
|
|
// read.
|
|
const hdf5MaxDatasetBytes = 2 << 30 // 2 GiB
|
|
|
|
// hdf5ByteExtent multiplies declared extents by an element width into a
|
|
// byte count, bounding every factor against budget *before* it is
|
|
// multiplied. Multiplying first and comparing the product afterwards is
|
|
// the defect the review found: a dataspace of 2^32 by 2^32 float64
|
|
// elements is 2^64 bytes, which wraps to zero, and a cap that runs after
|
|
// the multiplication therefore accepts a declaration the file cannot
|
|
// back, while a chunk of 2^32 by 2^29 elements looks empty the same way.
|
|
// Every partial product here stays inside budget, so no caller can be
|
|
// handed a wrapped count either. A zero extent is legal and yields zero
|
|
// bytes (an empty dataset); a negative one is a hostile header, as is a
|
|
// non-positive element width.
|
|
func hdf5ByteExtent(dims []int, width int, budget uint64) (uint64, error) {
|
|
total := uint64(1)
|
|
empty := false
|
|
for _, d := range dims {
|
|
if d < 0 {
|
|
return 0, base.Errf("a declared extent of %d", d)
|
|
}
|
|
if d == 0 {
|
|
empty = true
|
|
continue
|
|
}
|
|
if uint64(d) > budget/total {
|
|
return 0, base.Errf("the declared extents multiply past the %d byte budget", budget)
|
|
}
|
|
total *= uint64(d)
|
|
}
|
|
if empty {
|
|
return 0, nil
|
|
}
|
|
if width <= 0 {
|
|
return 0, base.Errf("a declared element width of %d bytes", width)
|
|
}
|
|
if total > budget/uint64(width) {
|
|
return 0, base.Errf("the declared extents multiply past the %d byte budget", budget)
|
|
}
|
|
return total * uint64(width), nil
|
|
}
|
|
|
|
// HDF5 dataset object header message types.
|
|
const (
|
|
hdf5MsgDataspace = 1
|
|
hdf5MsgLinkInfo = 2
|
|
hdf5MsgDatatype = 3
|
|
hdf5MsgFillValueOld = 4
|
|
hdf5MsgFillValue = 5
|
|
hdf5MsgLink = 6
|
|
hdf5MsgDataLayout = 8
|
|
hdf5MsgGroupInfo = 10
|
|
hdf5MsgFilterPipeline = 11
|
|
hdf5MsgAttribute = 12
|
|
hdf5MsgContinuation = 16
|
|
hdf5MsgSymbolTable = 17
|
|
)
|
|
|
|
// HDF5 filter identifiers.
|
|
const (
|
|
hdf5FilterDeflate = 1
|
|
hdf5FilterShuffle = 2
|
|
hdf5FilterFletcher32 = 3
|
|
)
|
|
|
|
// HDF5Dataset is one dataset of a file: its path from the root, its
|
|
// shape (row-major, as the file stores it), its values and its
|
|
// attributes. The attribute map also carries the attributes of the
|
|
// groups the dataset sits in, the nearest one winning, because that is
|
|
// where files usually put the units and titles that apply to a whole
|
|
// group.
|
|
type HDF5Dataset struct {
|
|
Path string
|
|
Shape []int
|
|
Values *core.Array
|
|
Attrs map[string]string
|
|
}
|
|
|
|
// LoadHDF5 reads every dataset of an HDF5 file, in path order, as
|
|
// numeric arrays. The values keep the file's own dtype where the core
|
|
// has one: fixed-point data lands by stored width and signedness
|
|
// (int8, uint8, int16, uint16, int32, uint32, and int64 as Int), the
|
|
// boolean enumeration convention lands Bool, and floating-point data
|
|
// lands float64 at width 8 and float32 at width 4. Unsigned 64-bit
|
|
// data is refused, because no core dtype holds every value of it.
|
|
//
|
|
// An object hard-linked under several names is read once: its datasets
|
|
// appear under the first path the traversal reaches, the links of a
|
|
// group being visited in their file order, and never under the later
|
|
// ones. A link that closes a cycle along the path to it is an error,
|
|
// because no linear listing of the file exists.
|
|
func LoadHDF5(path string) ([]HDF5Dataset, error) {
|
|
data, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nil, base.Errf("LoadHDF5: %w", err)
|
|
}
|
|
f, err := newHDF5File(data)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := []HDF5Dataset{}
|
|
if err := f.walk(f.rootAddress, newHDF5WalkState(), &out); err != nil {
|
|
return nil, err
|
|
}
|
|
slices.SortFunc(out, func(x, y HDF5Dataset) int { return strings.Compare(x.Path, y.Path) })
|
|
return out, nil
|
|
}
|
|
|
|
// hdf5File is a parsed superblock plus the raw bytes of the file every
|
|
// offset is resolved against.
|
|
type hdf5File struct {
|
|
data []byte
|
|
superblock byte
|
|
offSize int
|
|
lenSize int
|
|
rootAddress uint64
|
|
groupLeafK int
|
|
groupInnerK int
|
|
}
|
|
|
|
func newHDF5File(data []byte) (*hdf5File, error) {
|
|
const name = "LoadHDF5"
|
|
if len(data) < 96 || !bytes.Equal(data[:8], hdf5Magic) {
|
|
return nil, base.Errf("%s: not an HDF5 file, the signature is % x", name, data[:min(8, len(data))])
|
|
}
|
|
version := data[8]
|
|
f := &hdf5File{data: data, superblock: version}
|
|
switch version {
|
|
case 0, 1:
|
|
// The classic layout: the sizes and the root group entry follow
|
|
// the fixed header.
|
|
case 2, 3:
|
|
// The modern layout: the sizes follow the version byte, the
|
|
// four addresses follow without the group K values, and the
|
|
// root group is named by its object header address. The whole
|
|
// superblock carries a lookup3 checksum.
|
|
f.offSize = int(data[9])
|
|
f.lenSize = int(data[10])
|
|
switch f.offSize {
|
|
case 4, 8:
|
|
default:
|
|
return nil, base.Errf("%s: %d-byte addresses are not supported", name, f.offSize)
|
|
}
|
|
if f.lenSize != f.offSize && f.lenSize != 8 && f.lenSize != 4 {
|
|
return nil, base.Errf("%s: %d-byte lengths are not supported", name, f.lenSize)
|
|
}
|
|
stored := 12 + 4*f.offSize + 4
|
|
if len(data) < stored {
|
|
return nil, base.Errf("%s: superblock version %d is truncated", name, version)
|
|
}
|
|
// Every address the format carries is absolute, so a non-zero
|
|
// base would need base-relative resolution, which is not
|
|
// implemented; the extension holds the file space strategy and
|
|
// driver settings, which the reader does not interpret. Both
|
|
// are refused by name rather than read wrong.
|
|
if baseAddr := f.u64At(12, f.offSize); baseAddr != 0 {
|
|
return nil, base.Errf("%s: a non-zero base address of %d is not supported", name, baseAddr)
|
|
}
|
|
if ext := f.u64At(12+f.offSize, f.offSize); ext != math.MaxUint64 {
|
|
return nil, base.Errf("%s: the superblock extension at %d is not supported; rewrite the file with the default library version", name, ext)
|
|
}
|
|
if got, want := binary.LittleEndian.Uint32(data[12+4*f.offSize:]), hdf5Lookup3(data[:12+4*f.offSize]); got != want {
|
|
return nil, base.Errf("%s: the superblock fails its checksum (%#08x, want %#08x)", name, got, want)
|
|
}
|
|
f.rootAddress = f.u64At(12+3*f.offSize, f.offSize)
|
|
if f.rootAddress == math.MaxUint64 {
|
|
return nil, base.Errf("%s: the root group has no object header", name)
|
|
}
|
|
return f, nil
|
|
default:
|
|
return nil, base.Errf("%s: superblock version %d is not supported; rewrite the file with the default version", name, version)
|
|
}
|
|
f.offSize = int(data[13])
|
|
f.lenSize = int(data[14])
|
|
switch f.offSize {
|
|
case 4, 8:
|
|
default:
|
|
return nil, base.Errf("%s: %d-byte addresses are not supported", name, f.offSize)
|
|
}
|
|
if f.lenSize != f.offSize && f.lenSize != 8 && f.lenSize != 4 {
|
|
return nil, base.Errf("%s: %d-byte lengths are not supported", name, f.lenSize)
|
|
}
|
|
f.groupLeafK = int(binary.LittleEndian.Uint16(data[16:]))
|
|
f.groupInnerK = int(binary.LittleEndian.Uint16(data[18:]))
|
|
pos := 24
|
|
if version == 1 {
|
|
// Version 1 carries the indexed-storage K before the addresses.
|
|
pos = 28
|
|
}
|
|
// base address (every file this reader accepts carries 0 there:
|
|
// a non-zero base would require base-relative offset resolution,
|
|
// which is not implemented), free space information, end of file,
|
|
// driver information: four addresses, then the root group's symbol
|
|
// table entry, whose first field is the link name offset and is
|
|
// sized by the length size, not the address size.
|
|
pos += 4 * f.offSize
|
|
f.rootAddress = f.u64At(pos+f.lenSize, f.offSize)
|
|
if f.rootAddress == math.MaxUint64 {
|
|
return nil, base.Errf("%s: the root group has no object header", name)
|
|
}
|
|
return f, nil
|
|
}
|
|
|
|
// u64At reads an address or length of size bytes at an absolute file
|
|
// offset; every offset the reader resolves is absolute.
|
|
func (f *hdf5File) u64At(off int, size int) uint64 {
|
|
if off < 0 || off+size > len(f.data) {
|
|
return math.MaxUint64
|
|
}
|
|
switch size {
|
|
case 4:
|
|
return uint64(binary.LittleEndian.Uint32(f.data[off:]))
|
|
case 8:
|
|
return binary.LittleEndian.Uint64(f.data[off:])
|
|
}
|
|
return math.MaxUint64
|
|
}
|
|
|
|
// at reads an address field of the file's address size at off.
|
|
func (f *hdf5File) at(off int) uint64 { return f.u64At(off, f.offSize) }
|
|
|
|
// len reads a length field of the file's length size at off.
|
|
func (f *hdf5File) length(off int) uint64 { return f.u64At(off, f.lenSize) }
|
|
|
|
// bytes returns the file bytes of the given region, or nil when the
|
|
// region lies outside the file.
|
|
func (f *hdf5File) bytes(addr uint64, n uint64) []byte {
|
|
if addr == math.MaxUint64 || n > uint64(len(f.data)) {
|
|
return nil
|
|
}
|
|
start := addr
|
|
if start > uint64(len(f.data)) || n > uint64(len(f.data))-start {
|
|
return nil
|
|
}
|
|
return f.data[start : start+n]
|
|
}
|
|
|
|
// hdf5Message is one header message with its raw payload.
|
|
type hdf5Message struct {
|
|
typ uint16
|
|
data []byte
|
|
}
|
|
|
|
// messages reads an object header's messages, following continuation
|
|
// blocks. Versions 1 and 2 are supported: version 1 is what the
|
|
// library writes by default, version 2 what the "latest" library
|
|
// version writes.
|
|
func (f *hdf5File) messages(addr uint64) ([]hdf5Message, error) {
|
|
const name = "LoadHDF5"
|
|
head := f.bytes(addr, 4)
|
|
if head == nil {
|
|
return nil, base.Errf("%s: object header at %d lies outside the file", name, addr)
|
|
}
|
|
if bytes.Equal(head, hdf5ObjHdr2) {
|
|
return f.messagesV2(addr)
|
|
}
|
|
start := f.bytes(addr, 16)
|
|
if start == nil {
|
|
return nil, base.Errf("%s: object header at %d lies outside the file", name, addr)
|
|
}
|
|
if start[0] != 1 {
|
|
return nil, base.Errf("%s: object header version %d is not supported; rewrite the file with the default library version", name, start[0])
|
|
}
|
|
// version(1), reserved(1), messages(2), references(4), data size(4),
|
|
// padding(4), then the message data itself.
|
|
nmsg := int(binary.LittleEndian.Uint16(start[2:]))
|
|
dataSize := int(binary.LittleEndian.Uint32(start[8:]))
|
|
pos := addr + 16
|
|
region := f.bytes(pos, uint64(dataSize))
|
|
if region == nil {
|
|
return nil, base.Errf("%s: object header at %d is truncated", name, addr)
|
|
}
|
|
// The declared count sizes the preallocation, so it is capped by the
|
|
// region that would have to hold the messages: a header claiming
|
|
// 65535 messages costs 16 bytes in the file, and every message is at
|
|
// least eight bytes, so the region bounds the count that can follow.
|
|
out := make([]hdf5Message, 0, min(nmsg, len(region)/8+1))
|
|
p := 0
|
|
for i := range nmsg {
|
|
if p+8 > len(region) {
|
|
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
|
|
}
|
|
typ := binary.LittleEndian.Uint16(region[p:])
|
|
size := int(binary.LittleEndian.Uint16(region[p+2:]))
|
|
if p+8+size > len(region) {
|
|
return nil, base.Errf("%s: object header at %d ends inside message %d", name, addr, i)
|
|
}
|
|
body := region[p+8 : p+8+size]
|
|
if typ == hdf5MsgContinuation {
|
|
// offset, length: another block of messages. The declared
|
|
// count includes the messages the chain carries, so the
|
|
// link is the last message physically present in this
|
|
// block; every further link is followed inside messagesIn,
|
|
// whose visited set keeps a hostile self-referencing block
|
|
// from looping. The link body must hold an address and a
|
|
// length outright: reading the fields from whatever follows
|
|
// a short message would guess at bytes the message never
|
|
// carried.
|
|
if size < f.offSize+f.lenSize {
|
|
return nil, base.Errf("%s: object header at %d carries a continuation message shorter than an offset and a length", name, addr)
|
|
}
|
|
blockAddr := f.at(int(pos) + p + 8)
|
|
next := f.bytes(blockAddr, f.length(int(pos)+p+8+f.offSize))
|
|
if next == nil {
|
|
return nil, base.Errf("%s: object header continuation at %d lies outside the file", name, addr)
|
|
}
|
|
cont, err := f.messagesIn(next, blockAddr, map[uint64]bool{addr: true}, 0)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, cont...)
|
|
break
|
|
}
|
|
out = append(out, hdf5Message{typ: typ, data: body})
|
|
p += 8 + size
|
|
p = alignUp(p, 8)
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// messagesV2 reads a version 2 object header: the OHDR signature, a
|
|
// version and a flag byte, the flag-selected prefix fields, the size
|
|
// of the first chunk, then the messages and a lookup3 checksum over
|
|
// everything from the signature to the end of the chunk. The format
|
|
// inserts no alignment into the stream, so the walk reads the fields
|
|
// packed as they lie.
|
|
func (f *hdf5File) messagesV2(addr uint64) ([]hdf5Message, error) {
|
|
const name = "LoadHDF5"
|
|
head := f.bytes(addr, 6)
|
|
if head == nil {
|
|
return nil, base.Errf("%s: object header at %d lies outside the file", name, addr)
|
|
}
|
|
if head[4] != 2 {
|
|
return nil, base.Errf("%s: object header version %d is not supported; rewrite the file with the default library version", name, head[4])
|
|
}
|
|
flags := head[5]
|
|
if flags&^byte(0x3f) != 0 {
|
|
return nil, base.Errf("%s: the object header at %d carries unknown status flags %#02x", name, addr, flags)
|
|
}
|
|
p := addr + 6
|
|
if flags&0x20 != 0 {
|
|
p += 16 // access, modification, change and birth times
|
|
}
|
|
if flags&0x10 != 0 {
|
|
p += 4 // the max compact and min dense attribute counts
|
|
}
|
|
width := 1 << (flags & 0x03)
|
|
// The checksum of four bytes follows the chunk directly, so the
|
|
// size field, the chunk and the checksum must all lie inside the
|
|
// file; the bound is settled before the size is read, so a hostile
|
|
// header cannot point the read past the end.
|
|
if p+uint64(width)+4 > uint64(len(f.data)) {
|
|
return nil, base.Errf("%s: object header at %d is truncated", name, addr)
|
|
}
|
|
chunk0 := leN(f.data[p:], width)
|
|
if chunk0 > uint64(len(f.data))-p-uint64(width)-4 {
|
|
return nil, base.Errf("%s: object header at %d is truncated", name, addr)
|
|
}
|
|
start := p + uint64(width)
|
|
end := start + chunk0
|
|
if got, want := binary.LittleEndian.Uint32(f.data[end:]), hdf5Lookup3(f.data[addr:end]); got != want {
|
|
return nil, base.Errf("%s: the object header at %d fails its checksum (%#08x, want %#08x)", name, addr, got, want)
|
|
}
|
|
visited := map[uint64]bool{addr: true}
|
|
return f.messagesV2Stream(f.data[start:end], addr, flags&0x04 != 0, visited, 0)
|
|
}
|
|
|
|
// messagesV2Stream walks one version 2 message region: each message is
|
|
// one type byte, a two-byte size and a flag byte, widened by a
|
|
// two-byte creation order when the header tracks creation order. A
|
|
// continuation message hands the walk to a further OCHK block, whose
|
|
// checksum covers signature and messages alike; visited refuses a
|
|
// block reached twice and depth an unbounded chain. A writer may leave
|
|
// a gap of up to three bytes before the region's checksum; anything
|
|
// wider would be a message the region cannot hold.
|
|
func (f *hdf5File) messagesV2Stream(region []byte, addr uint64, ordered bool, visited map[uint64]bool, depth int) ([]hdf5Message, error) {
|
|
const name = "LoadHDF5"
|
|
if depth > hdf5MaxHeaderBlocks {
|
|
return nil, base.Errf("%s: the object header at %d chains through more than %d continuation blocks", name, addr, hdf5MaxHeaderBlocks)
|
|
}
|
|
out := []hdf5Message{}
|
|
p := 0
|
|
for p+4 <= len(region) {
|
|
typ := region[p]
|
|
size := int(binary.LittleEndian.Uint16(region[p+1:]))
|
|
p += 4
|
|
if ordered {
|
|
if p+2 > len(region) {
|
|
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
|
|
}
|
|
p += 2
|
|
}
|
|
if p+size > len(region) {
|
|
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
|
|
}
|
|
if typ == hdf5MsgContinuation {
|
|
// The link body is an offset of the address size followed by
|
|
// a length of the length size; a block that ends inside the
|
|
// link has neither, so the read is refused instead of taken
|
|
// from bytes past the block.
|
|
if p+f.offSize+f.lenSize > len(region) {
|
|
return nil, base.Errf("%s: object header at %d carries a continuation message shorter than an offset and a length", name, addr)
|
|
}
|
|
blockAddr := leN(region[p:], f.offSize)
|
|
block := f.bytes(blockAddr, leN(region[p+f.offSize:], f.lenSize))
|
|
if block == nil || len(block) < 8 || !bytes.Equal(block[:4], hdf5Chunk2) {
|
|
return nil, base.Errf("%s: no version 2 continuation block at %d", name, blockAddr)
|
|
}
|
|
if got, want := binary.LittleEndian.Uint32(block[len(block)-4:]), hdf5Lookup3(block[:len(block)-4]); got != want {
|
|
return nil, base.Errf("%s: the continuation block at %d fails its checksum (%#08x, want %#08x)", name, blockAddr, got, want)
|
|
}
|
|
if visited[blockAddr] {
|
|
return nil, base.Errf("%s: object header continuation block at %d is reachable twice", name, blockAddr)
|
|
}
|
|
visited[blockAddr] = true
|
|
cont, err := f.messagesV2Stream(block[4:len(block)-4], blockAddr, ordered, visited, depth+1)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, cont...)
|
|
} else if typ != 0 {
|
|
// A null message is the residue of a deletion and names no
|
|
// content; the live messages carry on behind it.
|
|
out = append(out, hdf5Message{typ: uint16(typ), data: region[p : p+size]})
|
|
}
|
|
p += size
|
|
}
|
|
if gap := len(region) - p; gap >= 4 {
|
|
return nil, base.Errf("%s: object header at %d ends inside a message", name, addr)
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// messagesIn reads the messages of one continuation block, whose size
|
|
// is the block itself. A continuation message inside the block chains
|
|
// to a further block; the visited set holds every block address already
|
|
// read, so a cycle is an error instead of an infinite walk, and depth
|
|
// bounds the chain the same way the version 2 walk bounds its own.
|
|
func (f *hdf5File) messagesIn(block []byte, addr uint64, visited map[uint64]bool, depth int) ([]hdf5Message, error) {
|
|
const name = "LoadHDF5"
|
|
if visited[addr] {
|
|
return nil, base.Errf("%s: object header continuation block at %d is reachable twice", name, addr)
|
|
}
|
|
visited[addr] = true
|
|
if depth > hdf5MaxHeaderBlocks {
|
|
return nil, base.Errf("%s: the object header at %d chains through more than %d continuation blocks", name, addr, hdf5MaxHeaderBlocks)
|
|
}
|
|
out := []hdf5Message{}
|
|
p := 0
|
|
for p+8 <= len(block) {
|
|
typ := binary.LittleEndian.Uint16(block[p:])
|
|
size := int(binary.LittleEndian.Uint16(block[p+2:]))
|
|
if p+8+size > len(block) {
|
|
return nil, base.Errf("%s: object header continuation block ends inside a message", name)
|
|
}
|
|
if typ == hdf5MsgContinuation {
|
|
// The link body is an offset of the address size followed
|
|
// by a length of the length size; a block that ends inside
|
|
// the link has neither, so the read is refused instead of
|
|
// taken from bytes past the block.
|
|
if p+8+f.offSize+f.lenSize > len(block) {
|
|
return nil, base.Errf("%s: object header continuation block at %d carries a continuation message shorter than an offset and a length", name, addr)
|
|
}
|
|
blockAddr := leN(block[p+8:], f.offSize)
|
|
next := f.bytes(blockAddr, leN(block[p+8+f.offSize:], f.lenSize))
|
|
if next == nil {
|
|
return nil, base.Errf("%s: object header continuation block at %d lies outside the file", name, blockAddr)
|
|
}
|
|
cont, err := f.messagesIn(next, blockAddr, visited, depth+1)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, cont...)
|
|
p = alignUp(p+8+size, 8)
|
|
continue
|
|
}
|
|
out = append(out, hdf5Message{typ: typ, data: block[p+8 : p+8+size]})
|
|
p = alignUp(p+8+size, 8)
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// leN reads a little-endian integer of n bytes from the head of b.
|
|
func leN(b []byte, n int) uint64 {
|
|
var v uint64
|
|
for i := 0; i < n && i < len(b); i++ {
|
|
v |= uint64(b[i]) << (8 * i)
|
|
}
|
|
return v
|
|
}
|
|
|
|
func alignUp(n, to int) int { return (n + to - 1) / to * to }
|
|
|
|
// hdf5MaxGroupDepth bounds how deeply groups may nest. It is a stack
|
|
// guard for the walk, not a memory guard: the traversal carries one path
|
|
// buffer and one attribute map for the whole file, so the memory a deep
|
|
// file costs grows with its depth rather than with its square. Real
|
|
// files nest a handful of levels deep; the cap only refuses the extreme.
|
|
const hdf5MaxGroupDepth = 512
|
|
|
|
// hdf5MaxHeaderBlocks bounds how many continuation blocks one object
|
|
// header may chain through. It is a stack guard for the version 2
|
|
// message walk, whose recursion descends one frame per block: real
|
|
// headers split over a handful of blocks, the cap only refuses the
|
|
// chain a hostile file builds to exhaust the stack.
|
|
const hdf5MaxHeaderBlocks = 512
|
|
|
|
// hdf5DatasetEnvelope is the fixed per-dataset charge the walk deducts
|
|
// from the read budget beyond the value bytes themselves: the result's
|
|
// struct, shape slice, path string and attribute map cost roughly the
|
|
// same for every dataset, however small its values are, and a file of
|
|
// millions of empty datasets must pay for the envelopes it orders.
|
|
// The envelope is a conservative upper estimate of that structure, not
|
|
// a measurement of it, and it is charged together with the dataset's
|
|
// path length, which the result copies out of the walk's buffer.
|
|
const hdf5DatasetEnvelope = 512
|
|
|
|
// hdf5WalkState is the traversal state shared by the whole walk: the
|
|
// object headers currently on the path, so a hard-link cycle is refused
|
|
// instead of recursed, the object headers already walked from any path,
|
|
// so a diamond of hard links is read once instead of once per path, the
|
|
// inherited attribute map, which every level adds to and then undoes,
|
|
// and the path buffer, which every level extends and then truncates.
|
|
// All are shared rather than copied per level: a copy per level makes
|
|
// the live memory grow with the square of the nesting, which is the
|
|
// hostile-file blow-up the review found.
|
|
type hdf5WalkState struct {
|
|
onPath map[uint64]bool
|
|
seen map[uint64]bool
|
|
attrs map[string]string
|
|
path []byte
|
|
depth int
|
|
// budget is the byte extent the walk may still hand out, the
|
|
// whole-file aggregate behind the per-dataset cap: a file whose
|
|
// datasets each fit the cap can still declare thousands of them,
|
|
// and the sum must refuse, not the machine. Every dataset pays its
|
|
// value bytes plus hdf5DatasetEnvelope and its path.
|
|
budget int64
|
|
}
|
|
|
|
func newHDF5WalkState() *hdf5WalkState {
|
|
return &hdf5WalkState{onPath: map[uint64]bool{}, seen: map[uint64]bool{}, attrs: map[string]string{}, path: []byte{'/'}, budget: hdf5MaxDatasetBytes}
|
|
}
|
|
|
|
// hdf5AttrUndo records one attribute this level overwrote, so leaving
|
|
// the level restores the outer value instead of dropping it.
|
|
type hdf5AttrUndo struct {
|
|
key string
|
|
prev string
|
|
had bool
|
|
}
|
|
|
|
// walk visits the group whose path is already in st.path and everything
|
|
// under it, appending the datasets. The attributes of a group are
|
|
// inherited by everything below it, the nearest group winning, so a
|
|
// dataset carries the units and titles its file hangs on the enclosing
|
|
// groups.
|
|
//
|
|
// An object reachable through several hard links (a diamond) is walked
|
|
// once, under the first path the traversal reaches it from: st.seen
|
|
// records every object header already walked and a second arrival
|
|
// skips. The first path wins deterministically because the links of a
|
|
// group are visited in the order their file stores them. An object that
|
|
// closes a cycle along the current path is different: it is refused
|
|
// with an error, because no order of visits can read a cycle at all.
|
|
func (f *hdf5File) walk(addr uint64, st *hdf5WalkState, out *[]HDF5Dataset) error {
|
|
const name = "LoadHDF5"
|
|
if st.onPath[addr] {
|
|
return base.Errf("%s: the group at object header %d is linked into itself along %q; a hard-link cycle cannot be walked", name, addr, string(st.path))
|
|
}
|
|
if st.seen[addr] {
|
|
// Already walked from another path: its datasets are in out
|
|
// under that, earlier, path and the file's content is read.
|
|
return nil
|
|
}
|
|
if st.depth >= hdf5MaxGroupDepth {
|
|
return base.Errf("%s: the groups nest deeper than %d levels", name, hdf5MaxGroupDepth)
|
|
}
|
|
st.onPath[addr] = true
|
|
st.seen[addr] = true
|
|
st.depth++
|
|
defer func() {
|
|
delete(st.onPath, addr)
|
|
st.depth--
|
|
}()
|
|
msgs, err := f.messages(addr)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
undo := make([]hdf5AttrUndo, 0, 4)
|
|
for k, v := range f.attributesOf(msgs) {
|
|
prev, had := st.attrs[k]
|
|
undo = append(undo, hdf5AttrUndo{key: k, prev: prev, had: had})
|
|
st.attrs[k] = v
|
|
}
|
|
defer func() {
|
|
for _, u := range slices.Backward(undo) {
|
|
if u.had {
|
|
st.attrs[u.key] = u.prev
|
|
continue
|
|
}
|
|
delete(st.attrs, u.key)
|
|
}
|
|
}()
|
|
links, err := f.links(msgs)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
var datasetMsgs []hdf5Message
|
|
for _, m := range msgs {
|
|
switch m.typ {
|
|
case hdf5MsgDataspace, hdf5MsgDatatype, hdf5MsgDataLayout, hdf5MsgFilterPipeline, hdf5MsgFillValue:
|
|
datasetMsgs = append(datasetMsgs, m)
|
|
}
|
|
}
|
|
// A group has no dataspace; a dataset does.
|
|
hasSpace := false
|
|
for _, m := range datasetMsgs {
|
|
if m.typ == hdf5MsgDataspace {
|
|
hasSpace = true
|
|
}
|
|
}
|
|
if hasSpace {
|
|
ds, err := f.dataset(string(st.path), datasetMsgs, st)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
// The dataset owns its attributes: a clone, because the walk's
|
|
// map keeps changing for the levels below this one.
|
|
ds.Attrs = maps.Clone(st.attrs)
|
|
*out = append(*out, ds)
|
|
return nil
|
|
}
|
|
for _, l := range links {
|
|
mark := len(st.path)
|
|
if st.path[mark-1] != '/' {
|
|
st.path = append(st.path, '/')
|
|
}
|
|
st.path = append(st.path, l.name...)
|
|
err := f.walk(l.address, st, out)
|
|
st.path = st.path[:mark]
|
|
if err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// hdf5Link is one named hard link of a group.
|
|
type hdf5Link struct {
|
|
name string
|
|
address uint64
|
|
}
|
|
|
|
// links collects a group's hard links from its link messages and, when
|
|
// present, its symbol table.
|
|
func (f *hdf5File) links(msgs []hdf5Message) ([]hdf5Link, error) {
|
|
out := []hdf5Link{}
|
|
for _, m := range msgs {
|
|
switch m.typ {
|
|
case hdf5MsgLink:
|
|
l, err := f.decodeLink(m.data)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if l.address != math.MaxUint64 {
|
|
out = append(out, l)
|
|
}
|
|
case hdf5MsgSymbolTable:
|
|
st, err := f.symbolTableLinks(m.data)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out = append(out, st...)
|
|
case hdf5MsgLinkInfo:
|
|
// Dense storage would need the fractal heap; a group whose
|
|
// links all live in the header carries the info message
|
|
// without using the dense structures.
|
|
}
|
|
}
|
|
if len(out) == 0 {
|
|
for _, m := range msgs {
|
|
if m.typ == hdf5MsgLinkInfo {
|
|
const name = "LoadHDF5"
|
|
return nil, base.Errf("%s: this group stores its links in a dense (fractal heap) structure, which is not supported", name)
|
|
}
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// decodeLink parses one link message, ignoring soft and external links
|
|
// (they name no object in this file).
|
|
func (f *hdf5File) decodeLink(m []byte) (hdf5Link, error) {
|
|
const name = "LoadHDF5"
|
|
if len(m) < 2 {
|
|
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
|
|
}
|
|
if m[0] != 1 {
|
|
return hdf5Link{}, base.Errf("%s: link message version %d is not supported", name, m[0])
|
|
}
|
|
flags := m[1]
|
|
p := 2
|
|
linkType := uint8(0)
|
|
if flags&0x08 != 0 {
|
|
if p >= len(m) {
|
|
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
|
|
}
|
|
linkType = m[p]
|
|
p++
|
|
}
|
|
if flags&0x04 != 0 {
|
|
if p+8 > len(m) {
|
|
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
|
|
}
|
|
p += 8 // creation order
|
|
}
|
|
nameLenSize := 1 << (flags & 0x03) // 1, 2, 4 or 8 bytes
|
|
nameLen, ok := readUintLE(m[p:], nameLenSize)
|
|
if !ok {
|
|
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
|
|
}
|
|
p += nameLenSize
|
|
if uint64(p)+nameLen > uint64(len(m)) {
|
|
return hdf5Link{}, base.Errf("%s: link message is truncated", name)
|
|
}
|
|
linkName := string(m[p : p+int(nameLen)])
|
|
p += int(nameLen)
|
|
if linkType != 0 {
|
|
// A soft or external link: skip it, it names no object here.
|
|
return hdf5Link{name: linkName, address: math.MaxUint64}, nil
|
|
}
|
|
addr := f.at2(m[p:])
|
|
return hdf5Link{name: linkName, address: addr}, nil
|
|
}
|
|
|
|
// at2 reads an address of the file's address size from a message
|
|
// payload, or returns the undefined address.
|
|
func (f *hdf5File) at2(b []byte) uint64 {
|
|
if len(b) < f.offSize {
|
|
return math.MaxUint64
|
|
}
|
|
switch f.offSize {
|
|
case 4:
|
|
return uint64(binary.LittleEndian.Uint32(b))
|
|
case 8:
|
|
return binary.LittleEndian.Uint64(b)
|
|
}
|
|
return math.MaxUint64
|
|
}
|
|
|
|
func readUintLE(b []byte, size int) (uint64, bool) {
|
|
if len(b) < size {
|
|
return 0, false
|
|
}
|
|
switch size {
|
|
case 1:
|
|
return uint64(b[0]), true
|
|
case 2:
|
|
return uint64(binary.LittleEndian.Uint16(b)), true
|
|
case 4:
|
|
return uint64(binary.LittleEndian.Uint32(b)), true
|
|
case 8:
|
|
return binary.LittleEndian.Uint64(b), true
|
|
}
|
|
return 0, false
|
|
}
|
|
|
|
// symbolTableLinks walks the group's version 1 B-tree of symbol table
|
|
// nodes and reads the names out of the local heap.
|
|
func (f *hdf5File) symbolTableLinks(m []byte) ([]hdf5Link, error) {
|
|
const name = "LoadHDF5"
|
|
if len(m) < 2*f.offSize {
|
|
return nil, base.Errf("%s: symbol table message is truncated", name)
|
|
}
|
|
treeAddr := f.at2(m)
|
|
heapAddr := f.at2(m[f.offSize:])
|
|
heap, err := f.localHeap(heapAddr)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out := []hdf5Link{}
|
|
set := map[uint64]bool{}
|
|
if err := f.treeLinks(treeAddr, heap, set, &out, 0); err != nil {
|
|
return nil, err
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// treeLinks descends a group B-tree, reading a symbol table node at
|
|
// every leaf.
|
|
func (f *hdf5File) treeLinks(addr uint64, heap []byte, seen map[uint64]bool, out *[]hdf5Link, depth int) error {
|
|
const name = "LoadHDF5"
|
|
if depth > 32 {
|
|
return base.Errf("%s: the group B-tree is more than 32 levels deep", name)
|
|
}
|
|
if addr == math.MaxUint64 || seen[addr] {
|
|
return nil
|
|
}
|
|
seen[addr] = true
|
|
// The version 1 B-tree header is the signature, the type, the level
|
|
// and the entry count (eight bytes) plus a left and a right sibling
|
|
// address, so it is 8+2*offSize wide, not the 24 an 8-byte-address
|
|
// file happens to make it. Keying the entries off a pinned 24 read
|
|
// a 4/4 file's first key as a sibling address and lost the node.
|
|
node := f.bytes(addr, uint64(8+2*f.offSize))
|
|
if node == nil || !bytes.Equal(node[:4], hdf5Tree) {
|
|
return base.Errf("%s: no B-tree node at %d", name, addr)
|
|
}
|
|
nodeType := node[4]
|
|
level := node[5]
|
|
entries := int(binary.LittleEndian.Uint16(node[6:]))
|
|
if nodeType != 0 {
|
|
return base.Errf("%s: B-tree node type %d is not a group", name, nodeType)
|
|
}
|
|
// Each entry is a key of one address size followed by a child
|
|
// address; the node ends with one trailing key.
|
|
entrySize := f.lenSize + f.offSize
|
|
base := int(addr) + 8 + 2*f.offSize
|
|
for i := range entries {
|
|
child := f.at(base + i*entrySize + f.lenSize)
|
|
if level == 0 {
|
|
if err := f.symbolNodeLinks(child, heap, seen, out); err != nil {
|
|
return err
|
|
}
|
|
continue
|
|
}
|
|
if err := f.treeLinks(child, heap, seen, out, depth+1); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// symbolNodeLinks reads one symbol table node's entries.
|
|
func (f *hdf5File) symbolNodeLinks(addr uint64, heap []byte, seen map[uint64]bool, out *[]hdf5Link) error {
|
|
const name = "LoadHDF5"
|
|
if addr == math.MaxUint64 || seen[addr] {
|
|
return nil
|
|
}
|
|
seen[addr] = true
|
|
node := f.bytes(addr, 8)
|
|
if node == nil || !bytes.Equal(node[:4], hdf5SymbolNode) {
|
|
return base.Errf("%s: no symbol table node at %d", name, addr)
|
|
}
|
|
count := int(binary.LittleEndian.Uint16(node[6:]))
|
|
// An entry is the heap offset (of the length size), the object
|
|
// address (of the address size), the cache type, a reserved word
|
|
// and the 16-byte scratch pad: lenSize+offSize+24, not the 40 an
|
|
// 8/8 file happens to make it. A pinned 40 walked a 4/4 node's
|
|
// entries into the middle of the first one and read its second
|
|
// link's name from the wrong heap offset.
|
|
entrySize := f.lenSize + f.offSize + 24
|
|
start := int(addr) + 8
|
|
for i := range count {
|
|
off := start + i*entrySize
|
|
nameOff := f.length(off)
|
|
objAddr := f.at(off + f.lenSize)
|
|
linkName, err := heapString(heap, nameOff)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
*out = append(*out, hdf5Link{name: linkName, address: objAddr})
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// localHeap reads a local heap's data segment, the pool of
|
|
// null-terminated names the symbol table entries point into.
|
|
func (f *hdf5File) localHeap(addr uint64) ([]byte, error) {
|
|
const name = "LoadHDF5"
|
|
// The header is the signature and version, the data segment size
|
|
// and the free-list head (both of the length size), then the data
|
|
// segment address of the address size. Sizing the header with the
|
|
// address size instead (8+3*offSize) sliced the address out of a
|
|
// header that a 4/8 file stores in 20 bytes at [24:], which
|
|
// panicked; every field here is sized by its own kind.
|
|
head := f.bytes(addr, uint64(8+2*f.lenSize+f.offSize))
|
|
if head == nil || !bytes.Equal(head[:4], hdf5LocalHeap) {
|
|
return nil, base.Errf("%s: no local heap at %d", name, addr)
|
|
}
|
|
dataAddr := f.at2(head[8+2*f.lenSize:])
|
|
size := f.length(int(addr) + 8) // data segment size, then the free list head
|
|
seg := f.bytes(dataAddr, size)
|
|
if seg == nil {
|
|
return nil, base.Errf("%s: the local heap at %d lies outside the file", name, addr)
|
|
}
|
|
return seg, nil
|
|
}
|
|
|
|
// heapString reads the null-terminated string at an offset in the heap.
|
|
func heapString(heap []byte, off uint64) (string, error) {
|
|
if off >= uint64(len(heap)) {
|
|
return "", base.Errf("LoadHDF5: a symbol table entry names a name at heap offset %d, past the %d-byte heap",
|
|
off, len(heap))
|
|
}
|
|
end := bytes.IndexByte(heap[off:], 0)
|
|
if end < 0 {
|
|
return "", base.Errf("LoadHDF5: the name at heap offset %d is not terminated", off)
|
|
}
|
|
return string(heap[off : off+uint64(end)]), nil
|
|
}
|
|
|
|
// attributesOf collects the attributes carried in an object header.
|
|
func (f *hdf5File) attributesOf(msgs []hdf5Message) map[string]string {
|
|
out := map[string]string{}
|
|
for _, m := range msgs {
|
|
if m.typ != hdf5MsgAttribute {
|
|
continue
|
|
}
|
|
linkName, value, ok := f.decodeAttribute(m.data)
|
|
if ok {
|
|
out[linkName] = value
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// hdf5Type is the parsed datatype of a dataset or attribute: enough of
|
|
// it to read the values.
|
|
type hdf5Type struct {
|
|
class byte
|
|
size int
|
|
signed bool
|
|
width int // fixed-point: bytes
|
|
isFloat bool
|
|
isBool bool // a boolean enumeration: lands core.Bool
|
|
bitOrder byte
|
|
}
|
|
|
|
// hdf5ClassName names a datatype class for a refusal message: the
|
|
// number alone tells whoever wrote the file little about what the
|
|
// reader missed.
|
|
func hdf5ClassName(class byte) string {
|
|
switch class {
|
|
case 2:
|
|
return "date and time"
|
|
case 4:
|
|
return "bit field"
|
|
case 5:
|
|
return "opaque"
|
|
case 6:
|
|
return "compound"
|
|
case 7:
|
|
return "reference"
|
|
case 8:
|
|
return "enumeration"
|
|
case 9:
|
|
return "variable-length"
|
|
}
|
|
return "unknown"
|
|
}
|
|
|
|
// decodeType parses a datatype message (version 1, which is what the
|
|
// default library version writes). Strings are refused for datasets and
|
|
// accepted for attributes, where a string is exactly what is wanted.
|
|
// An enumeration is accepted only in the boolean convention HDF5
|
|
// writers carry booleans in, and lands core.Bool; every other
|
|
// enumeration and every bit field is refused by name.
|
|
func decodeType(m []byte, allowStrings bool) (hdf5Type, error) {
|
|
const name = "LoadHDF5"
|
|
if len(m) < 8 {
|
|
return hdf5Type{}, base.Errf("%s: datatype message is truncated", name)
|
|
}
|
|
// The first byte carries the version in its high nibble and the
|
|
// datatype class in its low one.
|
|
classAndVersion := m[0]
|
|
version := classAndVersion >> 4
|
|
class := classAndVersion & 0x0f
|
|
if version != 1 {
|
|
return hdf5Type{}, base.Errf("%s: datatype message version %d is not supported", name, version)
|
|
}
|
|
size := int(binary.LittleEndian.Uint32(m[4:]))
|
|
t := hdf5Type{class: class, size: size, bitOrder: m[1] >> 0}
|
|
// The class bit field's bit 0 is the byte order: 0 little-endian,
|
|
// anything else big-endian. Every element below is decoded as
|
|
// little-endian, so a big-endian datatype, on a dataset or on an
|
|
// attribute alike, would come back as byte-swapped noise with no
|
|
// error; it is refused by name instead.
|
|
if (class == 0 || class == 1) && m[1]&0x01 != 0 {
|
|
return hdf5Type{}, base.Errf("%s: big-endian datatypes are not supported; rewrite the file in the little-endian order", name)
|
|
}
|
|
switch class {
|
|
case 0: // fixed-point
|
|
if size != 1 && size != 2 && size != 4 && size != 8 {
|
|
return hdf5Type{}, base.Errf("%s: %d-byte fixed-point values are not supported", name, size)
|
|
}
|
|
t.signed = m[1]&0x08 != 0
|
|
t.width = size
|
|
case 1: // floating-point
|
|
if len(m) < 20 {
|
|
return hdf5Type{}, base.Errf("%s: floating-point datatype message is truncated", name)
|
|
}
|
|
switch size {
|
|
case 4, 8:
|
|
default:
|
|
return hdf5Type{}, base.Errf("%s: %d-byte floating-point values are not supported", name, size)
|
|
}
|
|
t.isFloat = true
|
|
t.width = size
|
|
case 3: // string
|
|
if !allowStrings {
|
|
return hdf5Type{}, base.Errf("%s: string datasets are not supported (only numeric arrays are read)", name)
|
|
}
|
|
t.width = size
|
|
if t.width <= 0 {
|
|
return hdf5Type{}, base.Errf("%s: a string attribute declares %d bytes", name, size)
|
|
}
|
|
case 8: // enumeration: only the boolean convention is read
|
|
if err := hdf5EnumBool(m); err != nil {
|
|
return hdf5Type{}, err
|
|
}
|
|
t.isBool = true
|
|
t.width = size
|
|
case 9: // variable-length: the text lives in a global heap collection
|
|
if !allowStrings {
|
|
return hdf5Type{}, base.Errf("%s: variable-length datasets are not supported (only numeric arrays are read)", name)
|
|
}
|
|
t.class = 9
|
|
t.width = size // the descriptor: length, heap address, object index
|
|
default:
|
|
return hdf5Type{}, base.Errf("%s: datatype class %d (%s) is not supported", name, class, hdf5ClassName(class))
|
|
}
|
|
return t, nil
|
|
}
|
|
|
|
// hdf5EnumBool verifies that an enumeration datatype message carries
|
|
// the boolean convention HDF5 writers store booleans in: a one-byte
|
|
// unsigned base type whose member values are a subset of {0, 1}. The
|
|
// member names are irrelevant to the values, which is what the landing
|
|
// keys on, and any other enumeration is refused by name.
|
|
//
|
|
// The message layout comes from the HDF5 file format specification's
|
|
// datatype message, enumeration class: the class bit field carries the
|
|
// member count in its low sixteen bits, and the properties hold the
|
|
// base type as a complete datatype message, then the member names
|
|
// (each a NUL-terminated string stored in a multiple of eight bytes,
|
|
// padded from its own field's start) and the packed member values. The
|
|
// base type here is a complete version 1 fixed-point message: its own
|
|
// eight-byte header plus the bit offset and bit precision the
|
|
// specification's fixed-point property table defines, twelve bytes in
|
|
// total, the same shape the reader already requires of the twenty-byte
|
|
// version 1 floating-point message.
|
|
func hdf5EnumBool(m []byte) error {
|
|
const name = "LoadHDF5"
|
|
members := int(binary.LittleEndian.Uint16(m[1:]))
|
|
if m[3] != 0 {
|
|
return base.Errf("%s: enumeration datatype class 8 carries unknown bit field bits %#02x", name, m[3])
|
|
}
|
|
if members < 1 {
|
|
return base.Errf("%s: enumeration datatype class 8 declares %d members", name, members)
|
|
}
|
|
if size := binary.LittleEndian.Uint32(m[4:]); size != 1 {
|
|
return base.Errf("%s: enumeration datatype class 8 with %d-byte values is not supported; only the one-byte unsigned boolean convention is read", name, size)
|
|
}
|
|
// The base type: a complete version 1 fixed-point message.
|
|
if len(m) < 8+12 {
|
|
return base.Errf("%s: enumeration datatype message is truncated", name)
|
|
}
|
|
b := m[8:]
|
|
if bclass := b[0] & 0x0f; b[0]>>4 != 1 || bclass != 0 {
|
|
return base.Errf("%s: enumeration datatype class 8 with a base type of class %d version %d is not supported; only the one-byte unsigned boolean convention is read",
|
|
name, bclass, b[0]>>4)
|
|
}
|
|
if b[1]&0x01 != 0 {
|
|
return base.Errf("%s: big-endian datatypes are not supported; rewrite the file in the little-endian order", name)
|
|
}
|
|
if b[1]&0x08 != 0 {
|
|
return base.Errf("%s: enumeration datatype class 8 with a signed base type is not supported; only the one-byte unsigned boolean convention is read", name)
|
|
}
|
|
if bsize := binary.LittleEndian.Uint32(b[4:]); bsize != 1 {
|
|
return base.Errf("%s: enumeration datatype class 8 with a base type of %d bytes is not supported; only the one-byte unsigned boolean convention is read", name, bsize)
|
|
}
|
|
// The names: each field is a NUL-terminated string padded, from its
|
|
// own start, to a multiple of eight bytes. The walk only needs the
|
|
// fields' extent; the values follow them.
|
|
p := 8 + 12
|
|
for i := range members {
|
|
end := -1
|
|
for j := p; j < len(m); j++ {
|
|
if m[j] == 0 {
|
|
end = j
|
|
break
|
|
}
|
|
}
|
|
if end < 0 {
|
|
return base.Errf("%s: enumeration datatype message ends inside member name %d", name, i+1)
|
|
}
|
|
p += alignUp(end-p+1, 8)
|
|
if p > len(m) {
|
|
return base.Errf("%s: enumeration datatype message ends inside the member names", name)
|
|
}
|
|
}
|
|
// The values: packed, one base-width byte each.
|
|
if members > len(m)-p {
|
|
return base.Errf("%s: enumeration datatype message holds %d member values, fewer than its %d members", name, len(m)-p, members)
|
|
}
|
|
for i := range members {
|
|
if v := m[p+i]; v > 1 {
|
|
return base.Errf("%s: enumeration datatype class 8 declares member value %d, outside the boolean convention {0, 1}", name, v)
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// decodeDataspace parses a dataspace message into the dataset's shape.
|
|
// Version 1 is what the default library version writes, version 2 what
|
|
// the "latest" one writes: it drops the reserved bytes to one and
|
|
// always stores the dimensions in eight bytes.
|
|
func decodeDataspace(m []byte, lenSize int) ([]int, error) {
|
|
const name = "LoadHDF5"
|
|
if len(m) < 8 {
|
|
return nil, base.Errf("%s: dataspace message is truncated", name)
|
|
}
|
|
version := m[0]
|
|
rank := int(m[1])
|
|
flags := m[2]
|
|
if version != 1 && version != 2 {
|
|
return nil, base.Errf("%s: dataspace message version %d is not supported", name, version)
|
|
}
|
|
if rank == 0 {
|
|
return []int{1}, nil // a scalar: one element, no dimensions
|
|
}
|
|
pos, dimSize := 8, lenSize
|
|
if version == 2 {
|
|
pos, dimSize = 4, 8
|
|
}
|
|
shape := make([]int, rank)
|
|
for i := range rank {
|
|
if pos+dimSize > len(m) {
|
|
return nil, base.Errf("%s: dataspace message is truncated", name)
|
|
}
|
|
shape[i] = int(uint64At(m[pos:], dimSize))
|
|
pos += dimSize
|
|
}
|
|
if flags&0x02 != 0 {
|
|
return nil, base.Errf("%s: datasets with a dimension permutation are not supported", name)
|
|
}
|
|
if flags&0x01 != 0 {
|
|
// Maximum dimensions are not the shape; nothing to do with them.
|
|
pos += rank * dimSize
|
|
}
|
|
return shape, nil
|
|
}
|
|
|
|
func uint64At(b []byte, size int) uint64 {
|
|
v, _ := readUintLE(b, size)
|
|
return v
|
|
}
|
|
|
|
// hdf5Layout is a dataset's storage description. For chunked storage
|
|
// the message records one more dimension than the dataset has: the
|
|
// dimensions are the chunk's shape followed by the element size in
|
|
// bytes, and the chunk B-tree keys carry the same extra offset.
|
|
type hdf5Layout struct {
|
|
class byte
|
|
addr uint64
|
|
size uint64
|
|
compact []byte
|
|
dims []int
|
|
}
|
|
|
|
func decodeLayout(m []byte, offSize int) (hdf5Layout, error) {
|
|
const name = "LoadHDF5"
|
|
if len(m) < 2 {
|
|
return hdf5Layout{}, base.Errf("%s: data layout message is truncated", name)
|
|
}
|
|
version := m[0]
|
|
class := m[1]
|
|
switch version {
|
|
case 3, 4:
|
|
default:
|
|
return hdf5Layout{}, base.Errf("%s: data layout message version %d is not supported", name, version)
|
|
}
|
|
l := hdf5Layout{class: class}
|
|
switch class {
|
|
case 0: // compact: the data lives in the message
|
|
if len(m) < 4 {
|
|
return hdf5Layout{}, base.Errf("%s: compact layout message is truncated", name)
|
|
}
|
|
size := int(binary.LittleEndian.Uint16(m[2:]))
|
|
if 4+size > len(m) {
|
|
return hdf5Layout{}, base.Errf("%s: compact data runs past the message", name)
|
|
}
|
|
l.compact = m[4 : 4+size]
|
|
case 1: // contiguous
|
|
if len(m) < 2+offSize+8 {
|
|
return hdf5Layout{}, base.Errf("%s: contiguous layout message is truncated", name)
|
|
}
|
|
l.addr = uint64At(m[2:], offSize)
|
|
l.size = uint64At(m[2+offSize:], 8)
|
|
case 2: // chunked
|
|
if len(m) < 3+offSize {
|
|
return hdf5Layout{}, base.Errf("%s: chunked layout message is truncated", name)
|
|
}
|
|
rank := int(m[2])
|
|
p := 3
|
|
l.addr = uint64At(m[p:], offSize)
|
|
p += offSize
|
|
// Version 4 stores one flag byte before the chunk dimensions
|
|
// and widens them to 8 bytes when the flag asks for it.
|
|
dimSize := 4
|
|
if version == 4 {
|
|
if p >= len(m) {
|
|
return hdf5Layout{}, base.Errf("%s: chunked layout message is truncated", name)
|
|
}
|
|
if m[p]&0x01 != 0 {
|
|
dimSize = 8
|
|
}
|
|
p++
|
|
}
|
|
l.dims = make([]int, rank)
|
|
for i := range rank {
|
|
if p+dimSize > len(m) {
|
|
return hdf5Layout{}, base.Errf("%s: chunk dimensions run past the message", name)
|
|
}
|
|
l.dims[i] = int(uint64At(m[p:], dimSize))
|
|
p += dimSize
|
|
}
|
|
default:
|
|
return hdf5Layout{}, base.Errf("%s: storage class %d is not supported", name, class)
|
|
}
|
|
return l, nil
|
|
}
|
|
|
|
// hdf5Filter is one entry of a filter pipeline.
|
|
type hdf5Filter struct {
|
|
id uint16
|
|
values []uint32
|
|
}
|
|
|
|
func decodeFilters(m []byte) ([]hdf5Filter, error) {
|
|
const name = "LoadHDF5"
|
|
if len(m) < 2 {
|
|
return nil, base.Errf("%s: filter pipeline message is truncated", name)
|
|
}
|
|
version := m[0]
|
|
if version != 1 {
|
|
// Only version 1's layout is known; parsing another version on
|
|
// this layout would fail far away with an unrelated truncation
|
|
// error instead of naming the defect.
|
|
return nil, base.Errf("%s: unsupported filter pipeline message version %d", name, version)
|
|
}
|
|
count := int(m[1])
|
|
p := 2
|
|
p += 6 // reserved
|
|
out := make([]hdf5Filter, 0, count)
|
|
for range count {
|
|
if p+8 > len(m) {
|
|
return nil, base.Errf("%s: filter pipeline message is truncated", name)
|
|
}
|
|
filter := hdf5Filter{id: binary.LittleEndian.Uint16(m[p:])}
|
|
nameLen := int(binary.LittleEndian.Uint16(m[p+2:]))
|
|
nValues := int(binary.LittleEndian.Uint16(m[p+6:]))
|
|
p += 8
|
|
if p+nameLen > len(m) {
|
|
return nil, base.Errf("%s: filter name runs past the message", name)
|
|
}
|
|
p += nameLen
|
|
p = alignUp(p, 8)
|
|
for range nValues {
|
|
if p+4 > len(m) {
|
|
return nil, base.Errf("%s: filter values run past the message", name)
|
|
}
|
|
filter.values = append(filter.values, binary.LittleEndian.Uint32(m[p:]))
|
|
p += 4
|
|
}
|
|
p = alignUp(p, 8)
|
|
switch filter.id {
|
|
case hdf5FilterDeflate, hdf5FilterShuffle, hdf5FilterFletcher32:
|
|
default:
|
|
return nil, base.Errf("%s: filter %d is not supported", name, filter.id)
|
|
}
|
|
out = append(out, filter)
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// dataset reads one dataset's values.
|
|
func (f *hdf5File) dataset(path string, msgs []hdf5Message, st *hdf5WalkState) (HDF5Dataset, error) {
|
|
const name = "LoadHDF5"
|
|
var (
|
|
shape []int
|
|
dtype hdf5Type
|
|
layout hdf5Layout
|
|
filters []hdf5Filter
|
|
haveType bool
|
|
haveSpace bool
|
|
haveLay bool
|
|
)
|
|
for _, m := range msgs {
|
|
switch m.typ {
|
|
case hdf5MsgDataspace:
|
|
s, err := decodeDataspace(m.data, f.lenSize)
|
|
if err != nil {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
|
|
}
|
|
shape, haveSpace = s, true
|
|
case hdf5MsgDatatype:
|
|
t, err := decodeType(m.data, false)
|
|
if err != nil {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
|
|
}
|
|
dtype, haveType = t, true
|
|
case hdf5MsgDataLayout:
|
|
l, err := decodeLayout(m.data, f.offSize)
|
|
if err != nil {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
|
|
}
|
|
layout, haveLay = l, true
|
|
case hdf5MsgFilterPipeline:
|
|
fs, err := decodeFilters(m.data)
|
|
if err != nil {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err)
|
|
}
|
|
filters = fs
|
|
}
|
|
}
|
|
if !haveSpace || !haveType || !haveLay {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q is missing its %s", name, path,
|
|
missingOf(haveSpace, haveType, haveLay))
|
|
}
|
|
if layout.class == 2 {
|
|
// A latest-version file indexes its chunks with the version 2
|
|
// B-tree, which this reader does not walk; a classic file's
|
|
// chunks hang off the version 1 tree below.
|
|
if f.superblock >= 2 {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q is chunked through a version 2 B-tree, which is not supported; rewrite the file with the default library version", name, path)
|
|
}
|
|
// The chunked message's last dimension is the element size, not
|
|
// a chunk dimension: the rank must be one less, and the value
|
|
// must agree with the datatype, which is a strong check that
|
|
// both were read correctly.
|
|
if len(layout.dims) != len(shape)+1 {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d chunk dimensions for rank %d",
|
|
name, path, len(layout.dims), len(shape))
|
|
}
|
|
if got := layout.dims[len(layout.dims)-1]; got != dtype.width {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d-byte chunk elements against a %d-byte datatype",
|
|
name, path, got, dtype.width)
|
|
}
|
|
layout.dims = layout.dims[:len(shape)]
|
|
}
|
|
// Fixed-point data lands by its stored width and signedness; an
|
|
// unsigned 64-bit value has no exact core dtype and is refused
|
|
// before any storage is read, with the same message the value
|
|
// decode reported it with.
|
|
if !dtype.isFloat && !dtype.signed && dtype.width == 8 {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path,
|
|
base.Errf("unsigned 64-bit integers have no exact core dtype"))
|
|
}
|
|
// The size arithmetic bounds every extent against the budget before
|
|
// it is multiplied: a hostile dataspace must fail the cap, not wrap
|
|
// the product into a small want that passes it. The validated byte
|
|
// count is what every storage class below works from.
|
|
want, err := hdf5ByteExtent(shape, dtype.width, hdf5MaxDatasetBytes)
|
|
if err != nil {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w; larger datasets need mmap", name, path, err)
|
|
}
|
|
n := int(want)
|
|
// The per-dataset cap bounds one dataset; the walk's aggregate
|
|
// bounds their sum, so a file declaring thousands of cap-fitting
|
|
// datasets refuses here instead of allocating the sum. The sum is
|
|
// charged past the value bytes: every accepted dataset also costs
|
|
// its result structure and its copied path, which the envelope and
|
|
// the path length stand for, so empty datasets pay too.
|
|
charge := int64(n) + int64(len(st.path)) + hdf5DatasetEnvelope
|
|
if charge > st.budget {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d bytes, the file's datasets are past the %d byte read budget; larger files need mmap",
|
|
name, path, n, hdf5MaxDatasetBytes)
|
|
}
|
|
st.budget -= charge
|
|
if layout.class == 2 {
|
|
values, cerr := f.chunkedArray(path, shape, dtype, layout, filters, n)
|
|
if cerr != nil {
|
|
return HDF5Dataset{}, cerr
|
|
}
|
|
return HDF5Dataset{Path: path, Shape: shape, Values: values}, nil
|
|
}
|
|
raw, rerr := f.rawData(path, shape, dtype, layout, n)
|
|
if rerr != nil {
|
|
return HDF5Dataset{}, rerr
|
|
}
|
|
values, aerr := arrayFromRaw(raw, dtype, shape)
|
|
if aerr != nil {
|
|
return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, aerr)
|
|
}
|
|
return HDF5Dataset{Path: path, Shape: shape, Values: values}, nil
|
|
}
|
|
|
|
func missingOf(haveSpace, haveType, haveLay bool) string {
|
|
switch {
|
|
case !haveSpace:
|
|
return "dataspace"
|
|
case !haveType:
|
|
return "datatype"
|
|
case !haveLay:
|
|
return "storage layout"
|
|
}
|
|
return "messages"
|
|
}
|
|
|
|
// rawData gathers a dataset's bytes for the storage classes that hold
|
|
// them contiguously: compact storage is the message itself and
|
|
// contiguous storage is a span of the file. Chunked storage decodes
|
|
// into the values array directly, in chunkedArray. n is the dataset's
|
|
// byte size, validated against the read budget by the caller.
|
|
func (f *hdf5File) rawData(path string, shape []int, dtype hdf5Type, layout hdf5Layout, n int) ([]byte, error) {
|
|
const name = "LoadHDF5"
|
|
switch layout.class {
|
|
case 0:
|
|
if len(layout.compact) < n {
|
|
return nil, base.Errf("%s: dataset %q holds %d compact bytes, %d are needed",
|
|
name, path, len(layout.compact), n)
|
|
}
|
|
return layout.compact[:n], nil
|
|
case 1:
|
|
if layout.addr == math.MaxUint64 {
|
|
// The undefined address means storage was never allocated:
|
|
// for an empty dataset that is the whole answer (the fill
|
|
// value is the data, as the chunked branch already models),
|
|
// and for a non-empty one the bytes simply do not exist.
|
|
if n == 0 {
|
|
return []byte{}, nil
|
|
}
|
|
return nil, base.Errf("%s: dataset %q declares storage that was never allocated", name, path)
|
|
}
|
|
raw := f.bytes(layout.addr, uint64(n))
|
|
if raw == nil {
|
|
return nil, base.Errf("%s: dataset %q lies outside the file", name, path)
|
|
}
|
|
return raw, nil
|
|
}
|
|
return nil, base.Errf("%s: dataset %q has an unsupported storage class", name, path)
|
|
}
|
|
|
|
// hdf5ChunkStage carries what one dataset's chunk decode reuses across
|
|
// its chunks: the inflate output, the unshuffle staging, the inflate
|
|
// reader over the file's bytes, its limited view, and the index
|
|
// vectors the placement walk uses. A decode is serial, so the stage
|
|
// belongs to one call and nothing here is shared between concurrent
|
|
// loads.
|
|
type hdf5ChunkStage struct {
|
|
inflate []byte
|
|
unshuf []byte
|
|
zr io.ReadCloser
|
|
zrReset zlib.Resetter
|
|
br bytes.Reader
|
|
lim io.LimitedReader
|
|
offsets []uint64
|
|
vec []int
|
|
seen map[uint64]bool
|
|
}
|
|
|
|
// chunkedArray reassembles a chunked dataset through its version 1
|
|
// B-tree, running each chunk back through the filter pipeline and
|
|
// decoding each cell into the values array as it lands. bytes is the
|
|
// dataset's size, already validated against the read budget by the
|
|
// caller, so this function never re-derives it in int. The staging
|
|
// buffers, the inflate reader and the index vectors are one stage,
|
|
// reused across every chunk of the dataset.
|
|
func (f *hdf5File) chunkedArray(path string, shape []int, dtype hdf5Type, layout hdf5Layout, filters []hdf5Filter, bytes int) (*core.Array, error) {
|
|
const name = "LoadHDF5"
|
|
rank := len(shape)
|
|
if len(layout.dims) != rank {
|
|
return nil, base.Errf("%s: dataset %q declares %d chunk dimensions for rank %d", name, path, len(layout.dims), rank)
|
|
}
|
|
for _, c := range layout.dims {
|
|
// A chunk holds at least one element of the datatype; a zero or
|
|
// negative declared dimension is a hostile header, and both the
|
|
// strides below and the placement's overlap box degenerate on it.
|
|
if c <= 0 {
|
|
return nil, base.Errf("%s: dataset %q declares a %d-element chunk dimension", name, path, c)
|
|
}
|
|
}
|
|
// One chunk may not exceed the reader budget on its own, because the
|
|
// filters inflate into a buffer of exactly this size. The product is
|
|
// bounded extent by extent, so a hostile chunk dimension fails the
|
|
// cap instead of wrapping it.
|
|
chunkBytes, err := hdf5ByteExtent(layout.dims, dtype.width, hdf5MaxDatasetBytes)
|
|
if err != nil {
|
|
return nil, base.Errf("%s: dataset %q declares a chunk: %w", name, path, err)
|
|
}
|
|
cb := int(chunkBytes)
|
|
// The destination array and the per-cell decode carry the value
|
|
// mapping arrayFromRaw applies to contiguous bytes; the cells land
|
|
// in the payload as the walk places them, so no assembled copy of
|
|
// the dataset's bytes exists. Fixed-point cells land by stored width
|
|
// and signedness, exactly as the contiguous path lands them, and a
|
|
// boolean cell outside the enumeration's {0, 1} values is refused
|
|
// rather than coerced.
|
|
nElems := 0
|
|
if dtype.width > 0 {
|
|
nElems = bytes / dtype.width
|
|
}
|
|
var put func(data []byte, ci, oi int) error
|
|
var arr *core.Array
|
|
switch {
|
|
case dtype.isFloat && dtype.width == 8:
|
|
vals := make([]float64, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = math.Float64frombits(binary.LittleEndian.Uint64(data[ci*8:]))
|
|
return nil
|
|
}
|
|
arr, err = core.FloatsFromArray(vals, shape...)
|
|
case dtype.isFloat && dtype.width == 4:
|
|
vals := make([]float32, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = math.Float32frombits(binary.LittleEndian.Uint32(data[ci*4:]))
|
|
return nil
|
|
}
|
|
// The aliasing constructor: the decode fills the array's own
|
|
// payload through vals, which a copying constructor would have
|
|
// left behind as a separate slice.
|
|
arr, err = core.FromFloat32Slice(vals, shape...)
|
|
case dtype.isBool:
|
|
vals := make([]bool, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
b := data[ci]
|
|
if b > 1 {
|
|
return base.Errf("a boolean value of %d is outside the members 0 and 1", b)
|
|
}
|
|
vals[oi] = b != 0
|
|
return nil
|
|
}
|
|
arr, err = core.BoolsFromArray(vals, shape...)
|
|
case !dtype.signed && dtype.width == 8:
|
|
// Unreachable through LoadHDF5, which refuses the datatype
|
|
// first; the decode carries the same refusal for direct callers.
|
|
return nil, base.Errf("%s: dataset %q: %w", name, path,
|
|
base.Errf("unsigned 64-bit integers have no exact core dtype"))
|
|
case dtype.width == 1 && dtype.signed:
|
|
vals := make([]int8, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = int8(data[ci])
|
|
return nil
|
|
}
|
|
arr, err = core.Int8sFromArray(vals, shape...)
|
|
case dtype.width == 1:
|
|
vals := make([]uint8, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = data[ci]
|
|
return nil
|
|
}
|
|
arr, err = core.Uint8sFromArray(vals, shape...)
|
|
case dtype.width == 2 && dtype.signed:
|
|
vals := make([]int16, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = int16(binary.LittleEndian.Uint16(data[ci*2:]))
|
|
return nil
|
|
}
|
|
arr, err = core.Int16sFromArray(vals, shape...)
|
|
case dtype.width == 2:
|
|
vals := make([]uint16, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = binary.LittleEndian.Uint16(data[ci*2:])
|
|
return nil
|
|
}
|
|
arr, err = core.Uint16sFromArray(vals, shape...)
|
|
case dtype.width == 4 && dtype.signed:
|
|
vals := make([]int32, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = int32(binary.LittleEndian.Uint32(data[ci*4:]))
|
|
return nil
|
|
}
|
|
arr, err = core.Int32sFromArray(vals, shape...)
|
|
case dtype.width == 4:
|
|
vals := make([]uint32, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = binary.LittleEndian.Uint32(data[ci*4:])
|
|
return nil
|
|
}
|
|
arr, err = core.Uint32sFromArray(vals, shape...)
|
|
default:
|
|
// Signed 64-bit, the only fixed-point width left.
|
|
vals := make([]int64, nElems)
|
|
put = func(data []byte, ci, oi int) error {
|
|
vals[oi] = int64(binary.LittleEndian.Uint64(data[ci*8:]))
|
|
return nil
|
|
}
|
|
arr, err = core.IntsFromArray(vals, shape...)
|
|
}
|
|
if err != nil {
|
|
return nil, base.Errf("%s: dataset %q: %w", name, path, err)
|
|
}
|
|
if layout.addr == math.MaxUint64 {
|
|
return arr, nil // no chunks written: the fill value is the data
|
|
}
|
|
stage := &hdf5ChunkStage{
|
|
seen: map[uint64]bool{},
|
|
offsets: make([]uint64, rank),
|
|
vec: make([]int, 5*rank),
|
|
}
|
|
seenNodes := map[uint64]bool{}
|
|
return arr, f.chunkTreeWalk(layout.addr, layout, rank, seenNodes, func(addr uint64, offsets []uint64, mask uint32, size int) error {
|
|
if stage.seen[addr] {
|
|
return base.Errf("%s: dataset %q has two chunks at the same address", name, path)
|
|
}
|
|
stage.seen[addr] = true
|
|
stored := f.bytes(addr, uint64(size))
|
|
if stored == nil {
|
|
return base.Errf("%s: dataset %q has a chunk outside the file", name, path)
|
|
}
|
|
data, ferr := runFilters(stored, filters, mask, dtype, cb, stage)
|
|
if ferr != nil {
|
|
return base.Errf("%s: dataset %q: %w", name, path, ferr)
|
|
}
|
|
return placeCells(shape, layout.dims, offsets, data, dtype.width, cb, nElems, stage.vec, put)
|
|
}, 0, stage.offsets)
|
|
}
|
|
|
|
// placeCells walks the overlap of one chunk and the dataset, calling
|
|
// put for every cell with the chunk-local cell index and the output
|
|
// element index; the decode writes the value as the cell lands, and a
|
|
// cell put refuses stops the walk with that error.
|
|
// chunkBytes is the byte size of one complete chunk, computed by the
|
|
// caller with every dimension bounded against the reader budget, so
|
|
// the strides cannot wrap and a chunk is known to hold whole elements.
|
|
// The loop walks the overlap box in output coordinates, so the work is
|
|
// proportional to the placed elements: a chunk hanging past the
|
|
// shape's edge is padding the file may hold but the array has no room
|
|
// for, and one starting past an axis's end places nothing at all.
|
|
// Walking the whole declared chunk instead would let a hostile chunk
|
|
// dimension pin the reader for hours without touching a byte. vec
|
|
// carries the box bounds, the row-major strides of chunk and output,
|
|
// and the counter: five vectors of the rank, reused across chunks.
|
|
func placeCells(shape, chunkDim []int, offsets []uint64, data []byte, width, chunkBytes, maxOut int, vec []int, put func(data []byte, ci, oi int) error) error {
|
|
const name = "LoadHDF5"
|
|
if len(data) == 0 {
|
|
return nil
|
|
}
|
|
rank := len(shape)
|
|
if len(offsets) != rank {
|
|
return base.Errf("%s: a chunk carries %d offsets for rank %d", name, len(offsets), rank)
|
|
}
|
|
// The chunk must arrive complete: a truncated deflate stream would
|
|
// otherwise read past its data below.
|
|
if len(data) < chunkBytes {
|
|
return base.Errf("%s: a chunk holds %d bytes, a full chunk of %d is needed", name, len(data), chunkBytes)
|
|
}
|
|
lo := vec[:rank]
|
|
hi := vec[rank : 2*rank]
|
|
chunkStride := vec[2*rank : 3*rank]
|
|
outStride := vec[3*rank : 4*rank]
|
|
count := vec[4*rank : 5*rank]
|
|
for d := range rank {
|
|
if offsets[d] >= uint64(shape[d]) {
|
|
return nil
|
|
}
|
|
lo[d] = int(offsets[d])
|
|
hi[d] = min(lo[d]+chunkDim[d], shape[d])
|
|
if lo[d] == hi[d] {
|
|
return nil
|
|
}
|
|
}
|
|
// Row-major strides of the chunk and of the output. Both products
|
|
// are bounded: the chunk's by chunkBytes and the array's by the
|
|
// dataset's own validated byte count.
|
|
cs, os := 1, 1
|
|
for i := rank - 1; i >= 0; i-- {
|
|
chunkStride[i], outStride[i] = cs, os
|
|
cs *= chunkDim[i]
|
|
os *= shape[i]
|
|
}
|
|
// The element walk indexes both sides with a computed offset. The
|
|
// chunk's data may be a view of the file buffer, whose capacity
|
|
// runs past the chunk, so a stride that reached outside the chunk
|
|
// would read the file's other bytes silently instead of failing:
|
|
// the two limits are therefore checked explicitly, once, in element
|
|
// units.
|
|
maxChunk := len(data) / width
|
|
copy(count, lo)
|
|
for {
|
|
// The element's output coordinate and chunk-local coordinate.
|
|
oi, ci := 0, 0
|
|
for d := range rank {
|
|
oi += count[d] * outStride[d]
|
|
ci += (count[d] - lo[d]) * chunkStride[d]
|
|
}
|
|
if oi < 0 || oi >= maxOut || ci < 0 || ci >= maxChunk {
|
|
return base.Errf("%s: a chunk element lands at element %d of the %d-element array and %d of the %d-element chunk",
|
|
name, oi, maxOut, ci, maxChunk)
|
|
}
|
|
if err := put(data, ci, oi); err != nil {
|
|
return err
|
|
}
|
|
// Advance the odometer over the overlap box.
|
|
d := rank - 1
|
|
for d >= 0 {
|
|
count[d]++
|
|
if count[d] < hi[d] {
|
|
break
|
|
}
|
|
count[d] = lo[d]
|
|
d--
|
|
}
|
|
if d < 0 {
|
|
return nil
|
|
}
|
|
}
|
|
}
|
|
|
|
// chunkTree walks the chunk B-tree, calling place for every chunk,
|
|
// with a fresh per-entry offset vector. The decode reuses one vector
|
|
// across the whole walk through chunkTreeWalk instead.
|
|
func (f *hdf5File) chunkTree(addr uint64, layout hdf5Layout, rank int, nodes map[uint64]bool, place func(uint64, []uint64, uint32, int) error, depth int) error {
|
|
return f.chunkTreeWalk(addr, layout, rank, nodes, place, depth, make([]uint64, rank))
|
|
}
|
|
|
|
// chunkTreeWalk is the walk itself, with the per-entry offset vector
|
|
// supplied by the caller, so a decode allocates one for the dataset
|
|
// rather than one per chunk entry; a place callback consumes the
|
|
// offsets before it returns. nodes carries the addresses of the
|
|
// interior nodes already visited: a legal B-tree never revisits one,
|
|
// and without the set a hostile file whose node lists itself among its
|
|
// children would multiply the walk into entries^depth visits before
|
|
// the depth guard ever fires.
|
|
func (f *hdf5File) chunkTreeWalk(addr uint64, layout hdf5Layout, rank int, nodes map[uint64]bool, place func(uint64, []uint64, uint32, int) error, depth int, offsets []uint64) error {
|
|
const name = "LoadHDF5"
|
|
if depth > 32 {
|
|
return base.Errf("%s: the chunk B-tree is more than 32 levels deep", name)
|
|
}
|
|
if nodes[addr] {
|
|
return base.Errf("%s: the chunk B-tree revisits node %d", name, addr)
|
|
}
|
|
nodes[addr] = true
|
|
// The version 1 B-tree header is 8+2*offSize wide, the signature,
|
|
// type, level and entry count plus the two sibling addresses.
|
|
node := f.bytes(addr, uint64(8+2*f.offSize))
|
|
if node == nil || !bytes.Equal(node[:4], hdf5Tree) {
|
|
return base.Errf("%s: no chunk B-tree node at %d", name, addr)
|
|
}
|
|
if node[4] != 1 {
|
|
return base.Errf("%s: B-tree node type %d is not a chunk tree", name, node[4])
|
|
}
|
|
level := node[5]
|
|
entries := int(binary.LittleEndian.Uint16(node[6:]))
|
|
// The node header is 8+2*offSize wide, as in the group B-tree; the
|
|
// first key starts behind it.
|
|
nodeHeader := 8 + 2*f.offSize
|
|
// A v1 chunk key is the chunk size, the filter mask and one offset
|
|
// per dimension of the chunk *plus* the element size, which is why
|
|
// the key has one more offset than the dataset has axes; each entry
|
|
// is a key plus a child address.
|
|
keySize := 8 + 8*(rank+1)
|
|
entrySize := keySize + f.offSize
|
|
start := int(addr) + nodeHeader
|
|
for i := range entries {
|
|
p := start + i*entrySize
|
|
size := int(f.u64At(p, 4))
|
|
mask := uint32(f.u64At(p+4, 4))
|
|
for d := range rank {
|
|
offsets[d] = f.length(p + 8 + d*8)
|
|
}
|
|
child := f.at(p + keySize)
|
|
if level == 0 {
|
|
if err := place(child, offsets, mask, size); err != nil {
|
|
return err
|
|
}
|
|
continue
|
|
}
|
|
if err := f.chunkTreeWalk(child, layout, rank, nodes, place, depth+1, offsets); err != nil {
|
|
return err
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// runFilters undoes the filter pipeline for one chunk, in reverse
|
|
// order; a filter whose bit is set in the mask did not run on it. The
|
|
// returned bytes are either the stored chunk untouched or one of the
|
|
// stage's buffers, valid until the stage handles its next chunk.
|
|
func runFilters(data []byte, filters []hdf5Filter, mask uint32, dtype hdf5Type, chunkBytes int, stage *hdf5ChunkStage) ([]byte, error) {
|
|
const name = "LoadHDF5"
|
|
out := data
|
|
for i, filter := range slices.Backward(filters) {
|
|
if mask&(1<<uint(i)) != 0 {
|
|
continue
|
|
}
|
|
switch filter.id {
|
|
case hdf5FilterFletcher32:
|
|
if len(out) < 4 {
|
|
return nil, base.Errf("%s: a fletcher32 chunk is shorter than its checksum", name)
|
|
}
|
|
if err := checkFletcher32(out); err != nil {
|
|
return nil, err
|
|
}
|
|
out = out[:len(out)-4]
|
|
case hdf5FilterShuffle:
|
|
if dtype.width <= 1 {
|
|
// One byte per element: the transpose is the identity.
|
|
continue
|
|
}
|
|
if cap(stage.unshuf) < len(out) {
|
|
stage.unshuf = make([]byte, len(out))
|
|
}
|
|
dst := stage.unshuf[:len(out)]
|
|
unshuffleInto(dst, out, dtype.width)
|
|
out = dst
|
|
case hdf5FilterDeflate:
|
|
// The filter's value is the compression level the writer
|
|
// used, which inflate does not need.
|
|
inflated, err := inflate(out, chunkBytes, stage)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
out = inflated
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// inflate decompresses a deflate chunk into the stage's reused output
|
|
// buffer, refusing to grow past the chunk's own size. The filter
|
|
// stores a zlib stream, not a raw deflate one: the two-byte zlib
|
|
// header is present in the file. The reader is created once per
|
|
// dataset and reset over each next chunk's bytes; Reset re-initialises
|
|
// it to the state a fresh reader holds.
|
|
func inflate(data []byte, want int, stage *hdf5ChunkStage) ([]byte, error) {
|
|
stage.br.Reset(data)
|
|
if stage.zr == nil {
|
|
zr, zerr := zlib.NewReader(&stage.br)
|
|
if zerr != nil {
|
|
return nil, base.Errf("the deflate stream is corrupt: %w", zerr)
|
|
}
|
|
stage.zr, stage.zrReset = zr, zr.(zlib.Resetter)
|
|
} else if rerr := stage.zrReset.Reset(&stage.br, nil); rerr != nil {
|
|
return nil, base.Errf("the deflate stream is corrupt: %w", rerr)
|
|
}
|
|
// The chunk's size is known, so the output is sized once instead of
|
|
// grown from a small first guess. One byte past the chunk is what
|
|
// separates a stream that fills it from one that overflows it.
|
|
if cap(stage.inflate) < want+1 {
|
|
stage.inflate = make([]byte, 0, want+1)
|
|
}
|
|
out := stage.inflate[:0]
|
|
stage.lim.R = stage.zr
|
|
stage.lim.N = int64(want) + 1
|
|
for len(out) < cap(out) {
|
|
n, rerr := stage.lim.Read(out[len(out):cap(out)])
|
|
out = out[:len(out)+n]
|
|
if rerr != nil {
|
|
if rerr != io.EOF {
|
|
return nil, base.Errf("the deflate stream is corrupt: %w", rerr)
|
|
}
|
|
break
|
|
}
|
|
}
|
|
if len(out) > want {
|
|
return nil, base.Errf("the deflate stream inflates past the chunk size")
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// unshuffleInto undoes the shuffle filter into out, which holds the
|
|
// same length as data: the filter transposes the bytes of the
|
|
// elements, so the first block holds every element's first byte. The
|
|
// transpose runs over tiles of rows: a tile is a short contiguous run
|
|
// of out, and the reads that fill it walk one row of the source
|
|
// sequentially, so neither stream strides across the whole chunk per
|
|
// column. Every byte still lands where the flat transpose put it.
|
|
func unshuffleInto(out, data []byte, width int) {
|
|
n := len(data) / width
|
|
const tileRows = 64
|
|
for j0 := 0; j0 < n; j0 += tileRows {
|
|
j1 := min(j0+tileRows, n)
|
|
for i := range width {
|
|
src := data[i*n+j0 : i*n+j1]
|
|
dst := out[j0*width+i:]
|
|
for k, b := range src {
|
|
dst[k*width] = b
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// checkFletcher32 verifies the fletcher32 checksum HDF5 appends to a
|
|
// filtered chunk.
|
|
func checkFletcher32(data []byte) error {
|
|
body, sum := data[:len(data)-4], binary.LittleEndian.Uint32(data[len(data)-4:])
|
|
if got := fletcher32(body); got != sum {
|
|
return base.Errf("the fletcher32 checksum does not match: %08x, want %08x", got, sum)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// arrayFromRaw builds the array for a dataset from its bytes. The
|
|
// decoded buffer is exactly one element per value, so the array takes
|
|
// it over instead of copying it a second time. Fixed-point data lands
|
|
// the core dtype its stored width and signedness declare, a boolean
|
|
// enumeration lands Bool, and a boolean cell outside the members 0 and
|
|
// 1 is refused rather than coerced.
|
|
func arrayFromRaw(raw []byte, dtype hdf5Type, shape []int) (*core.Array, error) {
|
|
switch {
|
|
case dtype.isFloat && dtype.width == 8:
|
|
vals := make([]float64, len(raw)/8)
|
|
for i := range vals {
|
|
vals[i] = math.Float64frombits(binary.LittleEndian.Uint64(raw[i*8:]))
|
|
}
|
|
return core.FloatsFromArray(vals, shape...)
|
|
case dtype.isFloat && dtype.width == 4:
|
|
vals := make([]float32, len(raw)/4)
|
|
for i := range vals {
|
|
vals[i] = math.Float32frombits(binary.LittleEndian.Uint32(raw[i*4:]))
|
|
}
|
|
return core.FromFloat32s(vals, shape...)
|
|
case dtype.isBool:
|
|
vals := make([]bool, len(raw))
|
|
for i, b := range raw {
|
|
if b > 1 {
|
|
return nil, base.Errf("a boolean value of %d is outside the members 0 and 1", b)
|
|
}
|
|
vals[i] = b != 0
|
|
}
|
|
return core.BoolsFromArray(vals, shape...)
|
|
case !dtype.signed && dtype.width == 8:
|
|
return nil, base.Errf("unsigned 64-bit integers have no exact core dtype")
|
|
}
|
|
switch {
|
|
case dtype.width == 1 && dtype.signed:
|
|
vals := make([]int8, len(raw))
|
|
for i, b := range raw {
|
|
vals[i] = int8(b)
|
|
}
|
|
return core.Int8sFromArray(vals, shape...)
|
|
case dtype.width == 1:
|
|
// raw may be a view of the file buffer; the payload copies out
|
|
// of it before the array takes the copy over.
|
|
vals := make([]uint8, len(raw))
|
|
copy(vals, raw)
|
|
return core.Uint8sFromArray(vals, shape...)
|
|
case dtype.width == 2 && dtype.signed:
|
|
vals := make([]int16, len(raw)/2)
|
|
for i := range vals {
|
|
vals[i] = int16(binary.LittleEndian.Uint16(raw[i*2:]))
|
|
}
|
|
return core.Int16sFromArray(vals, shape...)
|
|
case dtype.width == 2:
|
|
vals := make([]uint16, len(raw)/2)
|
|
for i := range vals {
|
|
vals[i] = binary.LittleEndian.Uint16(raw[i*2:])
|
|
}
|
|
return core.Uint16sFromArray(vals, shape...)
|
|
case dtype.width == 4 && dtype.signed:
|
|
vals := make([]int32, len(raw)/4)
|
|
for i := range vals {
|
|
vals[i] = int32(binary.LittleEndian.Uint32(raw[i*4:]))
|
|
}
|
|
return core.Int32sFromArray(vals, shape...)
|
|
case dtype.width == 4:
|
|
vals := make([]uint32, len(raw)/4)
|
|
for i := range vals {
|
|
vals[i] = binary.LittleEndian.Uint32(raw[i*4:])
|
|
}
|
|
return core.Uint32sFromArray(vals, shape...)
|
|
case dtype.width == 8:
|
|
// Signed 64-bit: unsigned was refused above, and decodeType
|
|
// admits no other fixed-point width.
|
|
vals := make([]int64, len(raw)/8)
|
|
for i := range vals {
|
|
vals[i] = int64(binary.LittleEndian.Uint64(raw[i*8:]))
|
|
}
|
|
return core.IntsFromArray(vals, shape...)
|
|
}
|
|
return nil, base.Errf("%d-byte fixed-point values have no core dtype", dtype.width)
|
|
}
|
|
|
|
// decodeAttribute reads one attribute message: its name and a textual
|
|
// rendering of its value, the shape the package's attribute maps use.
|
|
func (f *hdf5File) decodeAttribute(m []byte) (string, string, bool) {
|
|
if len(m) < 8 || m[0] != 1 {
|
|
return "", "", false
|
|
}
|
|
nameSize := int(binary.LittleEndian.Uint16(m[2:]))
|
|
typeSize := int(binary.LittleEndian.Uint16(m[4:]))
|
|
spaceSize := int(binary.LittleEndian.Uint16(m[6:]))
|
|
p := 8
|
|
if p+nameSize+typeSize+spaceSize > len(m) {
|
|
return "", "", false
|
|
}
|
|
name := strings.TrimRight(string(m[p:p+nameSize]), "\x00")
|
|
// The name field is padded so the datatype starts on an eight-byte
|
|
// boundary. The padding is inside the message, so the datatype is
|
|
// checked against the body's own length: the slice underneath is a
|
|
// view of the file buffer, whose capacity runs past the body.
|
|
p = alignUp(p+nameSize, 8)
|
|
if p+typeSize > len(m) {
|
|
return "", "", false
|
|
}
|
|
dtype, err := decodeType(m[p:p+typeSize], true)
|
|
if err != nil {
|
|
return "", "", false
|
|
}
|
|
// The datatype message is padded to an eight-byte boundary before
|
|
// the dataspace begins.
|
|
p = alignUp(p+typeSize, 8)
|
|
if p+spaceSize > len(m) {
|
|
return "", "", false
|
|
}
|
|
// The dataspace's dimension fields are sized by the file's own
|
|
// length size, exactly as the dataset path's are.
|
|
shape, err := decodeDataspace(m[p:p+spaceSize], f.lenSize)
|
|
if err != nil {
|
|
return "", "", false
|
|
}
|
|
p += spaceSize
|
|
// The value's byte count is bounded extent by extent against what is
|
|
// left of the message: a hostile dataspace must fail here, not wrap
|
|
// the product into a negative that would slip past a check made
|
|
// afterwards and then size an allocation.
|
|
size, err := hdf5ByteExtent(shape, dtype.width, uint64(len(m)-p))
|
|
if err != nil {
|
|
return "", "", false
|
|
}
|
|
raw := m[p : p+int(size)]
|
|
if dtype.class == 3 {
|
|
// A fixed-length string: the bytes are the text.
|
|
return name, strings.TrimRight(string(raw), "\x00"), true
|
|
}
|
|
if dtype.class == 9 {
|
|
// A variable-length string: the attribute holds a descriptor
|
|
// naming a global heap object, and that object holds the text.
|
|
// The descriptor is four bytes of length, the collection
|
|
// address, then the object index.
|
|
if len(raw) < 8+f.offSize {
|
|
return "", "", false
|
|
}
|
|
length := int(binary.LittleEndian.Uint32(raw))
|
|
heapAddr := uint64At(raw[4:], f.offSize)
|
|
index := binary.LittleEndian.Uint32(raw[4+f.offSize:])
|
|
text, err := f.heapString(heapAddr, index, length)
|
|
if err != nil {
|
|
return "", "", false
|
|
}
|
|
return name, text, true
|
|
}
|
|
// The element count the numeric rendering below walks, derived from
|
|
// the validated byte count rather than from a second multiplication.
|
|
count := 0
|
|
if dtype.width > 0 {
|
|
count = len(raw) / dtype.width
|
|
}
|
|
vals := make([]string, 0, count)
|
|
for i := range count {
|
|
cell := raw[i*dtype.width : (i+1)*dtype.width]
|
|
switch {
|
|
case dtype.isFloat && dtype.width == 8:
|
|
vals = append(vals, strconv.FormatFloat(math.Float64frombits(binary.LittleEndian.Uint64(cell)), 'g', -1, 64))
|
|
case dtype.isFloat && dtype.width == 4:
|
|
vals = append(vals, strconv.FormatFloat(float64(math.Float32frombits(binary.LittleEndian.Uint32(cell))), 'g', -1, 32))
|
|
case dtype.isBool:
|
|
// The boolean enumeration renders as its own values; a cell
|
|
// outside them is a file that contradicts its datatype, and
|
|
// the attribute is dropped rather than guessed at.
|
|
if cell[0] > 1 {
|
|
return "", "", false
|
|
}
|
|
vals = append(vals, strconv.FormatInt(int64(cell[0]), 10))
|
|
case dtype.signed:
|
|
// The datatype's own signed bit decides how its stored bits
|
|
// read: an int8 attribute of -1 renders "-1", not 255.
|
|
u := uint64At(cell, dtype.width)
|
|
if width := uint(dtype.width); width < 64 {
|
|
mask := uint64(1)<<(8*width) - 1
|
|
if u&(uint64(1)<<(8*width-1)) != 0 {
|
|
u |= ^mask // sign-extend into the 64-bit read
|
|
}
|
|
}
|
|
vals = append(vals, strconv.FormatInt(int64(u), 10))
|
|
default:
|
|
vals = append(vals, strconv.FormatUint(uint64At(cell, dtype.width), 10))
|
|
}
|
|
}
|
|
if len(vals) == 1 {
|
|
return name, vals[0], true
|
|
}
|
|
return name, "[" + strings.Join(vals, ", ") + "]", true
|
|
}
|
|
|
|
// heapString reads a variable-length string out of a global heap
|
|
// collection: the descriptor names the collection and an object index,
|
|
// and the collection lists (index, reference count, size, bytes)
|
|
// records, each padded to eight bytes.
|
|
func (f *hdf5File) heapString(addr uint64, index uint32, length int) (string, error) {
|
|
const name = "LoadHDF5"
|
|
head := f.bytes(addr, 16)
|
|
if head == nil || !bytes.Equal(head[:4], []byte("GCOL")) {
|
|
return "", base.Errf("%s: no global heap collection at %d", name, addr)
|
|
}
|
|
size := f.length(int(addr) + 8) // the collection size, header included
|
|
collection := f.bytes(addr, size)
|
|
if collection == nil {
|
|
return "", base.Errf("%s: the global heap collection at %d lies outside the file", name, addr)
|
|
}
|
|
p := 8 + f.lenSize
|
|
for p+8+f.lenSize <= len(collection) {
|
|
objIndex := uint32(binary.LittleEndian.Uint16(collection[p:]))
|
|
objSize := uint64At(collection[p+8:], f.lenSize)
|
|
body := p + 8 + f.lenSize
|
|
// The bound compares in uint64: an int addition would wrap a
|
|
// hostile object size past the guard into a negative value and
|
|
// the slice below would panic.
|
|
if objSize > uint64(len(collection)-body) {
|
|
break
|
|
}
|
|
size := int(objSize)
|
|
if objIndex == index {
|
|
text := collection[body : body+size]
|
|
if length > 0 && length < len(text) {
|
|
text = text[:length]
|
|
}
|
|
return strings.TrimRight(string(text), "\x00"), nil
|
|
}
|
|
p = alignUp(body+size, 8)
|
|
}
|
|
return "", base.Errf("%s: the global heap object %d is not in the collection at %d", name, index, addr)
|
|
}
|
|
|
|
// fletcher32 is the checksum HDF5's fletcher32 filter appends: a
|
|
// 32-bit Fletcher sum over 16-bit words, ones-complement folded.
|
|
func fletcher32(data []byte) uint32 {
|
|
sum1, sum2 := uint32(0xffff), uint32(0xffff)
|
|
// The filter runs in blocks of 359 words (718 bytes).
|
|
for len(data) > 0 {
|
|
n := min(len(data), 718)
|
|
block := data[:n]
|
|
if n%2 != 0 {
|
|
block = append(append([]byte{}, block...), 0)
|
|
}
|
|
for i := 0; i+1 < len(block); i += 2 {
|
|
sum1 += uint32(binary.BigEndian.Uint16(block[i:]))
|
|
sum2 += sum1
|
|
}
|
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16)
|
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16)
|
|
data = data[n:]
|
|
}
|
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16)
|
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16)
|
|
return sum2<<16 | sum1
|
|
}
|
|
|
|
// hdf5Lookup3 is the Jenkins lookup3 hash in the little-endian,
|
|
// byte-wise form HDF5 stores as the checksum of superblock versions 2
|
|
// and 3 and of every chunk of an object header version 2, with the
|
|
// zero initial value the library uses. The main loop mixes whole
|
|
// 12-byte blocks and the switch folds the tail, whose bytes fall into
|
|
// the three words highest first.
|
|
func hdf5Lookup3(key []byte) uint32 {
|
|
a := uint32(0xdeadbeef) + uint32(len(key))
|
|
b := a
|
|
c := a
|
|
p := 0
|
|
for ; len(key)-p > 12; p += 12 {
|
|
a += binary.LittleEndian.Uint32(key[p:])
|
|
b += binary.LittleEndian.Uint32(key[p+4:])
|
|
c += binary.LittleEndian.Uint32(key[p+8:])
|
|
a, b, c = hdf5Lookup3Mix(a, b, c)
|
|
}
|
|
switch r := key[p:]; len(r) {
|
|
case 12:
|
|
c += uint32(r[11]) << 24
|
|
fallthrough
|
|
case 11:
|
|
c += uint32(r[10]) << 16
|
|
fallthrough
|
|
case 10:
|
|
c += uint32(r[9]) << 8
|
|
fallthrough
|
|
case 9:
|
|
c += uint32(r[8])
|
|
fallthrough
|
|
case 8:
|
|
b += uint32(r[7]) << 24
|
|
fallthrough
|
|
case 7:
|
|
b += uint32(r[6]) << 16
|
|
fallthrough
|
|
case 6:
|
|
b += uint32(r[5]) << 8
|
|
fallthrough
|
|
case 5:
|
|
b += uint32(r[4])
|
|
fallthrough
|
|
case 4:
|
|
a += uint32(r[3]) << 24
|
|
fallthrough
|
|
case 3:
|
|
a += uint32(r[2]) << 16
|
|
fallthrough
|
|
case 2:
|
|
a += uint32(r[1]) << 8
|
|
fallthrough
|
|
case 1:
|
|
a += uint32(r[0])
|
|
case 0:
|
|
return c
|
|
}
|
|
a, b, c = hdf5Lookup3Final(a, b, c)
|
|
return c
|
|
}
|
|
|
|
// hdf5Lookup3Mix is lookup3's inner round.
|
|
func hdf5Lookup3Mix(a, b, c uint32) (uint32, uint32, uint32) {
|
|
a -= c
|
|
a ^= bits.RotateLeft32(c, 4)
|
|
c += b
|
|
b -= a
|
|
b ^= bits.RotateLeft32(a, 6)
|
|
a += c
|
|
c -= b
|
|
c ^= bits.RotateLeft32(b, 8)
|
|
b += a
|
|
a -= c
|
|
a ^= bits.RotateLeft32(c, 16)
|
|
c += b
|
|
b -= a
|
|
b ^= bits.RotateLeft32(a, 19)
|
|
a += c
|
|
c -= b
|
|
c ^= bits.RotateLeft32(b, 4)
|
|
b += a
|
|
return a, b, c
|
|
}
|
|
|
|
// hdf5Lookup3Final is lookup3's closing avalanche.
|
|
func hdf5Lookup3Final(a, b, c uint32) (uint32, uint32, uint32) {
|
|
c ^= b
|
|
c -= bits.RotateLeft32(b, 14)
|
|
a ^= c
|
|
a -= bits.RotateLeft32(c, 11)
|
|
b ^= a
|
|
b -= bits.RotateLeft32(a, 25)
|
|
c ^= b
|
|
c -= bits.RotateLeft32(b, 16)
|
|
a ^= c
|
|
a -= bits.RotateLeft32(c, 4)
|
|
b ^= a
|
|
b -= bits.RotateLeft32(a, 14)
|
|
c ^= b
|
|
c -= bits.RotateLeft32(b, 24)
|
|
return a, b, c
|
|
}
|