// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package io import ( "bytes" "compress/zlib" "encoding/binary" "io" "maps" "math" "math/bits" "os" "slices" "strconv" "strings" "sourcedock.dev/petrbalvin/tensor/internal/base" "sourcedock.dev/petrbalvin/tensor/internal/core" ) // HDF5 (HDF5 1.8/1.10 file format), read-only, for the numeric arrays // scientific files carry: every dataset of a file comes back as an // Array with its path, shape and attributes. // // Supported: superblocks 0 to 3 (the classic layout and the "latest" // library version, whose checksums are the lookup3 sum and are // verified), object headers versions 1 and 2, groups stored as symbol // tables or as link messages, datasets stored contiguously, compactly // or in chunks through a version 1 B-tree, the deflate, shuffle and // fletcher32 filters, fixed-point and floating-point datatypes of the // usual widths, the boolean enumeration convention, and attributes in // the object header. Fixed-point datasets land the core dtype their // stored width and signedness declare; floating-point datasets land // float64 at width 8 and float32 at width 4. // // Refused with an error naming what is missing, never guessed at: the // superblock extension, the fractal-heap link storage of dense groups, // the version 2 B-tree a latest-version file uses to index chunks, // variable-length, compound, array and every non-boolean enumeration // datatype, bit fields, string datasets, and the szip, nbit and // scale-offset filters. // Superblock and object header signatures. var ( hdf5Magic = []byte{0x89, 'H', 'D', 'F', '\r', '\n', 0x1a, '\n'} hdf5Tree = []byte{'T', 'R', 'E', 'E'} hdf5SymbolNode = []byte{'S', 'N', 'O', 'D'} hdf5LocalHeap = []byte{'H', 'E', 'A', 'P'} hdf5ObjHdr2 = []byte{'O', 'H', 'D', 'R'} hdf5Chunk2 = []byte{'O', 'C', 'H', 'K'} ) // hdf5MaxDatasetBytes bounds the raw byte extent of one dataset and of // one chunk. The reader materialises whole datasets, so a declared size // beyond this budget is a hostile header, not data: without the cap a // 16-byte file could order a terabyte-scale allocation through the // dataspace, the fill-value path or a single chunk declaration. Honest // datasets above the budget belong behind mmap, not behind an eager // read. const hdf5MaxDatasetBytes = 2 << 30 // 2 GiB // hdf5ByteExtent multiplies declared extents by an element width into a // byte count, bounding every factor against budget *before* it is // multiplied. Multiplying first and comparing the product afterwards is // the defect the review found: a dataspace of 2^32 by 2^32 float64 // elements is 2^64 bytes, which wraps to zero, and a cap that runs after // the multiplication therefore accepts a declaration the file cannot // back, while a chunk of 2^32 by 2^29 elements looks empty the same way. // Every partial product here stays inside budget, so no caller can be // handed a wrapped count either. A zero extent is legal and yields zero // bytes (an empty dataset); a negative one is a hostile header, as is a // non-positive element width. func hdf5ByteExtent(dims []int, width int, budget uint64) (uint64, error) { total := uint64(1) empty := false for _, d := range dims { if d < 0 { return 0, base.Errf("a declared extent of %d", d) } if d == 0 { empty = true continue } if uint64(d) > budget/total { return 0, base.Errf("the declared extents multiply past the %d byte budget", budget) } total *= uint64(d) } if empty { return 0, nil } if width <= 0 { return 0, base.Errf("a declared element width of %d bytes", width) } if total > budget/uint64(width) { return 0, base.Errf("the declared extents multiply past the %d byte budget", budget) } return total * uint64(width), nil } // HDF5 dataset object header message types. const ( hdf5MsgDataspace = 1 hdf5MsgLinkInfo = 2 hdf5MsgDatatype = 3 hdf5MsgFillValueOld = 4 hdf5MsgFillValue = 5 hdf5MsgLink = 6 hdf5MsgDataLayout = 8 hdf5MsgGroupInfo = 10 hdf5MsgFilterPipeline = 11 hdf5MsgAttribute = 12 hdf5MsgContinuation = 16 hdf5MsgSymbolTable = 17 ) // HDF5 filter identifiers. const ( hdf5FilterDeflate = 1 hdf5FilterShuffle = 2 hdf5FilterFletcher32 = 3 ) // HDF5Dataset is one dataset of a file: its path from the root, its // shape (row-major, as the file stores it), its values and its // attributes. The attribute map also carries the attributes of the // groups the dataset sits in, the nearest one winning, because that is // where files usually put the units and titles that apply to a whole // group. type HDF5Dataset struct { Path string Shape []int Values *core.Array Attrs map[string]string } // LoadHDF5 reads every dataset of an HDF5 file, in path order, as // numeric arrays. The values keep the file's own dtype where the core // has one: fixed-point data lands by stored width and signedness // (int8, uint8, int16, uint16, int32, uint32, and int64 as Int), the // boolean enumeration convention lands Bool, and floating-point data // lands float64 at width 8 and float32 at width 4. Unsigned 64-bit // data is refused, because no core dtype holds every value of it. // // An object hard-linked under several names is read once: its datasets // appear under the first path the traversal reaches, the links of a // group being visited in their file order, and never under the later // ones. A link that closes a cycle along the path to it is an error, // because no linear listing of the file exists. func LoadHDF5(path string) ([]HDF5Dataset, error) { data, err := os.ReadFile(path) if err != nil { return nil, base.Errf("LoadHDF5: %w", err) } f, err := newHDF5File(data) if err != nil { return nil, err } out := []HDF5Dataset{} if err := f.walk(f.rootAddress, newHDF5WalkState(), &out); err != nil { return nil, err } slices.SortFunc(out, func(x, y HDF5Dataset) int { return strings.Compare(x.Path, y.Path) }) return out, nil } // hdf5File is a parsed superblock plus the raw bytes of the file every // offset is resolved against. type hdf5File struct { data []byte superblock byte offSize int lenSize int rootAddress uint64 groupLeafK int groupInnerK int } func newHDF5File(data []byte) (*hdf5File, error) { const name = "LoadHDF5" if len(data) < 96 || !bytes.Equal(data[:8], hdf5Magic) { return nil, base.Errf("%s: not an HDF5 file, the signature is % x", name, data[:min(8, len(data))]) } version := data[8] f := &hdf5File{data: data, superblock: version} switch version { case 0, 1: // The classic layout: the sizes and the root group entry follow // the fixed header. case 2, 3: // The modern layout: the sizes follow the version byte, the // four addresses follow without the group K values, and the // root group is named by its object header address. The whole // superblock carries a lookup3 checksum. f.offSize = int(data[9]) f.lenSize = int(data[10]) switch f.offSize { case 4, 8: default: return nil, base.Errf("%s: %d-byte addresses are not supported", name, f.offSize) } if f.lenSize != f.offSize && f.lenSize != 8 && f.lenSize != 4 { return nil, base.Errf("%s: %d-byte lengths are not supported", name, f.lenSize) } stored := 12 + 4*f.offSize + 4 if len(data) < stored { return nil, base.Errf("%s: superblock version %d is truncated", name, version) } // Every address the format carries is absolute, so a non-zero // base would need base-relative resolution, which is not // implemented; the extension holds the file space strategy and // driver settings, which the reader does not interpret. Both // are refused by name rather than read wrong. if baseAddr := f.u64At(12, f.offSize); baseAddr != 0 { return nil, base.Errf("%s: a non-zero base address of %d is not supported", name, baseAddr) } if ext := f.u64At(12+f.offSize, f.offSize); ext != math.MaxUint64 { return nil, base.Errf("%s: the superblock extension at %d is not supported; rewrite the file with the default library version", name, ext) } if got, want := binary.LittleEndian.Uint32(data[12+4*f.offSize:]), hdf5Lookup3(data[:12+4*f.offSize]); got != want { return nil, base.Errf("%s: the superblock fails its checksum (%#08x, want %#08x)", name, got, want) } f.rootAddress = f.u64At(12+3*f.offSize, f.offSize) if f.rootAddress == math.MaxUint64 { return nil, base.Errf("%s: the root group has no object header", name) } return f, nil default: return nil, base.Errf("%s: superblock version %d is not supported; rewrite the file with the default version", name, version) } f.offSize = int(data[13]) f.lenSize = int(data[14]) switch f.offSize { case 4, 8: default: return nil, base.Errf("%s: %d-byte addresses are not supported", name, f.offSize) } if f.lenSize != f.offSize && f.lenSize != 8 && f.lenSize != 4 { return nil, base.Errf("%s: %d-byte lengths are not supported", name, f.lenSize) } f.groupLeafK = int(binary.LittleEndian.Uint16(data[16:])) f.groupInnerK = int(binary.LittleEndian.Uint16(data[18:])) pos := 24 if version == 1 { // Version 1 carries the indexed-storage K before the addresses. pos = 28 } // base address (every file this reader accepts carries 0 there: // a non-zero base would require base-relative offset resolution, // which is not implemented), free space information, end of file, // driver information: four addresses, then the root group's symbol // table entry, whose first field is the link name offset and is // sized by the length size, not the address size. pos += 4 * f.offSize f.rootAddress = f.u64At(pos+f.lenSize, f.offSize) if f.rootAddress == math.MaxUint64 { return nil, base.Errf("%s: the root group has no object header", name) } return f, nil } // u64At reads an address or length of size bytes at an absolute file // offset; every offset the reader resolves is absolute. func (f *hdf5File) u64At(off int, size int) uint64 { if off < 0 || off+size > len(f.data) { return math.MaxUint64 } switch size { case 4: return uint64(binary.LittleEndian.Uint32(f.data[off:])) case 8: return binary.LittleEndian.Uint64(f.data[off:]) } return math.MaxUint64 } // at reads an address field of the file's address size at off. func (f *hdf5File) at(off int) uint64 { return f.u64At(off, f.offSize) } // len reads a length field of the file's length size at off. func (f *hdf5File) length(off int) uint64 { return f.u64At(off, f.lenSize) } // bytes returns the file bytes of the given region, or nil when the // region lies outside the file. func (f *hdf5File) bytes(addr uint64, n uint64) []byte { if addr == math.MaxUint64 || n > uint64(len(f.data)) { return nil } start := addr if start > uint64(len(f.data)) || n > uint64(len(f.data))-start { return nil } return f.data[start : start+n] } // hdf5Message is one header message with its raw payload. type hdf5Message struct { typ uint16 data []byte } // messages reads an object header's messages, following continuation // blocks. Versions 1 and 2 are supported: version 1 is what the // library writes by default, version 2 what the "latest" library // version writes. func (f *hdf5File) messages(addr uint64) ([]hdf5Message, error) { const name = "LoadHDF5" head := f.bytes(addr, 4) if head == nil { return nil, base.Errf("%s: object header at %d lies outside the file", name, addr) } if bytes.Equal(head, hdf5ObjHdr2) { return f.messagesV2(addr) } start := f.bytes(addr, 16) if start == nil { return nil, base.Errf("%s: object header at %d lies outside the file", name, addr) } if start[0] != 1 { return nil, base.Errf("%s: object header version %d is not supported; rewrite the file with the default library version", name, start[0]) } // version(1), reserved(1), messages(2), references(4), data size(4), // padding(4), then the message data itself. nmsg := int(binary.LittleEndian.Uint16(start[2:])) dataSize := int(binary.LittleEndian.Uint32(start[8:])) pos := addr + 16 region := f.bytes(pos, uint64(dataSize)) if region == nil { return nil, base.Errf("%s: object header at %d is truncated", name, addr) } // The declared count sizes the preallocation, so it is capped by the // region that would have to hold the messages: a header claiming // 65535 messages costs 16 bytes in the file, and every message is at // least eight bytes, so the region bounds the count that can follow. out := make([]hdf5Message, 0, min(nmsg, len(region)/8+1)) p := 0 for i := range nmsg { if p+8 > len(region) { return nil, base.Errf("%s: object header at %d ends inside a message", name, addr) } typ := binary.LittleEndian.Uint16(region[p:]) size := int(binary.LittleEndian.Uint16(region[p+2:])) if p+8+size > len(region) { return nil, base.Errf("%s: object header at %d ends inside message %d", name, addr, i) } body := region[p+8 : p+8+size] if typ == hdf5MsgContinuation { // offset, length: another block of messages. The declared // count includes the messages the chain carries, so the // link is the last message physically present in this // block; every further link is followed inside messagesIn, // whose visited set keeps a hostile self-referencing block // from looping. The link body must hold an address and a // length outright: reading the fields from whatever follows // a short message would guess at bytes the message never // carried. if size < f.offSize+f.lenSize { return nil, base.Errf("%s: object header at %d carries a continuation message shorter than an offset and a length", name, addr) } blockAddr := f.at(int(pos) + p + 8) next := f.bytes(blockAddr, f.length(int(pos)+p+8+f.offSize)) if next == nil { return nil, base.Errf("%s: object header continuation at %d lies outside the file", name, addr) } cont, err := f.messagesIn(next, blockAddr, map[uint64]bool{addr: true}, 0) if err != nil { return nil, err } out = append(out, cont...) break } out = append(out, hdf5Message{typ: typ, data: body}) p += 8 + size p = alignUp(p, 8) } return out, nil } // messagesV2 reads a version 2 object header: the OHDR signature, a // version and a flag byte, the flag-selected prefix fields, the size // of the first chunk, then the messages and a lookup3 checksum over // everything from the signature to the end of the chunk. The format // inserts no alignment into the stream, so the walk reads the fields // packed as they lie. func (f *hdf5File) messagesV2(addr uint64) ([]hdf5Message, error) { const name = "LoadHDF5" head := f.bytes(addr, 6) if head == nil { return nil, base.Errf("%s: object header at %d lies outside the file", name, addr) } if head[4] != 2 { return nil, base.Errf("%s: object header version %d is not supported; rewrite the file with the default library version", name, head[4]) } flags := head[5] if flags&^byte(0x3f) != 0 { return nil, base.Errf("%s: the object header at %d carries unknown status flags %#02x", name, addr, flags) } p := addr + 6 if flags&0x20 != 0 { p += 16 // access, modification, change and birth times } if flags&0x10 != 0 { p += 4 // the max compact and min dense attribute counts } width := 1 << (flags & 0x03) // The checksum of four bytes follows the chunk directly, so the // size field, the chunk and the checksum must all lie inside the // file; the bound is settled before the size is read, so a hostile // header cannot point the read past the end. if p+uint64(width)+4 > uint64(len(f.data)) { return nil, base.Errf("%s: object header at %d is truncated", name, addr) } chunk0 := leN(f.data[p:], width) if chunk0 > uint64(len(f.data))-p-uint64(width)-4 { return nil, base.Errf("%s: object header at %d is truncated", name, addr) } start := p + uint64(width) end := start + chunk0 if got, want := binary.LittleEndian.Uint32(f.data[end:]), hdf5Lookup3(f.data[addr:end]); got != want { return nil, base.Errf("%s: the object header at %d fails its checksum (%#08x, want %#08x)", name, addr, got, want) } visited := map[uint64]bool{addr: true} return f.messagesV2Stream(f.data[start:end], addr, flags&0x04 != 0, visited, 0) } // messagesV2Stream walks one version 2 message region: each message is // one type byte, a two-byte size and a flag byte, widened by a // two-byte creation order when the header tracks creation order. A // continuation message hands the walk to a further OCHK block, whose // checksum covers signature and messages alike; visited refuses a // block reached twice and depth an unbounded chain. A writer may leave // a gap of up to three bytes before the region's checksum; anything // wider would be a message the region cannot hold. func (f *hdf5File) messagesV2Stream(region []byte, addr uint64, ordered bool, visited map[uint64]bool, depth int) ([]hdf5Message, error) { const name = "LoadHDF5" if depth > hdf5MaxHeaderBlocks { return nil, base.Errf("%s: the object header at %d chains through more than %d continuation blocks", name, addr, hdf5MaxHeaderBlocks) } out := []hdf5Message{} p := 0 for p+4 <= len(region) { typ := region[p] size := int(binary.LittleEndian.Uint16(region[p+1:])) p += 4 if ordered { if p+2 > len(region) { return nil, base.Errf("%s: object header at %d ends inside a message", name, addr) } p += 2 } if p+size > len(region) { return nil, base.Errf("%s: object header at %d ends inside a message", name, addr) } if typ == hdf5MsgContinuation { // The link body is an offset of the address size followed by // a length of the length size; a block that ends inside the // link has neither, so the read is refused instead of taken // from bytes past the block. if p+f.offSize+f.lenSize > len(region) { return nil, base.Errf("%s: object header at %d carries a continuation message shorter than an offset and a length", name, addr) } blockAddr := leN(region[p:], f.offSize) block := f.bytes(blockAddr, leN(region[p+f.offSize:], f.lenSize)) if block == nil || len(block) < 8 || !bytes.Equal(block[:4], hdf5Chunk2) { return nil, base.Errf("%s: no version 2 continuation block at %d", name, blockAddr) } if got, want := binary.LittleEndian.Uint32(block[len(block)-4:]), hdf5Lookup3(block[:len(block)-4]); got != want { return nil, base.Errf("%s: the continuation block at %d fails its checksum (%#08x, want %#08x)", name, blockAddr, got, want) } if visited[blockAddr] { return nil, base.Errf("%s: object header continuation block at %d is reachable twice", name, blockAddr) } visited[blockAddr] = true cont, err := f.messagesV2Stream(block[4:len(block)-4], blockAddr, ordered, visited, depth+1) if err != nil { return nil, err } out = append(out, cont...) } else if typ != 0 { // A null message is the residue of a deletion and names no // content; the live messages carry on behind it. out = append(out, hdf5Message{typ: uint16(typ), data: region[p : p+size]}) } p += size } if gap := len(region) - p; gap >= 4 { return nil, base.Errf("%s: object header at %d ends inside a message", name, addr) } return out, nil } // messagesIn reads the messages of one continuation block, whose size // is the block itself. A continuation message inside the block chains // to a further block; the visited set holds every block address already // read, so a cycle is an error instead of an infinite walk, and depth // bounds the chain the same way the version 2 walk bounds its own. func (f *hdf5File) messagesIn(block []byte, addr uint64, visited map[uint64]bool, depth int) ([]hdf5Message, error) { const name = "LoadHDF5" if visited[addr] { return nil, base.Errf("%s: object header continuation block at %d is reachable twice", name, addr) } visited[addr] = true if depth > hdf5MaxHeaderBlocks { return nil, base.Errf("%s: the object header at %d chains through more than %d continuation blocks", name, addr, hdf5MaxHeaderBlocks) } out := []hdf5Message{} p := 0 for p+8 <= len(block) { typ := binary.LittleEndian.Uint16(block[p:]) size := int(binary.LittleEndian.Uint16(block[p+2:])) if p+8+size > len(block) { return nil, base.Errf("%s: object header continuation block ends inside a message", name) } if typ == hdf5MsgContinuation { // The link body is an offset of the address size followed // by a length of the length size; a block that ends inside // the link has neither, so the read is refused instead of // taken from bytes past the block. if p+8+f.offSize+f.lenSize > len(block) { return nil, base.Errf("%s: object header continuation block at %d carries a continuation message shorter than an offset and a length", name, addr) } blockAddr := leN(block[p+8:], f.offSize) next := f.bytes(blockAddr, leN(block[p+8+f.offSize:], f.lenSize)) if next == nil { return nil, base.Errf("%s: object header continuation block at %d lies outside the file", name, blockAddr) } cont, err := f.messagesIn(next, blockAddr, visited, depth+1) if err != nil { return nil, err } out = append(out, cont...) p = alignUp(p+8+size, 8) continue } out = append(out, hdf5Message{typ: typ, data: block[p+8 : p+8+size]}) p = alignUp(p+8+size, 8) } return out, nil } // leN reads a little-endian integer of n bytes from the head of b. func leN(b []byte, n int) uint64 { var v uint64 for i := 0; i < n && i < len(b); i++ { v |= uint64(b[i]) << (8 * i) } return v } func alignUp(n, to int) int { return (n + to - 1) / to * to } // hdf5MaxGroupDepth bounds how deeply groups may nest. It is a stack // guard for the walk, not a memory guard: the traversal carries one path // buffer and one attribute map for the whole file, so the memory a deep // file costs grows with its depth rather than with its square. Real // files nest a handful of levels deep; the cap only refuses the extreme. const hdf5MaxGroupDepth = 512 // hdf5MaxHeaderBlocks bounds how many continuation blocks one object // header may chain through. It is a stack guard for the version 2 // message walk, whose recursion descends one frame per block: real // headers split over a handful of blocks, the cap only refuses the // chain a hostile file builds to exhaust the stack. const hdf5MaxHeaderBlocks = 512 // hdf5DatasetEnvelope is the fixed per-dataset charge the walk deducts // from the read budget beyond the value bytes themselves: the result's // struct, shape slice, path string and attribute map cost roughly the // same for every dataset, however small its values are, and a file of // millions of empty datasets must pay for the envelopes it orders. // The envelope is a conservative upper estimate of that structure, not // a measurement of it, and it is charged together with the dataset's // path length, which the result copies out of the walk's buffer. const hdf5DatasetEnvelope = 512 // hdf5WalkState is the traversal state shared by the whole walk: the // object headers currently on the path, so a hard-link cycle is refused // instead of recursed, the object headers already walked from any path, // so a diamond of hard links is read once instead of once per path, the // inherited attribute map, which every level adds to and then undoes, // and the path buffer, which every level extends and then truncates. // All are shared rather than copied per level: a copy per level makes // the live memory grow with the square of the nesting, which is the // hostile-file blow-up the review found. type hdf5WalkState struct { onPath map[uint64]bool seen map[uint64]bool attrs map[string]string path []byte depth int // budget is the byte extent the walk may still hand out, the // whole-file aggregate behind the per-dataset cap: a file whose // datasets each fit the cap can still declare thousands of them, // and the sum must refuse, not the machine. Every dataset pays its // value bytes plus hdf5DatasetEnvelope and its path. budget int64 } func newHDF5WalkState() *hdf5WalkState { return &hdf5WalkState{onPath: map[uint64]bool{}, seen: map[uint64]bool{}, attrs: map[string]string{}, path: []byte{'/'}, budget: hdf5MaxDatasetBytes} } // hdf5AttrUndo records one attribute this level overwrote, so leaving // the level restores the outer value instead of dropping it. type hdf5AttrUndo struct { key string prev string had bool } // walk visits the group whose path is already in st.path and everything // under it, appending the datasets. The attributes of a group are // inherited by everything below it, the nearest group winning, so a // dataset carries the units and titles its file hangs on the enclosing // groups. // // An object reachable through several hard links (a diamond) is walked // once, under the first path the traversal reaches it from: st.seen // records every object header already walked and a second arrival // skips. The first path wins deterministically because the links of a // group are visited in the order their file stores them. An object that // closes a cycle along the current path is different: it is refused // with an error, because no order of visits can read a cycle at all. func (f *hdf5File) walk(addr uint64, st *hdf5WalkState, out *[]HDF5Dataset) error { const name = "LoadHDF5" if st.onPath[addr] { return base.Errf("%s: the group at object header %d is linked into itself along %q; a hard-link cycle cannot be walked", name, addr, string(st.path)) } if st.seen[addr] { // Already walked from another path: its datasets are in out // under that, earlier, path and the file's content is read. return nil } if st.depth >= hdf5MaxGroupDepth { return base.Errf("%s: the groups nest deeper than %d levels", name, hdf5MaxGroupDepth) } st.onPath[addr] = true st.seen[addr] = true st.depth++ defer func() { delete(st.onPath, addr) st.depth-- }() msgs, err := f.messages(addr) if err != nil { return err } undo := make([]hdf5AttrUndo, 0, 4) for k, v := range f.attributesOf(msgs) { prev, had := st.attrs[k] undo = append(undo, hdf5AttrUndo{key: k, prev: prev, had: had}) st.attrs[k] = v } defer func() { for _, u := range slices.Backward(undo) { if u.had { st.attrs[u.key] = u.prev continue } delete(st.attrs, u.key) } }() links, err := f.links(msgs) if err != nil { return err } var datasetMsgs []hdf5Message for _, m := range msgs { switch m.typ { case hdf5MsgDataspace, hdf5MsgDatatype, hdf5MsgDataLayout, hdf5MsgFilterPipeline, hdf5MsgFillValue: datasetMsgs = append(datasetMsgs, m) } } // A group has no dataspace; a dataset does. hasSpace := false for _, m := range datasetMsgs { if m.typ == hdf5MsgDataspace { hasSpace = true } } if hasSpace { ds, err := f.dataset(string(st.path), datasetMsgs, st) if err != nil { return err } // The dataset owns its attributes: a clone, because the walk's // map keeps changing for the levels below this one. ds.Attrs = maps.Clone(st.attrs) *out = append(*out, ds) return nil } for _, l := range links { mark := len(st.path) if st.path[mark-1] != '/' { st.path = append(st.path, '/') } st.path = append(st.path, l.name...) err := f.walk(l.address, st, out) st.path = st.path[:mark] if err != nil { return err } } return nil } // hdf5Link is one named hard link of a group. type hdf5Link struct { name string address uint64 } // links collects a group's hard links from its link messages and, when // present, its symbol table. func (f *hdf5File) links(msgs []hdf5Message) ([]hdf5Link, error) { out := []hdf5Link{} for _, m := range msgs { switch m.typ { case hdf5MsgLink: l, err := f.decodeLink(m.data) if err != nil { return nil, err } if l.address != math.MaxUint64 { out = append(out, l) } case hdf5MsgSymbolTable: st, err := f.symbolTableLinks(m.data) if err != nil { return nil, err } out = append(out, st...) case hdf5MsgLinkInfo: // Dense storage would need the fractal heap; a group whose // links all live in the header carries the info message // without using the dense structures. } } if len(out) == 0 { for _, m := range msgs { if m.typ == hdf5MsgLinkInfo { const name = "LoadHDF5" return nil, base.Errf("%s: this group stores its links in a dense (fractal heap) structure, which is not supported", name) } } } return out, nil } // decodeLink parses one link message, ignoring soft and external links // (they name no object in this file). func (f *hdf5File) decodeLink(m []byte) (hdf5Link, error) { const name = "LoadHDF5" if len(m) < 2 { return hdf5Link{}, base.Errf("%s: link message is truncated", name) } if m[0] != 1 { return hdf5Link{}, base.Errf("%s: link message version %d is not supported", name, m[0]) } flags := m[1] p := 2 linkType := uint8(0) if flags&0x08 != 0 { if p >= len(m) { return hdf5Link{}, base.Errf("%s: link message is truncated", name) } linkType = m[p] p++ } if flags&0x04 != 0 { if p+8 > len(m) { return hdf5Link{}, base.Errf("%s: link message is truncated", name) } p += 8 // creation order } nameLenSize := 1 << (flags & 0x03) // 1, 2, 4 or 8 bytes nameLen, ok := readUintLE(m[p:], nameLenSize) if !ok { return hdf5Link{}, base.Errf("%s: link message is truncated", name) } p += nameLenSize if uint64(p)+nameLen > uint64(len(m)) { return hdf5Link{}, base.Errf("%s: link message is truncated", name) } linkName := string(m[p : p+int(nameLen)]) p += int(nameLen) if linkType != 0 { // A soft or external link: skip it, it names no object here. return hdf5Link{name: linkName, address: math.MaxUint64}, nil } addr := f.at2(m[p:]) return hdf5Link{name: linkName, address: addr}, nil } // at2 reads an address of the file's address size from a message // payload, or returns the undefined address. func (f *hdf5File) at2(b []byte) uint64 { if len(b) < f.offSize { return math.MaxUint64 } switch f.offSize { case 4: return uint64(binary.LittleEndian.Uint32(b)) case 8: return binary.LittleEndian.Uint64(b) } return math.MaxUint64 } func readUintLE(b []byte, size int) (uint64, bool) { if len(b) < size { return 0, false } switch size { case 1: return uint64(b[0]), true case 2: return uint64(binary.LittleEndian.Uint16(b)), true case 4: return uint64(binary.LittleEndian.Uint32(b)), true case 8: return binary.LittleEndian.Uint64(b), true } return 0, false } // symbolTableLinks walks the group's version 1 B-tree of symbol table // nodes and reads the names out of the local heap. func (f *hdf5File) symbolTableLinks(m []byte) ([]hdf5Link, error) { const name = "LoadHDF5" if len(m) < 2*f.offSize { return nil, base.Errf("%s: symbol table message is truncated", name) } treeAddr := f.at2(m) heapAddr := f.at2(m[f.offSize:]) heap, err := f.localHeap(heapAddr) if err != nil { return nil, err } out := []hdf5Link{} set := map[uint64]bool{} if err := f.treeLinks(treeAddr, heap, set, &out, 0); err != nil { return nil, err } return out, nil } // treeLinks descends a group B-tree, reading a symbol table node at // every leaf. func (f *hdf5File) treeLinks(addr uint64, heap []byte, seen map[uint64]bool, out *[]hdf5Link, depth int) error { const name = "LoadHDF5" if depth > 32 { return base.Errf("%s: the group B-tree is more than 32 levels deep", name) } if addr == math.MaxUint64 || seen[addr] { return nil } seen[addr] = true // The version 1 B-tree header is the signature, the type, the level // and the entry count (eight bytes) plus a left and a right sibling // address, so it is 8+2*offSize wide, not the 24 an 8-byte-address // file happens to make it. Keying the entries off a pinned 24 read // a 4/4 file's first key as a sibling address and lost the node. node := f.bytes(addr, uint64(8+2*f.offSize)) if node == nil || !bytes.Equal(node[:4], hdf5Tree) { return base.Errf("%s: no B-tree node at %d", name, addr) } nodeType := node[4] level := node[5] entries := int(binary.LittleEndian.Uint16(node[6:])) if nodeType != 0 { return base.Errf("%s: B-tree node type %d is not a group", name, nodeType) } // Each entry is a key of one address size followed by a child // address; the node ends with one trailing key. entrySize := f.lenSize + f.offSize base := int(addr) + 8 + 2*f.offSize for i := range entries { child := f.at(base + i*entrySize + f.lenSize) if level == 0 { if err := f.symbolNodeLinks(child, heap, seen, out); err != nil { return err } continue } if err := f.treeLinks(child, heap, seen, out, depth+1); err != nil { return err } } return nil } // symbolNodeLinks reads one symbol table node's entries. func (f *hdf5File) symbolNodeLinks(addr uint64, heap []byte, seen map[uint64]bool, out *[]hdf5Link) error { const name = "LoadHDF5" if addr == math.MaxUint64 || seen[addr] { return nil } seen[addr] = true node := f.bytes(addr, 8) if node == nil || !bytes.Equal(node[:4], hdf5SymbolNode) { return base.Errf("%s: no symbol table node at %d", name, addr) } count := int(binary.LittleEndian.Uint16(node[6:])) // An entry is the heap offset (of the length size), the object // address (of the address size), the cache type, a reserved word // and the 16-byte scratch pad: lenSize+offSize+24, not the 40 an // 8/8 file happens to make it. A pinned 40 walked a 4/4 node's // entries into the middle of the first one and read its second // link's name from the wrong heap offset. entrySize := f.lenSize + f.offSize + 24 start := int(addr) + 8 for i := range count { off := start + i*entrySize nameOff := f.length(off) objAddr := f.at(off + f.lenSize) linkName, err := heapString(heap, nameOff) if err != nil { return err } *out = append(*out, hdf5Link{name: linkName, address: objAddr}) } return nil } // localHeap reads a local heap's data segment, the pool of // null-terminated names the symbol table entries point into. func (f *hdf5File) localHeap(addr uint64) ([]byte, error) { const name = "LoadHDF5" // The header is the signature and version, the data segment size // and the free-list head (both of the length size), then the data // segment address of the address size. Sizing the header with the // address size instead (8+3*offSize) sliced the address out of a // header that a 4/8 file stores in 20 bytes at [24:], which // panicked; every field here is sized by its own kind. head := f.bytes(addr, uint64(8+2*f.lenSize+f.offSize)) if head == nil || !bytes.Equal(head[:4], hdf5LocalHeap) { return nil, base.Errf("%s: no local heap at %d", name, addr) } dataAddr := f.at2(head[8+2*f.lenSize:]) size := f.length(int(addr) + 8) // data segment size, then the free list head seg := f.bytes(dataAddr, size) if seg == nil { return nil, base.Errf("%s: the local heap at %d lies outside the file", name, addr) } return seg, nil } // heapString reads the null-terminated string at an offset in the heap. func heapString(heap []byte, off uint64) (string, error) { if off >= uint64(len(heap)) { return "", base.Errf("LoadHDF5: a symbol table entry names a name at heap offset %d, past the %d-byte heap", off, len(heap)) } end := bytes.IndexByte(heap[off:], 0) if end < 0 { return "", base.Errf("LoadHDF5: the name at heap offset %d is not terminated", off) } return string(heap[off : off+uint64(end)]), nil } // attributesOf collects the attributes carried in an object header. func (f *hdf5File) attributesOf(msgs []hdf5Message) map[string]string { out := map[string]string{} for _, m := range msgs { if m.typ != hdf5MsgAttribute { continue } linkName, value, ok := f.decodeAttribute(m.data) if ok { out[linkName] = value } } return out } // hdf5Type is the parsed datatype of a dataset or attribute: enough of // it to read the values. type hdf5Type struct { class byte size int signed bool width int // fixed-point: bytes isFloat bool isBool bool // a boolean enumeration: lands core.Bool bitOrder byte } // hdf5ClassName names a datatype class for a refusal message: the // number alone tells whoever wrote the file little about what the // reader missed. func hdf5ClassName(class byte) string { switch class { case 2: return "date and time" case 4: return "bit field" case 5: return "opaque" case 6: return "compound" case 7: return "reference" case 8: return "enumeration" case 9: return "variable-length" } return "unknown" } // decodeType parses a datatype message (version 1, which is what the // default library version writes). Strings are refused for datasets and // accepted for attributes, where a string is exactly what is wanted. // An enumeration is accepted only in the boolean convention HDF5 // writers carry booleans in, and lands core.Bool; every other // enumeration and every bit field is refused by name. func decodeType(m []byte, allowStrings bool) (hdf5Type, error) { const name = "LoadHDF5" if len(m) < 8 { return hdf5Type{}, base.Errf("%s: datatype message is truncated", name) } // The first byte carries the version in its high nibble and the // datatype class in its low one. classAndVersion := m[0] version := classAndVersion >> 4 class := classAndVersion & 0x0f if version != 1 { return hdf5Type{}, base.Errf("%s: datatype message version %d is not supported", name, version) } size := int(binary.LittleEndian.Uint32(m[4:])) t := hdf5Type{class: class, size: size, bitOrder: m[1] >> 0} // The class bit field's bit 0 is the byte order: 0 little-endian, // anything else big-endian. Every element below is decoded as // little-endian, so a big-endian datatype, on a dataset or on an // attribute alike, would come back as byte-swapped noise with no // error; it is refused by name instead. if (class == 0 || class == 1) && m[1]&0x01 != 0 { return hdf5Type{}, base.Errf("%s: big-endian datatypes are not supported; rewrite the file in the little-endian order", name) } switch class { case 0: // fixed-point if size != 1 && size != 2 && size != 4 && size != 8 { return hdf5Type{}, base.Errf("%s: %d-byte fixed-point values are not supported", name, size) } t.signed = m[1]&0x08 != 0 t.width = size case 1: // floating-point if len(m) < 20 { return hdf5Type{}, base.Errf("%s: floating-point datatype message is truncated", name) } switch size { case 4, 8: default: return hdf5Type{}, base.Errf("%s: %d-byte floating-point values are not supported", name, size) } t.isFloat = true t.width = size case 3: // string if !allowStrings { return hdf5Type{}, base.Errf("%s: string datasets are not supported (only numeric arrays are read)", name) } t.width = size if t.width <= 0 { return hdf5Type{}, base.Errf("%s: a string attribute declares %d bytes", name, size) } case 8: // enumeration: only the boolean convention is read if err := hdf5EnumBool(m); err != nil { return hdf5Type{}, err } t.isBool = true t.width = size case 9: // variable-length: the text lives in a global heap collection if !allowStrings { return hdf5Type{}, base.Errf("%s: variable-length datasets are not supported (only numeric arrays are read)", name) } t.class = 9 t.width = size // the descriptor: length, heap address, object index default: return hdf5Type{}, base.Errf("%s: datatype class %d (%s) is not supported", name, class, hdf5ClassName(class)) } return t, nil } // hdf5EnumBool verifies that an enumeration datatype message carries // the boolean convention HDF5 writers store booleans in: a one-byte // unsigned base type whose member values are a subset of {0, 1}. The // member names are irrelevant to the values, which is what the landing // keys on, and any other enumeration is refused by name. // // The message layout comes from the HDF5 file format specification's // datatype message, enumeration class: the class bit field carries the // member count in its low sixteen bits, and the properties hold the // base type as a complete datatype message, then the member names // (each a NUL-terminated string stored in a multiple of eight bytes, // padded from its own field's start) and the packed member values. The // base type here is a complete version 1 fixed-point message: its own // eight-byte header plus the bit offset and bit precision the // specification's fixed-point property table defines, twelve bytes in // total, the same shape the reader already requires of the twenty-byte // version 1 floating-point message. func hdf5EnumBool(m []byte) error { const name = "LoadHDF5" members := int(binary.LittleEndian.Uint16(m[1:])) if m[3] != 0 { return base.Errf("%s: enumeration datatype class 8 carries unknown bit field bits %#02x", name, m[3]) } if members < 1 { return base.Errf("%s: enumeration datatype class 8 declares %d members", name, members) } if size := binary.LittleEndian.Uint32(m[4:]); size != 1 { return base.Errf("%s: enumeration datatype class 8 with %d-byte values is not supported; only the one-byte unsigned boolean convention is read", name, size) } // The base type: a complete version 1 fixed-point message. if len(m) < 8+12 { return base.Errf("%s: enumeration datatype message is truncated", name) } b := m[8:] if bclass := b[0] & 0x0f; b[0]>>4 != 1 || bclass != 0 { return base.Errf("%s: enumeration datatype class 8 with a base type of class %d version %d is not supported; only the one-byte unsigned boolean convention is read", name, bclass, b[0]>>4) } if b[1]&0x01 != 0 { return base.Errf("%s: big-endian datatypes are not supported; rewrite the file in the little-endian order", name) } if b[1]&0x08 != 0 { return base.Errf("%s: enumeration datatype class 8 with a signed base type is not supported; only the one-byte unsigned boolean convention is read", name) } if bsize := binary.LittleEndian.Uint32(b[4:]); bsize != 1 { return base.Errf("%s: enumeration datatype class 8 with a base type of %d bytes is not supported; only the one-byte unsigned boolean convention is read", name, bsize) } // The names: each field is a NUL-terminated string padded, from its // own start, to a multiple of eight bytes. The walk only needs the // fields' extent; the values follow them. p := 8 + 12 for i := range members { end := -1 for j := p; j < len(m); j++ { if m[j] == 0 { end = j break } } if end < 0 { return base.Errf("%s: enumeration datatype message ends inside member name %d", name, i+1) } p += alignUp(end-p+1, 8) if p > len(m) { return base.Errf("%s: enumeration datatype message ends inside the member names", name) } } // The values: packed, one base-width byte each. if members > len(m)-p { return base.Errf("%s: enumeration datatype message holds %d member values, fewer than its %d members", name, len(m)-p, members) } for i := range members { if v := m[p+i]; v > 1 { return base.Errf("%s: enumeration datatype class 8 declares member value %d, outside the boolean convention {0, 1}", name, v) } } return nil } // decodeDataspace parses a dataspace message into the dataset's shape. // Version 1 is what the default library version writes, version 2 what // the "latest" one writes: it drops the reserved bytes to one and // always stores the dimensions in eight bytes. func decodeDataspace(m []byte, lenSize int) ([]int, error) { const name = "LoadHDF5" if len(m) < 8 { return nil, base.Errf("%s: dataspace message is truncated", name) } version := m[0] rank := int(m[1]) flags := m[2] if version != 1 && version != 2 { return nil, base.Errf("%s: dataspace message version %d is not supported", name, version) } if rank == 0 { return []int{1}, nil // a scalar: one element, no dimensions } pos, dimSize := 8, lenSize if version == 2 { pos, dimSize = 4, 8 } shape := make([]int, rank) for i := range rank { if pos+dimSize > len(m) { return nil, base.Errf("%s: dataspace message is truncated", name) } shape[i] = int(uint64At(m[pos:], dimSize)) pos += dimSize } if flags&0x02 != 0 { return nil, base.Errf("%s: datasets with a dimension permutation are not supported", name) } if flags&0x01 != 0 { // Maximum dimensions are not the shape; nothing to do with them. pos += rank * dimSize } return shape, nil } func uint64At(b []byte, size int) uint64 { v, _ := readUintLE(b, size) return v } // hdf5Layout is a dataset's storage description. For chunked storage // the message records one more dimension than the dataset has: the // dimensions are the chunk's shape followed by the element size in // bytes, and the chunk B-tree keys carry the same extra offset. type hdf5Layout struct { class byte addr uint64 size uint64 compact []byte dims []int } func decodeLayout(m []byte, offSize int) (hdf5Layout, error) { const name = "LoadHDF5" if len(m) < 2 { return hdf5Layout{}, base.Errf("%s: data layout message is truncated", name) } version := m[0] class := m[1] switch version { case 3, 4: default: return hdf5Layout{}, base.Errf("%s: data layout message version %d is not supported", name, version) } l := hdf5Layout{class: class} switch class { case 0: // compact: the data lives in the message if len(m) < 4 { return hdf5Layout{}, base.Errf("%s: compact layout message is truncated", name) } size := int(binary.LittleEndian.Uint16(m[2:])) if 4+size > len(m) { return hdf5Layout{}, base.Errf("%s: compact data runs past the message", name) } l.compact = m[4 : 4+size] case 1: // contiguous if len(m) < 2+offSize+8 { return hdf5Layout{}, base.Errf("%s: contiguous layout message is truncated", name) } l.addr = uint64At(m[2:], offSize) l.size = uint64At(m[2+offSize:], 8) case 2: // chunked if len(m) < 3+offSize { return hdf5Layout{}, base.Errf("%s: chunked layout message is truncated", name) } rank := int(m[2]) p := 3 l.addr = uint64At(m[p:], offSize) p += offSize // Version 4 stores one flag byte before the chunk dimensions // and widens them to 8 bytes when the flag asks for it. dimSize := 4 if version == 4 { if p >= len(m) { return hdf5Layout{}, base.Errf("%s: chunked layout message is truncated", name) } if m[p]&0x01 != 0 { dimSize = 8 } p++ } l.dims = make([]int, rank) for i := range rank { if p+dimSize > len(m) { return hdf5Layout{}, base.Errf("%s: chunk dimensions run past the message", name) } l.dims[i] = int(uint64At(m[p:], dimSize)) p += dimSize } default: return hdf5Layout{}, base.Errf("%s: storage class %d is not supported", name, class) } return l, nil } // hdf5Filter is one entry of a filter pipeline. type hdf5Filter struct { id uint16 values []uint32 } func decodeFilters(m []byte) ([]hdf5Filter, error) { const name = "LoadHDF5" if len(m) < 2 { return nil, base.Errf("%s: filter pipeline message is truncated", name) } version := m[0] if version != 1 { // Only version 1's layout is known; parsing another version on // this layout would fail far away with an unrelated truncation // error instead of naming the defect. return nil, base.Errf("%s: unsupported filter pipeline message version %d", name, version) } count := int(m[1]) p := 2 p += 6 // reserved out := make([]hdf5Filter, 0, count) for range count { if p+8 > len(m) { return nil, base.Errf("%s: filter pipeline message is truncated", name) } filter := hdf5Filter{id: binary.LittleEndian.Uint16(m[p:])} nameLen := int(binary.LittleEndian.Uint16(m[p+2:])) nValues := int(binary.LittleEndian.Uint16(m[p+6:])) p += 8 if p+nameLen > len(m) { return nil, base.Errf("%s: filter name runs past the message", name) } p += nameLen p = alignUp(p, 8) for range nValues { if p+4 > len(m) { return nil, base.Errf("%s: filter values run past the message", name) } filter.values = append(filter.values, binary.LittleEndian.Uint32(m[p:])) p += 4 } p = alignUp(p, 8) switch filter.id { case hdf5FilterDeflate, hdf5FilterShuffle, hdf5FilterFletcher32: default: return nil, base.Errf("%s: filter %d is not supported", name, filter.id) } out = append(out, filter) } return out, nil } // dataset reads one dataset's values. func (f *hdf5File) dataset(path string, msgs []hdf5Message, st *hdf5WalkState) (HDF5Dataset, error) { const name = "LoadHDF5" var ( shape []int dtype hdf5Type layout hdf5Layout filters []hdf5Filter haveType bool haveSpace bool haveLay bool ) for _, m := range msgs { switch m.typ { case hdf5MsgDataspace: s, err := decodeDataspace(m.data, f.lenSize) if err != nil { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err) } shape, haveSpace = s, true case hdf5MsgDatatype: t, err := decodeType(m.data, false) if err != nil { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err) } dtype, haveType = t, true case hdf5MsgDataLayout: l, err := decodeLayout(m.data, f.offSize) if err != nil { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err) } layout, haveLay = l, true case hdf5MsgFilterPipeline: fs, err := decodeFilters(m.data) if err != nil { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, err) } filters = fs } } if !haveSpace || !haveType || !haveLay { return HDF5Dataset{}, base.Errf("%s: dataset %q is missing its %s", name, path, missingOf(haveSpace, haveType, haveLay)) } if layout.class == 2 { // A latest-version file indexes its chunks with the version 2 // B-tree, which this reader does not walk; a classic file's // chunks hang off the version 1 tree below. if f.superblock >= 2 { return HDF5Dataset{}, base.Errf("%s: dataset %q is chunked through a version 2 B-tree, which is not supported; rewrite the file with the default library version", name, path) } // The chunked message's last dimension is the element size, not // a chunk dimension: the rank must be one less, and the value // must agree with the datatype, which is a strong check that // both were read correctly. if len(layout.dims) != len(shape)+1 { return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d chunk dimensions for rank %d", name, path, len(layout.dims), len(shape)) } if got := layout.dims[len(layout.dims)-1]; got != dtype.width { return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d-byte chunk elements against a %d-byte datatype", name, path, got, dtype.width) } layout.dims = layout.dims[:len(shape)] } // Fixed-point data lands by its stored width and signedness; an // unsigned 64-bit value has no exact core dtype and is refused // before any storage is read, with the same message the value // decode reported it with. if !dtype.isFloat && !dtype.signed && dtype.width == 8 { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, base.Errf("unsigned 64-bit integers have no exact core dtype")) } // The size arithmetic bounds every extent against the budget before // it is multiplied: a hostile dataspace must fail the cap, not wrap // the product into a small want that passes it. The validated byte // count is what every storage class below works from. want, err := hdf5ByteExtent(shape, dtype.width, hdf5MaxDatasetBytes) if err != nil { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w; larger datasets need mmap", name, path, err) } n := int(want) // The per-dataset cap bounds one dataset; the walk's aggregate // bounds their sum, so a file declaring thousands of cap-fitting // datasets refuses here instead of allocating the sum. The sum is // charged past the value bytes: every accepted dataset also costs // its result structure and its copied path, which the envelope and // the path length stand for, so empty datasets pay too. charge := int64(n) + int64(len(st.path)) + hdf5DatasetEnvelope if charge > st.budget { return HDF5Dataset{}, base.Errf("%s: dataset %q declares %d bytes, the file's datasets are past the %d byte read budget; larger files need mmap", name, path, n, hdf5MaxDatasetBytes) } st.budget -= charge if layout.class == 2 { values, cerr := f.chunkedArray(path, shape, dtype, layout, filters, n) if cerr != nil { return HDF5Dataset{}, cerr } return HDF5Dataset{Path: path, Shape: shape, Values: values}, nil } raw, rerr := f.rawData(path, shape, dtype, layout, n) if rerr != nil { return HDF5Dataset{}, rerr } values, aerr := arrayFromRaw(raw, dtype, shape) if aerr != nil { return HDF5Dataset{}, base.Errf("%s: dataset %q: %w", name, path, aerr) } return HDF5Dataset{Path: path, Shape: shape, Values: values}, nil } func missingOf(haveSpace, haveType, haveLay bool) string { switch { case !haveSpace: return "dataspace" case !haveType: return "datatype" case !haveLay: return "storage layout" } return "messages" } // rawData gathers a dataset's bytes for the storage classes that hold // them contiguously: compact storage is the message itself and // contiguous storage is a span of the file. Chunked storage decodes // into the values array directly, in chunkedArray. n is the dataset's // byte size, validated against the read budget by the caller. func (f *hdf5File) rawData(path string, shape []int, dtype hdf5Type, layout hdf5Layout, n int) ([]byte, error) { const name = "LoadHDF5" switch layout.class { case 0: if len(layout.compact) < n { return nil, base.Errf("%s: dataset %q holds %d compact bytes, %d are needed", name, path, len(layout.compact), n) } return layout.compact[:n], nil case 1: if layout.addr == math.MaxUint64 { // The undefined address means storage was never allocated: // for an empty dataset that is the whole answer (the fill // value is the data, as the chunked branch already models), // and for a non-empty one the bytes simply do not exist. if n == 0 { return []byte{}, nil } return nil, base.Errf("%s: dataset %q declares storage that was never allocated", name, path) } raw := f.bytes(layout.addr, uint64(n)) if raw == nil { return nil, base.Errf("%s: dataset %q lies outside the file", name, path) } return raw, nil } return nil, base.Errf("%s: dataset %q has an unsupported storage class", name, path) } // hdf5ChunkStage carries what one dataset's chunk decode reuses across // its chunks: the inflate output, the unshuffle staging, the inflate // reader over the file's bytes, its limited view, and the index // vectors the placement walk uses. A decode is serial, so the stage // belongs to one call and nothing here is shared between concurrent // loads. type hdf5ChunkStage struct { inflate []byte unshuf []byte zr io.ReadCloser zrReset zlib.Resetter br bytes.Reader lim io.LimitedReader offsets []uint64 vec []int seen map[uint64]bool } // chunkedArray reassembles a chunked dataset through its version 1 // B-tree, running each chunk back through the filter pipeline and // decoding each cell into the values array as it lands. bytes is the // dataset's size, already validated against the read budget by the // caller, so this function never re-derives it in int. The staging // buffers, the inflate reader and the index vectors are one stage, // reused across every chunk of the dataset. func (f *hdf5File) chunkedArray(path string, shape []int, dtype hdf5Type, layout hdf5Layout, filters []hdf5Filter, bytes int) (*core.Array, error) { const name = "LoadHDF5" rank := len(shape) if len(layout.dims) != rank { return nil, base.Errf("%s: dataset %q declares %d chunk dimensions for rank %d", name, path, len(layout.dims), rank) } for _, c := range layout.dims { // A chunk holds at least one element of the datatype; a zero or // negative declared dimension is a hostile header, and both the // strides below and the placement's overlap box degenerate on it. if c <= 0 { return nil, base.Errf("%s: dataset %q declares a %d-element chunk dimension", name, path, c) } } // One chunk may not exceed the reader budget on its own, because the // filters inflate into a buffer of exactly this size. The product is // bounded extent by extent, so a hostile chunk dimension fails the // cap instead of wrapping it. chunkBytes, err := hdf5ByteExtent(layout.dims, dtype.width, hdf5MaxDatasetBytes) if err != nil { return nil, base.Errf("%s: dataset %q declares a chunk: %w", name, path, err) } cb := int(chunkBytes) // The destination array and the per-cell decode carry the value // mapping arrayFromRaw applies to contiguous bytes; the cells land // in the payload as the walk places them, so no assembled copy of // the dataset's bytes exists. Fixed-point cells land by stored width // and signedness, exactly as the contiguous path lands them, and a // boolean cell outside the enumeration's {0, 1} values is refused // rather than coerced. nElems := 0 if dtype.width > 0 { nElems = bytes / dtype.width } var put func(data []byte, ci, oi int) error var arr *core.Array switch { case dtype.isFloat && dtype.width == 8: vals := make([]float64, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = math.Float64frombits(binary.LittleEndian.Uint64(data[ci*8:])) return nil } arr, err = core.FloatsFromArray(vals, shape...) case dtype.isFloat && dtype.width == 4: vals := make([]float32, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = math.Float32frombits(binary.LittleEndian.Uint32(data[ci*4:])) return nil } // The aliasing constructor: the decode fills the array's own // payload through vals, which a copying constructor would have // left behind as a separate slice. arr, err = core.FromFloat32Slice(vals, shape...) case dtype.isBool: vals := make([]bool, nElems) put = func(data []byte, ci, oi int) error { b := data[ci] if b > 1 { return base.Errf("a boolean value of %d is outside the members 0 and 1", b) } vals[oi] = b != 0 return nil } arr, err = core.BoolsFromArray(vals, shape...) case !dtype.signed && dtype.width == 8: // Unreachable through LoadHDF5, which refuses the datatype // first; the decode carries the same refusal for direct callers. return nil, base.Errf("%s: dataset %q: %w", name, path, base.Errf("unsigned 64-bit integers have no exact core dtype")) case dtype.width == 1 && dtype.signed: vals := make([]int8, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = int8(data[ci]) return nil } arr, err = core.Int8sFromArray(vals, shape...) case dtype.width == 1: vals := make([]uint8, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = data[ci] return nil } arr, err = core.Uint8sFromArray(vals, shape...) case dtype.width == 2 && dtype.signed: vals := make([]int16, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = int16(binary.LittleEndian.Uint16(data[ci*2:])) return nil } arr, err = core.Int16sFromArray(vals, shape...) case dtype.width == 2: vals := make([]uint16, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = binary.LittleEndian.Uint16(data[ci*2:]) return nil } arr, err = core.Uint16sFromArray(vals, shape...) case dtype.width == 4 && dtype.signed: vals := make([]int32, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = int32(binary.LittleEndian.Uint32(data[ci*4:])) return nil } arr, err = core.Int32sFromArray(vals, shape...) case dtype.width == 4: vals := make([]uint32, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = binary.LittleEndian.Uint32(data[ci*4:]) return nil } arr, err = core.Uint32sFromArray(vals, shape...) default: // Signed 64-bit, the only fixed-point width left. vals := make([]int64, nElems) put = func(data []byte, ci, oi int) error { vals[oi] = int64(binary.LittleEndian.Uint64(data[ci*8:])) return nil } arr, err = core.IntsFromArray(vals, shape...) } if err != nil { return nil, base.Errf("%s: dataset %q: %w", name, path, err) } if layout.addr == math.MaxUint64 { return arr, nil // no chunks written: the fill value is the data } stage := &hdf5ChunkStage{ seen: map[uint64]bool{}, offsets: make([]uint64, rank), vec: make([]int, 5*rank), } seenNodes := map[uint64]bool{} return arr, f.chunkTreeWalk(layout.addr, layout, rank, seenNodes, func(addr uint64, offsets []uint64, mask uint32, size int) error { if stage.seen[addr] { return base.Errf("%s: dataset %q has two chunks at the same address", name, path) } stage.seen[addr] = true stored := f.bytes(addr, uint64(size)) if stored == nil { return base.Errf("%s: dataset %q has a chunk outside the file", name, path) } data, ferr := runFilters(stored, filters, mask, dtype, cb, stage) if ferr != nil { return base.Errf("%s: dataset %q: %w", name, path, ferr) } return placeCells(shape, layout.dims, offsets, data, dtype.width, cb, nElems, stage.vec, put) }, 0, stage.offsets) } // placeCells walks the overlap of one chunk and the dataset, calling // put for every cell with the chunk-local cell index and the output // element index; the decode writes the value as the cell lands, and a // cell put refuses stops the walk with that error. // chunkBytes is the byte size of one complete chunk, computed by the // caller with every dimension bounded against the reader budget, so // the strides cannot wrap and a chunk is known to hold whole elements. // The loop walks the overlap box in output coordinates, so the work is // proportional to the placed elements: a chunk hanging past the // shape's edge is padding the file may hold but the array has no room // for, and one starting past an axis's end places nothing at all. // Walking the whole declared chunk instead would let a hostile chunk // dimension pin the reader for hours without touching a byte. vec // carries the box bounds, the row-major strides of chunk and output, // and the counter: five vectors of the rank, reused across chunks. func placeCells(shape, chunkDim []int, offsets []uint64, data []byte, width, chunkBytes, maxOut int, vec []int, put func(data []byte, ci, oi int) error) error { const name = "LoadHDF5" if len(data) == 0 { return nil } rank := len(shape) if len(offsets) != rank { return base.Errf("%s: a chunk carries %d offsets for rank %d", name, len(offsets), rank) } // The chunk must arrive complete: a truncated deflate stream would // otherwise read past its data below. if len(data) < chunkBytes { return base.Errf("%s: a chunk holds %d bytes, a full chunk of %d is needed", name, len(data), chunkBytes) } lo := vec[:rank] hi := vec[rank : 2*rank] chunkStride := vec[2*rank : 3*rank] outStride := vec[3*rank : 4*rank] count := vec[4*rank : 5*rank] for d := range rank { if offsets[d] >= uint64(shape[d]) { return nil } lo[d] = int(offsets[d]) hi[d] = min(lo[d]+chunkDim[d], shape[d]) if lo[d] == hi[d] { return nil } } // Row-major strides of the chunk and of the output. Both products // are bounded: the chunk's by chunkBytes and the array's by the // dataset's own validated byte count. cs, os := 1, 1 for i := rank - 1; i >= 0; i-- { chunkStride[i], outStride[i] = cs, os cs *= chunkDim[i] os *= shape[i] } // The element walk indexes both sides with a computed offset. The // chunk's data may be a view of the file buffer, whose capacity // runs past the chunk, so a stride that reached outside the chunk // would read the file's other bytes silently instead of failing: // the two limits are therefore checked explicitly, once, in element // units. maxChunk := len(data) / width copy(count, lo) for { // The element's output coordinate and chunk-local coordinate. oi, ci := 0, 0 for d := range rank { oi += count[d] * outStride[d] ci += (count[d] - lo[d]) * chunkStride[d] } if oi < 0 || oi >= maxOut || ci < 0 || ci >= maxChunk { return base.Errf("%s: a chunk element lands at element %d of the %d-element array and %d of the %d-element chunk", name, oi, maxOut, ci, maxChunk) } if err := put(data, ci, oi); err != nil { return err } // Advance the odometer over the overlap box. d := rank - 1 for d >= 0 { count[d]++ if count[d] < hi[d] { break } count[d] = lo[d] d-- } if d < 0 { return nil } } } // chunkTree walks the chunk B-tree, calling place for every chunk, // with a fresh per-entry offset vector. The decode reuses one vector // across the whole walk through chunkTreeWalk instead. func (f *hdf5File) chunkTree(addr uint64, layout hdf5Layout, rank int, nodes map[uint64]bool, place func(uint64, []uint64, uint32, int) error, depth int) error { return f.chunkTreeWalk(addr, layout, rank, nodes, place, depth, make([]uint64, rank)) } // chunkTreeWalk is the walk itself, with the per-entry offset vector // supplied by the caller, so a decode allocates one for the dataset // rather than one per chunk entry; a place callback consumes the // offsets before it returns. nodes carries the addresses of the // interior nodes already visited: a legal B-tree never revisits one, // and without the set a hostile file whose node lists itself among its // children would multiply the walk into entries^depth visits before // the depth guard ever fires. func (f *hdf5File) chunkTreeWalk(addr uint64, layout hdf5Layout, rank int, nodes map[uint64]bool, place func(uint64, []uint64, uint32, int) error, depth int, offsets []uint64) error { const name = "LoadHDF5" if depth > 32 { return base.Errf("%s: the chunk B-tree is more than 32 levels deep", name) } if nodes[addr] { return base.Errf("%s: the chunk B-tree revisits node %d", name, addr) } nodes[addr] = true // The version 1 B-tree header is 8+2*offSize wide, the signature, // type, level and entry count plus the two sibling addresses. node := f.bytes(addr, uint64(8+2*f.offSize)) if node == nil || !bytes.Equal(node[:4], hdf5Tree) { return base.Errf("%s: no chunk B-tree node at %d", name, addr) } if node[4] != 1 { return base.Errf("%s: B-tree node type %d is not a chunk tree", name, node[4]) } level := node[5] entries := int(binary.LittleEndian.Uint16(node[6:])) // The node header is 8+2*offSize wide, as in the group B-tree; the // first key starts behind it. nodeHeader := 8 + 2*f.offSize // A v1 chunk key is the chunk size, the filter mask and one offset // per dimension of the chunk *plus* the element size, which is why // the key has one more offset than the dataset has axes; each entry // is a key plus a child address. keySize := 8 + 8*(rank+1) entrySize := keySize + f.offSize start := int(addr) + nodeHeader for i := range entries { p := start + i*entrySize size := int(f.u64At(p, 4)) mask := uint32(f.u64At(p+4, 4)) for d := range rank { offsets[d] = f.length(p + 8 + d*8) } child := f.at(p + keySize) if level == 0 { if err := place(child, offsets, mask, size); err != nil { return err } continue } if err := f.chunkTreeWalk(child, layout, rank, nodes, place, depth+1, offsets); err != nil { return err } } return nil } // runFilters undoes the filter pipeline for one chunk, in reverse // order; a filter whose bit is set in the mask did not run on it. The // returned bytes are either the stored chunk untouched or one of the // stage's buffers, valid until the stage handles its next chunk. func runFilters(data []byte, filters []hdf5Filter, mask uint32, dtype hdf5Type, chunkBytes int, stage *hdf5ChunkStage) ([]byte, error) { const name = "LoadHDF5" out := data for i, filter := range slices.Backward(filters) { if mask&(1< want { return nil, base.Errf("the deflate stream inflates past the chunk size") } return out, nil } // unshuffleInto undoes the shuffle filter into out, which holds the // same length as data: the filter transposes the bytes of the // elements, so the first block holds every element's first byte. The // transpose runs over tiles of rows: a tile is a short contiguous run // of out, and the reads that fill it walk one row of the source // sequentially, so neither stream strides across the whole chunk per // column. Every byte still lands where the flat transpose put it. func unshuffleInto(out, data []byte, width int) { n := len(data) / width const tileRows = 64 for j0 := 0; j0 < n; j0 += tileRows { j1 := min(j0+tileRows, n) for i := range width { src := data[i*n+j0 : i*n+j1] dst := out[j0*width+i:] for k, b := range src { dst[k*width] = b } } } } // checkFletcher32 verifies the fletcher32 checksum HDF5 appends to a // filtered chunk. func checkFletcher32(data []byte) error { body, sum := data[:len(data)-4], binary.LittleEndian.Uint32(data[len(data)-4:]) if got := fletcher32(body); got != sum { return base.Errf("the fletcher32 checksum does not match: %08x, want %08x", got, sum) } return nil } // arrayFromRaw builds the array for a dataset from its bytes. The // decoded buffer is exactly one element per value, so the array takes // it over instead of copying it a second time. Fixed-point data lands // the core dtype its stored width and signedness declare, a boolean // enumeration lands Bool, and a boolean cell outside the members 0 and // 1 is refused rather than coerced. func arrayFromRaw(raw []byte, dtype hdf5Type, shape []int) (*core.Array, error) { switch { case dtype.isFloat && dtype.width == 8: vals := make([]float64, len(raw)/8) for i := range vals { vals[i] = math.Float64frombits(binary.LittleEndian.Uint64(raw[i*8:])) } return core.FloatsFromArray(vals, shape...) case dtype.isFloat && dtype.width == 4: vals := make([]float32, len(raw)/4) for i := range vals { vals[i] = math.Float32frombits(binary.LittleEndian.Uint32(raw[i*4:])) } return core.FromFloat32s(vals, shape...) case dtype.isBool: vals := make([]bool, len(raw)) for i, b := range raw { if b > 1 { return nil, base.Errf("a boolean value of %d is outside the members 0 and 1", b) } vals[i] = b != 0 } return core.BoolsFromArray(vals, shape...) case !dtype.signed && dtype.width == 8: return nil, base.Errf("unsigned 64-bit integers have no exact core dtype") } switch { case dtype.width == 1 && dtype.signed: vals := make([]int8, len(raw)) for i, b := range raw { vals[i] = int8(b) } return core.Int8sFromArray(vals, shape...) case dtype.width == 1: // raw may be a view of the file buffer; the payload copies out // of it before the array takes the copy over. vals := make([]uint8, len(raw)) copy(vals, raw) return core.Uint8sFromArray(vals, shape...) case dtype.width == 2 && dtype.signed: vals := make([]int16, len(raw)/2) for i := range vals { vals[i] = int16(binary.LittleEndian.Uint16(raw[i*2:])) } return core.Int16sFromArray(vals, shape...) case dtype.width == 2: vals := make([]uint16, len(raw)/2) for i := range vals { vals[i] = binary.LittleEndian.Uint16(raw[i*2:]) } return core.Uint16sFromArray(vals, shape...) case dtype.width == 4 && dtype.signed: vals := make([]int32, len(raw)/4) for i := range vals { vals[i] = int32(binary.LittleEndian.Uint32(raw[i*4:])) } return core.Int32sFromArray(vals, shape...) case dtype.width == 4: vals := make([]uint32, len(raw)/4) for i := range vals { vals[i] = binary.LittleEndian.Uint32(raw[i*4:]) } return core.Uint32sFromArray(vals, shape...) case dtype.width == 8: // Signed 64-bit: unsigned was refused above, and decodeType // admits no other fixed-point width. vals := make([]int64, len(raw)/8) for i := range vals { vals[i] = int64(binary.LittleEndian.Uint64(raw[i*8:])) } return core.IntsFromArray(vals, shape...) } return nil, base.Errf("%d-byte fixed-point values have no core dtype", dtype.width) } // decodeAttribute reads one attribute message: its name and a textual // rendering of its value, the shape the package's attribute maps use. func (f *hdf5File) decodeAttribute(m []byte) (string, string, bool) { if len(m) < 8 || m[0] != 1 { return "", "", false } nameSize := int(binary.LittleEndian.Uint16(m[2:])) typeSize := int(binary.LittleEndian.Uint16(m[4:])) spaceSize := int(binary.LittleEndian.Uint16(m[6:])) p := 8 if p+nameSize+typeSize+spaceSize > len(m) { return "", "", false } name := strings.TrimRight(string(m[p:p+nameSize]), "\x00") // The name field is padded so the datatype starts on an eight-byte // boundary. The padding is inside the message, so the datatype is // checked against the body's own length: the slice underneath is a // view of the file buffer, whose capacity runs past the body. p = alignUp(p+nameSize, 8) if p+typeSize > len(m) { return "", "", false } dtype, err := decodeType(m[p:p+typeSize], true) if err != nil { return "", "", false } // The datatype message is padded to an eight-byte boundary before // the dataspace begins. p = alignUp(p+typeSize, 8) if p+spaceSize > len(m) { return "", "", false } // The dataspace's dimension fields are sized by the file's own // length size, exactly as the dataset path's are. shape, err := decodeDataspace(m[p:p+spaceSize], f.lenSize) if err != nil { return "", "", false } p += spaceSize // The value's byte count is bounded extent by extent against what is // left of the message: a hostile dataspace must fail here, not wrap // the product into a negative that would slip past a check made // afterwards and then size an allocation. size, err := hdf5ByteExtent(shape, dtype.width, uint64(len(m)-p)) if err != nil { return "", "", false } raw := m[p : p+int(size)] if dtype.class == 3 { // A fixed-length string: the bytes are the text. return name, strings.TrimRight(string(raw), "\x00"), true } if dtype.class == 9 { // A variable-length string: the attribute holds a descriptor // naming a global heap object, and that object holds the text. // The descriptor is four bytes of length, the collection // address, then the object index. if len(raw) < 8+f.offSize { return "", "", false } length := int(binary.LittleEndian.Uint32(raw)) heapAddr := uint64At(raw[4:], f.offSize) index := binary.LittleEndian.Uint32(raw[4+f.offSize:]) text, err := f.heapString(heapAddr, index, length) if err != nil { return "", "", false } return name, text, true } // The element count the numeric rendering below walks, derived from // the validated byte count rather than from a second multiplication. count := 0 if dtype.width > 0 { count = len(raw) / dtype.width } vals := make([]string, 0, count) for i := range count { cell := raw[i*dtype.width : (i+1)*dtype.width] switch { case dtype.isFloat && dtype.width == 8: vals = append(vals, strconv.FormatFloat(math.Float64frombits(binary.LittleEndian.Uint64(cell)), 'g', -1, 64)) case dtype.isFloat && dtype.width == 4: vals = append(vals, strconv.FormatFloat(float64(math.Float32frombits(binary.LittleEndian.Uint32(cell))), 'g', -1, 32)) case dtype.isBool: // The boolean enumeration renders as its own values; a cell // outside them is a file that contradicts its datatype, and // the attribute is dropped rather than guessed at. if cell[0] > 1 { return "", "", false } vals = append(vals, strconv.FormatInt(int64(cell[0]), 10)) case dtype.signed: // The datatype's own signed bit decides how its stored bits // read: an int8 attribute of -1 renders "-1", not 255. u := uint64At(cell, dtype.width) if width := uint(dtype.width); width < 64 { mask := uint64(1)<<(8*width) - 1 if u&(uint64(1)<<(8*width-1)) != 0 { u |= ^mask // sign-extend into the 64-bit read } } vals = append(vals, strconv.FormatInt(int64(u), 10)) default: vals = append(vals, strconv.FormatUint(uint64At(cell, dtype.width), 10)) } } if len(vals) == 1 { return name, vals[0], true } return name, "[" + strings.Join(vals, ", ") + "]", true } // heapString reads a variable-length string out of a global heap // collection: the descriptor names the collection and an object index, // and the collection lists (index, reference count, size, bytes) // records, each padded to eight bytes. func (f *hdf5File) heapString(addr uint64, index uint32, length int) (string, error) { const name = "LoadHDF5" head := f.bytes(addr, 16) if head == nil || !bytes.Equal(head[:4], []byte("GCOL")) { return "", base.Errf("%s: no global heap collection at %d", name, addr) } size := f.length(int(addr) + 8) // the collection size, header included collection := f.bytes(addr, size) if collection == nil { return "", base.Errf("%s: the global heap collection at %d lies outside the file", name, addr) } p := 8 + f.lenSize for p+8+f.lenSize <= len(collection) { objIndex := uint32(binary.LittleEndian.Uint16(collection[p:])) objSize := uint64At(collection[p+8:], f.lenSize) body := p + 8 + f.lenSize // The bound compares in uint64: an int addition would wrap a // hostile object size past the guard into a negative value and // the slice below would panic. if objSize > uint64(len(collection)-body) { break } size := int(objSize) if objIndex == index { text := collection[body : body+size] if length > 0 && length < len(text) { text = text[:length] } return strings.TrimRight(string(text), "\x00"), nil } p = alignUp(body+size, 8) } return "", base.Errf("%s: the global heap object %d is not in the collection at %d", name, index, addr) } // fletcher32 is the checksum HDF5's fletcher32 filter appends: a // 32-bit Fletcher sum over 16-bit words, ones-complement folded. func fletcher32(data []byte) uint32 { sum1, sum2 := uint32(0xffff), uint32(0xffff) // The filter runs in blocks of 359 words (718 bytes). for len(data) > 0 { n := min(len(data), 718) block := data[:n] if n%2 != 0 { block = append(append([]byte{}, block...), 0) } for i := 0; i+1 < len(block); i += 2 { sum1 += uint32(binary.BigEndian.Uint16(block[i:])) sum2 += sum1 } sum1 = (sum1 & 0xffff) + (sum1 >> 16) sum2 = (sum2 & 0xffff) + (sum2 >> 16) data = data[n:] } sum1 = (sum1 & 0xffff) + (sum1 >> 16) sum2 = (sum2 & 0xffff) + (sum2 >> 16) return sum2<<16 | sum1 } // hdf5Lookup3 is the Jenkins lookup3 hash in the little-endian, // byte-wise form HDF5 stores as the checksum of superblock versions 2 // and 3 and of every chunk of an object header version 2, with the // zero initial value the library uses. The main loop mixes whole // 12-byte blocks and the switch folds the tail, whose bytes fall into // the three words highest first. func hdf5Lookup3(key []byte) uint32 { a := uint32(0xdeadbeef) + uint32(len(key)) b := a c := a p := 0 for ; len(key)-p > 12; p += 12 { a += binary.LittleEndian.Uint32(key[p:]) b += binary.LittleEndian.Uint32(key[p+4:]) c += binary.LittleEndian.Uint32(key[p+8:]) a, b, c = hdf5Lookup3Mix(a, b, c) } switch r := key[p:]; len(r) { case 12: c += uint32(r[11]) << 24 fallthrough case 11: c += uint32(r[10]) << 16 fallthrough case 10: c += uint32(r[9]) << 8 fallthrough case 9: c += uint32(r[8]) fallthrough case 8: b += uint32(r[7]) << 24 fallthrough case 7: b += uint32(r[6]) << 16 fallthrough case 6: b += uint32(r[5]) << 8 fallthrough case 5: b += uint32(r[4]) fallthrough case 4: a += uint32(r[3]) << 24 fallthrough case 3: a += uint32(r[2]) << 16 fallthrough case 2: a += uint32(r[1]) << 8 fallthrough case 1: a += uint32(r[0]) case 0: return c } a, b, c = hdf5Lookup3Final(a, b, c) return c } // hdf5Lookup3Mix is lookup3's inner round. func hdf5Lookup3Mix(a, b, c uint32) (uint32, uint32, uint32) { a -= c a ^= bits.RotateLeft32(c, 4) c += b b -= a b ^= bits.RotateLeft32(a, 6) a += c c -= b c ^= bits.RotateLeft32(b, 8) b += a a -= c a ^= bits.RotateLeft32(c, 16) c += b b -= a b ^= bits.RotateLeft32(a, 19) a += c c -= b c ^= bits.RotateLeft32(b, 4) b += a return a, b, c } // hdf5Lookup3Final is lookup3's closing avalanche. func hdf5Lookup3Final(a, b, c uint32) (uint32, uint32, uint32) { c ^= b c -= bits.RotateLeft32(b, 14) a ^= c a -= bits.RotateLeft32(c, 11) b ^= a b -= bits.RotateLeft32(a, 25) c ^= b c -= bits.RotateLeft32(b, 16) a ^= c a -= bits.RotateLeft32(c, 4) b ^= a b -= bits.RotateLeft32(a, 14) c ^= b c -= bits.RotateLeft32(b, 24) return a, b, c }