450 lines
17 KiB
Go
450 lines
17 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
package io
|
|
|
|
// Object header and message writing for the HDF5 writer: the messages
|
|
// here are shaped exactly as the reader's decoders in hdf5.go parse
|
|
// them, which the fixtures under testdata/h5 pin byte for byte where
|
|
// it matters.
|
|
|
|
import (
|
|
"encoding/binary"
|
|
"math"
|
|
"strconv"
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
|
)
|
|
|
|
// hdf5OutMsg is one message of an object header: a type, the payload
|
|
// the format defines for it.
|
|
type hdf5OutMsg struct {
|
|
typ uint16
|
|
data []byte
|
|
}
|
|
|
|
// hdf5HeaderV1 builds a version 1 object header image, the one the
|
|
// classic layout uses: the fixed head names the message count and the
|
|
// size of the message region, and every message is preceded by an
|
|
// eight-byte header. The message's stored size includes the padding
|
|
// that keeps the next message on an eight-byte boundary of the header,
|
|
// which is how the reference library writes every message class.
|
|
func hdf5HeaderV1(msgs []hdf5OutMsg) ([]byte, error) {
|
|
if len(msgs) > 0xffff {
|
|
return nil, base.Errf("SaveHDF5: an object header would carry %d messages, past the %d the format counts", len(msgs), 0xffff)
|
|
}
|
|
body := []byte{}
|
|
for _, m := range msgs {
|
|
data := hdf5PadField(m.data)
|
|
if len(data) > 0xffff {
|
|
return nil, base.Errf("SaveHDF5: a message of %d bytes is past the %d a version 1 header stores", len(data), 0xffff)
|
|
}
|
|
body = binary.LittleEndian.AppendUint16(body, m.typ)
|
|
body = binary.LittleEndian.AppendUint16(body, uint16(len(data)))
|
|
body = append(body, 0, 0, 0, 0) // message flags and reserved
|
|
body = append(body, data...)
|
|
}
|
|
head := []byte{1, 0}
|
|
head = binary.LittleEndian.AppendUint16(head, uint16(len(msgs)))
|
|
head = binary.LittleEndian.AppendUint32(head, 1) // reference count
|
|
head = binary.LittleEndian.AppendUint32(head, uint32(len(body)))
|
|
head = binary.LittleEndian.AppendUint32(head, 0) // padding to eight
|
|
return append(head, body...), nil
|
|
}
|
|
|
|
// hdf5HeaderV2 builds a version 2 object header image, the one the
|
|
// latest layout uses: the messages are packed without alignment and
|
|
// the whole header, signature to last message, closes with a lookup3
|
|
// checksum. The size field is one, two or four bytes, whichever holds
|
|
// it.
|
|
func hdf5HeaderV2(msgs []hdf5OutMsg) ([]byte, error) {
|
|
body := []byte{}
|
|
for _, m := range msgs {
|
|
if len(m.data) > 0xffff {
|
|
return nil, base.Errf("SaveHDF5: a message of %d bytes is past the %d a version 2 header stores", len(m.data), 0xffff)
|
|
}
|
|
body = append(body, byte(m.typ))
|
|
body = binary.LittleEndian.AppendUint16(body, uint16(len(m.data)))
|
|
body = append(body, 0) // message flags
|
|
body = append(body, m.data...)
|
|
}
|
|
// The size of the size field is itself a mask: the stored value is
|
|
// the base-2 logarithm of the width, 0 for one byte, 1 for two and
|
|
// 2 for four, which the reader decodes as 1 << stored.
|
|
if len(body) > math.MaxUint32 {
|
|
return nil, base.Errf("SaveHDF5: an object header of %d bytes is past the %d the version 2 size field holds", len(body), uint64(math.MaxUint32))
|
|
}
|
|
var logWidth byte
|
|
switch {
|
|
case len(body) > 0xffff:
|
|
logWidth = 2
|
|
case len(body) > 0xff:
|
|
logWidth = 1
|
|
}
|
|
width := byte(1) << logWidth
|
|
out := append([]byte{}, hdf5ObjHdr2...)
|
|
out = append(out, 2, logWidth)
|
|
for i := range width {
|
|
out = append(out, byte(len(body)>>(8*i)))
|
|
}
|
|
out = append(out, body...)
|
|
return binary.LittleEndian.AppendUint32(out, hdf5Lookup3(out)), nil
|
|
}
|
|
|
|
// headerAt writes a built header image over a placeholder allocated
|
|
// earlier, keeping every address that already points here valid.
|
|
func (w *hdf5Writer) headerAt(addr uint64, image []byte) {
|
|
copy(w.buf[addr:], image)
|
|
}
|
|
|
|
// hdf5AppendAlign appends the zero bytes that align b to its next
|
|
// eight-byte boundary.
|
|
func hdf5AppendAlign(b []byte) []byte {
|
|
if r := len(b) % 8; r != 0 {
|
|
return append(b, make([]byte, 8-r)...)
|
|
}
|
|
return b
|
|
}
|
|
|
|
// hdf5PadField pads a name or a value field of a message to its next
|
|
// eight-byte boundary; a field already aligned stays as it is.
|
|
func hdf5PadField(b []byte) []byte { return hdf5AppendAlign(b) }
|
|
|
|
// Datatype messages, version 1. The class bit field's first byte
|
|
// carries the byte order (cleared: little-endian) and, for fixed
|
|
// point, the signed bit; the floating-point classes carry the IEEE 754
|
|
// layout the fixtures store, down to the exponent bias.
|
|
var (
|
|
hdf5Float32Type = []byte{
|
|
0x11, // version 1, class 1 (floating-point)
|
|
0x20, 0x1f, 0x00, // little-endian, the sign at bit 31
|
|
0x04, 0x00, 0x00, 0x00, // four-byte elements
|
|
0x00, 0x00, // bit offset 0
|
|
0x20, 0x00, // 32 bits of precision
|
|
0x17, 0x08, 0x00, 0x17, // exponent at 23 of 8, mantissa at 0 of 23
|
|
0x7f, 0x00, 0x00, 0x00, // exponent bias 127
|
|
}
|
|
hdf5Float64Type = []byte{
|
|
0x11, // version 1, class 1 (floating-point)
|
|
0x20, 0x3f, 0x00, // little-endian, the sign at bit 63
|
|
0x08, 0x00, 0x00, 0x00, // eight-byte elements
|
|
0x00, 0x00, // bit offset 0
|
|
0x40, 0x00, // 64 bits of precision
|
|
0x34, 0x0b, 0x00, 0x34, // exponent at 52 of 11, mantissa at 0 of 52
|
|
0xff, 0x03, 0x00, 0x00, // exponent bias 1023
|
|
}
|
|
)
|
|
|
|
// hdf5IntType writes a fixed-point datatype message of the given
|
|
// element width and signedness. The class bit field's byte order bit
|
|
// stays clear (little-endian) and bit 0x08 carries two's-complement
|
|
// signedness, the bit decodeType keys its landing on. The message is
|
|
// twelve bytes: the eight-byte header plus the bit offset and bit
|
|
// precision the format's fixed-point property table defines.
|
|
func hdf5IntType(width int, signed bool) []byte {
|
|
flags := byte(0) // little-endian, unsigned
|
|
if signed {
|
|
flags = 0x08 // bit 3: two's complement
|
|
}
|
|
b := []byte{0x10, flags, 0, 0} // version 1, class 0 (fixed-point)
|
|
b = binary.LittleEndian.AppendUint32(b, uint32(width))
|
|
b = binary.LittleEndian.AppendUint16(b, 0)
|
|
b = binary.LittleEndian.AppendUint16(b, uint16(8*width))
|
|
return b
|
|
}
|
|
|
|
// hdf5BoolType writes the boolean enumeration datatype message, in the
|
|
// exact shape the reader's hdf5EnumBool admits: class 8 with a member
|
|
// count of two in the class bit field's low sixteen bits and the
|
|
// reserved byte zero, a base type that is a complete one-byte unsigned
|
|
// little-endian fixed-point message, the member names each NUL
|
|
// terminated and padded from its own field start to a multiple of eight
|
|
// bytes (the message version 1 convention) and the packed member values
|
|
// 0 and 1 behind the names. The member names carry no semantics; the
|
|
// values are what the landing reads, and the payload stores them, one
|
|
// byte per element.
|
|
func hdf5BoolType() []byte {
|
|
b := []byte{0x18, 0x02, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00} // version 1, class 8, two members, one-byte values
|
|
b = append(b, hdf5IntType(1, false)...) // the base type
|
|
for _, name := range []string{"FALSE", "TRUE"} {
|
|
start := len(b)
|
|
b = append(b, name...)
|
|
b = append(b, 0)
|
|
for (len(b)-start)%8 != 0 {
|
|
b = append(b, 0)
|
|
}
|
|
}
|
|
return append(b, 0, 1) // FALSE = 0, TRUE = 1
|
|
}
|
|
|
|
// hdf5StringType writes a fixed-length string datatype message padded
|
|
// with NUL bytes, the padding the reference writes for fixed strings.
|
|
func hdf5StringType(width int) []byte {
|
|
b := []byte{0x13, 0x01, 0, 0} // version 1, class 3, NUL-padded, ASCII
|
|
return binary.LittleEndian.AppendUint32(b, uint32(width))
|
|
}
|
|
|
|
// hdf5SpaceV1 writes a version 1 dataspace message: the shape, with
|
|
// the maximum dimensions (the same extents) behind it, as the
|
|
// reference writes. A scalar dataset declares rank 0.
|
|
func hdf5SpaceV1(shape []int) []byte {
|
|
if len(shape) == 0 {
|
|
return []byte{1, 0, 0, 0, 0, 0, 0, 0}
|
|
}
|
|
b := []byte{1, byte(len(shape)), 0x01, 0}
|
|
b = binary.LittleEndian.AppendUint32(b, 0)
|
|
for _, d := range shape {
|
|
b = binary.LittleEndian.AppendUint64(b, uint64(d))
|
|
}
|
|
for _, d := range shape {
|
|
b = binary.LittleEndian.AppendUint64(b, uint64(d))
|
|
}
|
|
return b
|
|
}
|
|
|
|
// hdf5SpaceV2 writes a version 2 dataspace message, the one the latest
|
|
// layout uses: the fourth byte names the dataspace class, which the
|
|
// reference library sets to simple (one) for every ranked extent.
|
|
func hdf5SpaceV2(shape []int) []byte {
|
|
if len(shape) == 0 {
|
|
return []byte{2, 0, 0, 0} // a scalar: class 0, no dimensions
|
|
}
|
|
b := []byte{2, byte(len(shape)), 0x01, 1}
|
|
for _, d := range shape {
|
|
b = binary.LittleEndian.AppendUint64(b, uint64(d))
|
|
}
|
|
for _, d := range shape {
|
|
b = binary.LittleEndian.AppendUint64(b, uint64(d))
|
|
}
|
|
return b
|
|
}
|
|
|
|
// hdf5FillValueMsg writes the fill value message: the version 2 form
|
|
// of the classic layout declares a defined, all-zero fill (allocated
|
|
// incrementally for contiguous storage and late for chunked, as the
|
|
// reference does), the version 3 form of the latest layout carries no
|
|
// fill value at all.
|
|
func hdf5FillValueMsg(latest bool, chunked bool) []byte {
|
|
if latest {
|
|
return []byte{3, 0x0a}
|
|
}
|
|
if chunked {
|
|
return []byte{2, 3, 2, 1, 0, 0, 0, 0}
|
|
}
|
|
return []byte{2, 2, 2, 1, 0, 0, 0, 0}
|
|
}
|
|
|
|
// hdf5LayoutContiguous writes a version 3 contiguous layout message:
|
|
// the data's address and its byte extent. An empty dataset names no
|
|
// storage: the undefined address stands for it.
|
|
func hdf5LayoutContiguous(addr, size uint64) []byte {
|
|
b := []byte{3, 1}
|
|
b = binary.LittleEndian.AppendUint64(b, addr)
|
|
return binary.LittleEndian.AppendUint64(b, size)
|
|
}
|
|
|
|
// hdf5LayoutChunked writes a version 3 chunked layout message: the
|
|
// B-tree address, then one more dimension than the dataset has, the
|
|
// chunk's shape followed by the element size.
|
|
func hdf5LayoutChunked(addr uint64, chunk []int, width int) []byte {
|
|
b := []byte{3, 2, byte(len(chunk) + 1)}
|
|
b = binary.LittleEndian.AppendUint64(b, addr)
|
|
for _, c := range chunk {
|
|
b = binary.LittleEndian.AppendUint32(b, uint32(c))
|
|
}
|
|
return binary.LittleEndian.AppendUint32(b, uint32(width))
|
|
}
|
|
|
|
// hdf5OutFilter is one entry of a filter pipeline message: the
|
|
// identifier, the name the reference stores and the client values.
|
|
type hdf5OutFilter struct {
|
|
id uint16
|
|
name string
|
|
values []uint32
|
|
}
|
|
|
|
// hdf5FilterMessage writes a version 1 filter pipeline message. Every
|
|
// entry carries the optional flag the reference sets, its name padded
|
|
// to eight bytes and its client values padded to the same boundary.
|
|
func hdf5FilterMessage(filters []hdf5OutFilter) []byte {
|
|
b := []byte{1, byte(len(filters)), 0, 0, 0, 0, 0, 0}
|
|
for _, f := range filters {
|
|
name := append([]byte(f.name), 0)
|
|
b = binary.LittleEndian.AppendUint16(b, f.id)
|
|
b = binary.LittleEndian.AppendUint16(b, uint16(len(name)))
|
|
b = binary.LittleEndian.AppendUint16(b, 1)
|
|
b = binary.LittleEndian.AppendUint16(b, uint16(len(f.values)))
|
|
b = hdf5PadField(append(b, name...))
|
|
for _, v := range f.values {
|
|
b = binary.LittleEndian.AppendUint32(b, v)
|
|
}
|
|
b = hdf5AppendAlign(b)
|
|
}
|
|
return b
|
|
}
|
|
|
|
// hdf5AttrMessage writes one version 1 attribute message: the name
|
|
// padded to eight bytes, the datatype padded to eight, the dataspace,
|
|
// then the value parsed out of its text. The reader renders every
|
|
// attribute it reads as text, so the writer parses text back into the
|
|
// typed attribute it names: a whole number becomes int64, a decimal
|
|
// float64, a bracketed list an int64 or float64 array and anything
|
|
// else a fixed-length string.
|
|
func hdf5AttrMessage(a hdf5Attr) (hdf5OutMsg, error) {
|
|
dtype, value, shape, err := hdf5AttrValue(a.name, a.text)
|
|
if err != nil {
|
|
return hdf5OutMsg{}, err
|
|
}
|
|
// The name size counts the name and its NUL terminator; the field
|
|
// itself is padded to eight bytes.
|
|
nameField := hdf5PadField(append([]byte(a.name), 0))
|
|
space := hdf5SpaceV1(shape)
|
|
body := []byte{1, 0}
|
|
body = binary.LittleEndian.AppendUint16(body, uint16(len(a.name)+1))
|
|
body = binary.LittleEndian.AppendUint16(body, uint16(len(dtype)))
|
|
body = binary.LittleEndian.AppendUint16(body, uint16(len(space)))
|
|
body = append(body, nameField...)
|
|
body = hdf5PadField(append(body, dtype...))
|
|
body = append(body, space...)
|
|
body = append(body, value...)
|
|
return hdf5OutMsg{typ: hdf5MsgAttribute, data: body}, nil
|
|
}
|
|
|
|
// hdf5AttrValue parses an attribute's text into a datatype message,
|
|
// the raw value bytes and the dataspace shape (nil for a scalar). It
|
|
// is the inverse of the reader's attribute rendering: FormatInt and
|
|
// FormatFloat output parse back to the values they were printed from,
|
|
// bit for bit, and every other text becomes a fixed-length string.
|
|
func hdf5AttrValue(name, text string) ([]byte, []byte, []int, error) {
|
|
if strings.HasPrefix(text, "[") && strings.HasSuffix(text, "]") {
|
|
inner := strings.TrimSpace(text[1 : len(text)-1])
|
|
if inner == "" {
|
|
// An empty array: an int64 attribute of extent zero, which
|
|
// the reader renders back as "[]".
|
|
return hdf5IntType(8, true), nil, []int{0}, nil
|
|
}
|
|
parts := strings.Split(inner, ", ")
|
|
ints := make([]int64, 0, len(parts))
|
|
floats := make([]float64, 0, len(parts))
|
|
asFloat := false
|
|
for i, p := range parts {
|
|
if v, err := strconv.ParseInt(p, 10, 64); err == nil && !asFloat {
|
|
ints = append(ints, v)
|
|
floats = append(floats, float64(v))
|
|
continue
|
|
}
|
|
v, err := strconv.ParseFloat(p, 64)
|
|
if err != nil {
|
|
return nil, nil, nil, base.Errf("the attribute %q holds the array value %q whose element %d is not a number", name, text, i)
|
|
}
|
|
asFloat = true
|
|
floats = append(floats, v)
|
|
}
|
|
if asFloat {
|
|
raw := make([]byte, 0, 8*len(floats))
|
|
for _, v := range floats {
|
|
raw = binary.LittleEndian.AppendUint64(raw, math.Float64bits(v))
|
|
}
|
|
return hdf5Float64Type, raw, []int{len(floats)}, nil
|
|
}
|
|
raw := make([]byte, 0, 8*len(ints))
|
|
for _, v := range ints {
|
|
raw = binary.LittleEndian.AppendUint64(raw, uint64(v))
|
|
}
|
|
return hdf5IntType(8, true), raw, []int{len(ints)}, nil
|
|
}
|
|
if v, err := strconv.ParseInt(text, 10, 64); err == nil {
|
|
return hdf5IntType(8, true), binary.LittleEndian.AppendUint64(nil, uint64(v)), nil, nil
|
|
}
|
|
if v, err := strconv.ParseFloat(text, 64); err == nil {
|
|
return hdf5Float64Type, binary.LittleEndian.AppendUint64(nil, math.Float64bits(v)), nil, nil
|
|
}
|
|
width := max(len(text), 1)
|
|
value := append([]byte(text), make([]byte, width-len(text))...)
|
|
return hdf5StringType(width), value, nil, nil
|
|
}
|
|
|
|
// writeDataset writes one dataset: the dataspace, datatype and fill
|
|
// value messages, the filter pipeline and layout of its storage and
|
|
// its attributes, then the header of the layout the file uses.
|
|
func (w *hdf5Writer) writeDataset(s *hdf5OutSet) (uint64, error) {
|
|
msgs := make([]hdf5OutMsg, 0, 6)
|
|
if w.latest {
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataspace, data: hdf5SpaceV2(s.shape)})
|
|
} else {
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataspace, data: hdf5SpaceV1(s.shape)})
|
|
}
|
|
switch s.class {
|
|
case 0:
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5IntType(s.width, s.signed)})
|
|
case 1:
|
|
if s.width == 4 {
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5Float32Type})
|
|
} else {
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5Float64Type})
|
|
}
|
|
case 3:
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5StringType(s.width)})
|
|
case 8:
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5BoolType()})
|
|
default:
|
|
// Every class the plan can produce has its message case above;
|
|
// an unlisted one would emit a file with no datatype message,
|
|
// which the reader refuses later with less context than this.
|
|
return 0, base.Errf("SaveHDF5: dataset %q: the writer emits no datatype message for class %d", s.path, s.class)
|
|
}
|
|
// Chunked storage is the filter pipeline's only carrier, so a
|
|
// dataset with filters goes chunked; strings keep their raw bytes,
|
|
// and a dataset with no elements has nothing for the filters to
|
|
// compress, so it stays contiguous at an undefined address either
|
|
// way.
|
|
chunked := len(s.shape) > 0 && s.nbytes > 0 && (w.opts.Gzip != 0 || w.opts.Shuffle) && s.class != 3
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgFillValue, data: hdf5FillValueMsg(w.latest, chunked)})
|
|
if chunked {
|
|
entries := w.filterEntries(s.width)
|
|
if len(entries) > 0 {
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgFilterPipeline, data: hdf5FilterMessage(entries)})
|
|
}
|
|
tree, chunk, err := w.writeChunks(s)
|
|
if err != nil {
|
|
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, err)
|
|
}
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataLayout, data: hdf5LayoutChunked(tree, chunk, s.width)})
|
|
} else {
|
|
addr := uint64(math.MaxUint64)
|
|
var size uint64
|
|
if s.nbytes > 0 {
|
|
// The payload is encoded where the format will read it:
|
|
// the reservation is its final address, so the values pass
|
|
// through the writer once instead of building a block and
|
|
// then copying it in.
|
|
addr = w.reserve(s.nbytes)
|
|
if encErr := s.encode(w.buf[addr:]); encErr != nil {
|
|
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, encErr)
|
|
}
|
|
w.pad8()
|
|
size = uint64(s.nbytes)
|
|
}
|
|
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataLayout, data: hdf5LayoutContiguous(addr, size)})
|
|
}
|
|
for _, a := range s.attrs {
|
|
m, err := hdf5AttrMessage(a)
|
|
if err != nil {
|
|
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, err)
|
|
}
|
|
msgs = append(msgs, m)
|
|
}
|
|
var image []byte
|
|
var err error
|
|
if w.latest {
|
|
image, err = hdf5HeaderV2(msgs)
|
|
} else {
|
|
image, err = hdf5HeaderV1(msgs)
|
|
}
|
|
if err != nil {
|
|
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, err)
|
|
}
|
|
return w.bytes(image), nil
|
|
}
|