feat: initial release
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-03 10:00:00 +02:00
commit af4ee19703
617 changed files with 191195 additions and 0 deletions
+449
View File
@@ -0,0 +1,449 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package io
// Object header and message writing for the HDF5 writer: the messages
// here are shaped exactly as the reader's decoders in hdf5.go parse
// them, which the fixtures under testdata/h5 pin byte for byte where
// it matters.
import (
"encoding/binary"
"math"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/tensor/internal/base"
)
// hdf5OutMsg is one message of an object header: a type, the payload
// the format defines for it.
type hdf5OutMsg struct {
typ uint16
data []byte
}
// hdf5HeaderV1 builds a version 1 object header image, the one the
// classic layout uses: the fixed head names the message count and the
// size of the message region, and every message is preceded by an
// eight-byte header. The message's stored size includes the padding
// that keeps the next message on an eight-byte boundary of the header,
// which is how the reference library writes every message class.
func hdf5HeaderV1(msgs []hdf5OutMsg) ([]byte, error) {
if len(msgs) > 0xffff {
return nil, base.Errf("SaveHDF5: an object header would carry %d messages, past the %d the format counts", len(msgs), 0xffff)
}
body := []byte{}
for _, m := range msgs {
data := hdf5PadField(m.data)
if len(data) > 0xffff {
return nil, base.Errf("SaveHDF5: a message of %d bytes is past the %d a version 1 header stores", len(data), 0xffff)
}
body = binary.LittleEndian.AppendUint16(body, m.typ)
body = binary.LittleEndian.AppendUint16(body, uint16(len(data)))
body = append(body, 0, 0, 0, 0) // message flags and reserved
body = append(body, data...)
}
head := []byte{1, 0}
head = binary.LittleEndian.AppendUint16(head, uint16(len(msgs)))
head = binary.LittleEndian.AppendUint32(head, 1) // reference count
head = binary.LittleEndian.AppendUint32(head, uint32(len(body)))
head = binary.LittleEndian.AppendUint32(head, 0) // padding to eight
return append(head, body...), nil
}
// hdf5HeaderV2 builds a version 2 object header image, the one the
// latest layout uses: the messages are packed without alignment and
// the whole header, signature to last message, closes with a lookup3
// checksum. The size field is one, two or four bytes, whichever holds
// it.
func hdf5HeaderV2(msgs []hdf5OutMsg) ([]byte, error) {
body := []byte{}
for _, m := range msgs {
if len(m.data) > 0xffff {
return nil, base.Errf("SaveHDF5: a message of %d bytes is past the %d a version 2 header stores", len(m.data), 0xffff)
}
body = append(body, byte(m.typ))
body = binary.LittleEndian.AppendUint16(body, uint16(len(m.data)))
body = append(body, 0) // message flags
body = append(body, m.data...)
}
// The size of the size field is itself a mask: the stored value is
// the base-2 logarithm of the width, 0 for one byte, 1 for two and
// 2 for four, which the reader decodes as 1 << stored.
if len(body) > math.MaxUint32 {
return nil, base.Errf("SaveHDF5: an object header of %d bytes is past the %d the version 2 size field holds", len(body), uint64(math.MaxUint32))
}
var logWidth byte
switch {
case len(body) > 0xffff:
logWidth = 2
case len(body) > 0xff:
logWidth = 1
}
width := byte(1) << logWidth
out := append([]byte{}, hdf5ObjHdr2...)
out = append(out, 2, logWidth)
for i := range width {
out = append(out, byte(len(body)>>(8*i)))
}
out = append(out, body...)
return binary.LittleEndian.AppendUint32(out, hdf5Lookup3(out)), nil
}
// headerAt writes a built header image over a placeholder allocated
// earlier, keeping every address that already points here valid.
func (w *hdf5Writer) headerAt(addr uint64, image []byte) {
copy(w.buf[addr:], image)
}
// hdf5AppendAlign appends the zero bytes that align b to its next
// eight-byte boundary.
func hdf5AppendAlign(b []byte) []byte {
if r := len(b) % 8; r != 0 {
return append(b, make([]byte, 8-r)...)
}
return b
}
// hdf5PadField pads a name or a value field of a message to its next
// eight-byte boundary; a field already aligned stays as it is.
func hdf5PadField(b []byte) []byte { return hdf5AppendAlign(b) }
// Datatype messages, version 1. The class bit field's first byte
// carries the byte order (cleared: little-endian) and, for fixed
// point, the signed bit; the floating-point classes carry the IEEE 754
// layout the fixtures store, down to the exponent bias.
var (
hdf5Float32Type = []byte{
0x11, // version 1, class 1 (floating-point)
0x20, 0x1f, 0x00, // little-endian, the sign at bit 31
0x04, 0x00, 0x00, 0x00, // four-byte elements
0x00, 0x00, // bit offset 0
0x20, 0x00, // 32 bits of precision
0x17, 0x08, 0x00, 0x17, // exponent at 23 of 8, mantissa at 0 of 23
0x7f, 0x00, 0x00, 0x00, // exponent bias 127
}
hdf5Float64Type = []byte{
0x11, // version 1, class 1 (floating-point)
0x20, 0x3f, 0x00, // little-endian, the sign at bit 63
0x08, 0x00, 0x00, 0x00, // eight-byte elements
0x00, 0x00, // bit offset 0
0x40, 0x00, // 64 bits of precision
0x34, 0x0b, 0x00, 0x34, // exponent at 52 of 11, mantissa at 0 of 52
0xff, 0x03, 0x00, 0x00, // exponent bias 1023
}
)
// hdf5IntType writes a fixed-point datatype message of the given
// element width and signedness. The class bit field's byte order bit
// stays clear (little-endian) and bit 0x08 carries two's-complement
// signedness, the bit decodeType keys its landing on. The message is
// twelve bytes: the eight-byte header plus the bit offset and bit
// precision the format's fixed-point property table defines.
func hdf5IntType(width int, signed bool) []byte {
flags := byte(0) // little-endian, unsigned
if signed {
flags = 0x08 // bit 3: two's complement
}
b := []byte{0x10, flags, 0, 0} // version 1, class 0 (fixed-point)
b = binary.LittleEndian.AppendUint32(b, uint32(width))
b = binary.LittleEndian.AppendUint16(b, 0)
b = binary.LittleEndian.AppendUint16(b, uint16(8*width))
return b
}
// hdf5BoolType writes the boolean enumeration datatype message, in the
// exact shape the reader's hdf5EnumBool admits: class 8 with a member
// count of two in the class bit field's low sixteen bits and the
// reserved byte zero, a base type that is a complete one-byte unsigned
// little-endian fixed-point message, the member names each NUL
// terminated and padded from its own field start to a multiple of eight
// bytes (the message version 1 convention) and the packed member values
// 0 and 1 behind the names. The member names carry no semantics; the
// values are what the landing reads, and the payload stores them, one
// byte per element.
func hdf5BoolType() []byte {
b := []byte{0x18, 0x02, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00} // version 1, class 8, two members, one-byte values
b = append(b, hdf5IntType(1, false)...) // the base type
for _, name := range []string{"FALSE", "TRUE"} {
start := len(b)
b = append(b, name...)
b = append(b, 0)
for (len(b)-start)%8 != 0 {
b = append(b, 0)
}
}
return append(b, 0, 1) // FALSE = 0, TRUE = 1
}
// hdf5StringType writes a fixed-length string datatype message padded
// with NUL bytes, the padding the reference writes for fixed strings.
func hdf5StringType(width int) []byte {
b := []byte{0x13, 0x01, 0, 0} // version 1, class 3, NUL-padded, ASCII
return binary.LittleEndian.AppendUint32(b, uint32(width))
}
// hdf5SpaceV1 writes a version 1 dataspace message: the shape, with
// the maximum dimensions (the same extents) behind it, as the
// reference writes. A scalar dataset declares rank 0.
func hdf5SpaceV1(shape []int) []byte {
if len(shape) == 0 {
return []byte{1, 0, 0, 0, 0, 0, 0, 0}
}
b := []byte{1, byte(len(shape)), 0x01, 0}
b = binary.LittleEndian.AppendUint32(b, 0)
for _, d := range shape {
b = binary.LittleEndian.AppendUint64(b, uint64(d))
}
for _, d := range shape {
b = binary.LittleEndian.AppendUint64(b, uint64(d))
}
return b
}
// hdf5SpaceV2 writes a version 2 dataspace message, the one the latest
// layout uses: the fourth byte names the dataspace class, which the
// reference library sets to simple (one) for every ranked extent.
func hdf5SpaceV2(shape []int) []byte {
if len(shape) == 0 {
return []byte{2, 0, 0, 0} // a scalar: class 0, no dimensions
}
b := []byte{2, byte(len(shape)), 0x01, 1}
for _, d := range shape {
b = binary.LittleEndian.AppendUint64(b, uint64(d))
}
for _, d := range shape {
b = binary.LittleEndian.AppendUint64(b, uint64(d))
}
return b
}
// hdf5FillValueMsg writes the fill value message: the version 2 form
// of the classic layout declares a defined, all-zero fill (allocated
// incrementally for contiguous storage and late for chunked, as the
// reference does), the version 3 form of the latest layout carries no
// fill value at all.
func hdf5FillValueMsg(latest bool, chunked bool) []byte {
if latest {
return []byte{3, 0x0a}
}
if chunked {
return []byte{2, 3, 2, 1, 0, 0, 0, 0}
}
return []byte{2, 2, 2, 1, 0, 0, 0, 0}
}
// hdf5LayoutContiguous writes a version 3 contiguous layout message:
// the data's address and its byte extent. An empty dataset names no
// storage: the undefined address stands for it.
func hdf5LayoutContiguous(addr, size uint64) []byte {
b := []byte{3, 1}
b = binary.LittleEndian.AppendUint64(b, addr)
return binary.LittleEndian.AppendUint64(b, size)
}
// hdf5LayoutChunked writes a version 3 chunked layout message: the
// B-tree address, then one more dimension than the dataset has, the
// chunk's shape followed by the element size.
func hdf5LayoutChunked(addr uint64, chunk []int, width int) []byte {
b := []byte{3, 2, byte(len(chunk) + 1)}
b = binary.LittleEndian.AppendUint64(b, addr)
for _, c := range chunk {
b = binary.LittleEndian.AppendUint32(b, uint32(c))
}
return binary.LittleEndian.AppendUint32(b, uint32(width))
}
// hdf5OutFilter is one entry of a filter pipeline message: the
// identifier, the name the reference stores and the client values.
type hdf5OutFilter struct {
id uint16
name string
values []uint32
}
// hdf5FilterMessage writes a version 1 filter pipeline message. Every
// entry carries the optional flag the reference sets, its name padded
// to eight bytes and its client values padded to the same boundary.
func hdf5FilterMessage(filters []hdf5OutFilter) []byte {
b := []byte{1, byte(len(filters)), 0, 0, 0, 0, 0, 0}
for _, f := range filters {
name := append([]byte(f.name), 0)
b = binary.LittleEndian.AppendUint16(b, f.id)
b = binary.LittleEndian.AppendUint16(b, uint16(len(name)))
b = binary.LittleEndian.AppendUint16(b, 1)
b = binary.LittleEndian.AppendUint16(b, uint16(len(f.values)))
b = hdf5PadField(append(b, name...))
for _, v := range f.values {
b = binary.LittleEndian.AppendUint32(b, v)
}
b = hdf5AppendAlign(b)
}
return b
}
// hdf5AttrMessage writes one version 1 attribute message: the name
// padded to eight bytes, the datatype padded to eight, the dataspace,
// then the value parsed out of its text. The reader renders every
// attribute it reads as text, so the writer parses text back into the
// typed attribute it names: a whole number becomes int64, a decimal
// float64, a bracketed list an int64 or float64 array and anything
// else a fixed-length string.
func hdf5AttrMessage(a hdf5Attr) (hdf5OutMsg, error) {
dtype, value, shape, err := hdf5AttrValue(a.name, a.text)
if err != nil {
return hdf5OutMsg{}, err
}
// The name size counts the name and its NUL terminator; the field
// itself is padded to eight bytes.
nameField := hdf5PadField(append([]byte(a.name), 0))
space := hdf5SpaceV1(shape)
body := []byte{1, 0}
body = binary.LittleEndian.AppendUint16(body, uint16(len(a.name)+1))
body = binary.LittleEndian.AppendUint16(body, uint16(len(dtype)))
body = binary.LittleEndian.AppendUint16(body, uint16(len(space)))
body = append(body, nameField...)
body = hdf5PadField(append(body, dtype...))
body = append(body, space...)
body = append(body, value...)
return hdf5OutMsg{typ: hdf5MsgAttribute, data: body}, nil
}
// hdf5AttrValue parses an attribute's text into a datatype message,
// the raw value bytes and the dataspace shape (nil for a scalar). It
// is the inverse of the reader's attribute rendering: FormatInt and
// FormatFloat output parse back to the values they were printed from,
// bit for bit, and every other text becomes a fixed-length string.
func hdf5AttrValue(name, text string) ([]byte, []byte, []int, error) {
if strings.HasPrefix(text, "[") && strings.HasSuffix(text, "]") {
inner := strings.TrimSpace(text[1 : len(text)-1])
if inner == "" {
// An empty array: an int64 attribute of extent zero, which
// the reader renders back as "[]".
return hdf5IntType(8, true), nil, []int{0}, nil
}
parts := strings.Split(inner, ", ")
ints := make([]int64, 0, len(parts))
floats := make([]float64, 0, len(parts))
asFloat := false
for i, p := range parts {
if v, err := strconv.ParseInt(p, 10, 64); err == nil && !asFloat {
ints = append(ints, v)
floats = append(floats, float64(v))
continue
}
v, err := strconv.ParseFloat(p, 64)
if err != nil {
return nil, nil, nil, base.Errf("the attribute %q holds the array value %q whose element %d is not a number", name, text, i)
}
asFloat = true
floats = append(floats, v)
}
if asFloat {
raw := make([]byte, 0, 8*len(floats))
for _, v := range floats {
raw = binary.LittleEndian.AppendUint64(raw, math.Float64bits(v))
}
return hdf5Float64Type, raw, []int{len(floats)}, nil
}
raw := make([]byte, 0, 8*len(ints))
for _, v := range ints {
raw = binary.LittleEndian.AppendUint64(raw, uint64(v))
}
return hdf5IntType(8, true), raw, []int{len(ints)}, nil
}
if v, err := strconv.ParseInt(text, 10, 64); err == nil {
return hdf5IntType(8, true), binary.LittleEndian.AppendUint64(nil, uint64(v)), nil, nil
}
if v, err := strconv.ParseFloat(text, 64); err == nil {
return hdf5Float64Type, binary.LittleEndian.AppendUint64(nil, math.Float64bits(v)), nil, nil
}
width := max(len(text), 1)
value := append([]byte(text), make([]byte, width-len(text))...)
return hdf5StringType(width), value, nil, nil
}
// writeDataset writes one dataset: the dataspace, datatype and fill
// value messages, the filter pipeline and layout of its storage and
// its attributes, then the header of the layout the file uses.
func (w *hdf5Writer) writeDataset(s *hdf5OutSet) (uint64, error) {
msgs := make([]hdf5OutMsg, 0, 6)
if w.latest {
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataspace, data: hdf5SpaceV2(s.shape)})
} else {
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataspace, data: hdf5SpaceV1(s.shape)})
}
switch s.class {
case 0:
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5IntType(s.width, s.signed)})
case 1:
if s.width == 4 {
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5Float32Type})
} else {
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5Float64Type})
}
case 3:
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5StringType(s.width)})
case 8:
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDatatype, data: hdf5BoolType()})
default:
// Every class the plan can produce has its message case above;
// an unlisted one would emit a file with no datatype message,
// which the reader refuses later with less context than this.
return 0, base.Errf("SaveHDF5: dataset %q: the writer emits no datatype message for class %d", s.path, s.class)
}
// Chunked storage is the filter pipeline's only carrier, so a
// dataset with filters goes chunked; strings keep their raw bytes,
// and a dataset with no elements has nothing for the filters to
// compress, so it stays contiguous at an undefined address either
// way.
chunked := len(s.shape) > 0 && s.nbytes > 0 && (w.opts.Gzip != 0 || w.opts.Shuffle) && s.class != 3
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgFillValue, data: hdf5FillValueMsg(w.latest, chunked)})
if chunked {
entries := w.filterEntries(s.width)
if len(entries) > 0 {
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgFilterPipeline, data: hdf5FilterMessage(entries)})
}
tree, chunk, err := w.writeChunks(s)
if err != nil {
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, err)
}
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataLayout, data: hdf5LayoutChunked(tree, chunk, s.width)})
} else {
addr := uint64(math.MaxUint64)
var size uint64
if s.nbytes > 0 {
// The payload is encoded where the format will read it:
// the reservation is its final address, so the values pass
// through the writer once instead of building a block and
// then copying it in.
addr = w.reserve(s.nbytes)
if encErr := s.encode(w.buf[addr:]); encErr != nil {
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, encErr)
}
w.pad8()
size = uint64(s.nbytes)
}
msgs = append(msgs, hdf5OutMsg{typ: hdf5MsgDataLayout, data: hdf5LayoutContiguous(addr, size)})
}
for _, a := range s.attrs {
m, err := hdf5AttrMessage(a)
if err != nil {
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, err)
}
msgs = append(msgs, m)
}
var image []byte
var err error
if w.latest {
image, err = hdf5HeaderV2(msgs)
} else {
image, err = hdf5HeaderV1(msgs)
}
if err != nil {
return 0, base.Errf("SaveHDF5: dataset %q: %w", s.path, err)
}
return w.bytes(image), nil
}