Files
tensor/io/fitstable.go
T

1083 lines
34 KiB
Go
Raw Normal View History

2026-09-03 10:00:00 +02:00
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package io
import (
"bytes"
"encoding/binary"
"fmt"
"math"
"os"
"strconv"
"strings"
"unsafe"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
// FITS table extensions. Catalogues, source lists and observation
// logs live in the format's table HDUs: a binary table (XTENSION
// 'BINTABLE') packs each column big-endian according to its TFORM
// descriptor, an ASCII table (XTENSION 'TABLE') lays fixed-width text
// columns into rows of NAXIS1 characters. Both share the header
// vocabulary: TTYPEn names a column, TUNITn gives its unit, TFIELDS
// counts them.
// FITSTableColumn describes one column of a FITS table. Form carries
// the TFORM-style descriptor: "D" (float64), "E" (float32), "K"
// (int64), "L" (logical), "B" (unsigned byte), "I" (16-bit) or "J"
// (32-bit) for a binary table, and "nA" (a character string of n
// characters) for either kind; "Inw", "Fw.d", "Ew.d" or "Dw.d" lay out
// an ASCII table column, where w is the column width in characters. A
// numeric form reads its values from Data (Float answers D, Float32
// answers E, Int answers K, L, B, I and J); a character form reads
// them from Text.
type FITSTableColumn struct {
Name string
Unit string
Form string
Data *core.Array
Text []string
}
// FITSTable is a parsed table extension. Names, Units, Columns and
// Text run parallel to the file's column list: a numeric column holds
// its values in Columns and nil in Text, a character column holds its
// strings in Text and nil in Columns.
type FITSTable struct {
Kind string // "BINTABLE" or "TABLE"
Names []string
Units []string
Columns []*core.Array
Text [][]string
Rows int
Headers map[string]string
}
// SaveFITSTable writes a primary HDU followed by one table extension:
// a binary table by default, an ASCII table when ascii is set. The
// headers land in the extension's header beside the column cards. A
// column with a numeric form but nil Data, a character form without
// Text, mismatched row counts, an unknown form, an empty column list,
// or a value that does not fit its ASCII form's width is an error.
func SaveFITSTable(path string, ascii bool, cols []FITSTableColumn, headers map[string]string) error {
const name = "SaveFITSTable"
if len(cols) == 0 {
return base.Errf("%s: at least one column is required", name)
}
rows := -1
forms := make([]string, len(cols))
for i := range cols {
c := &cols[i]
if err := fitsCheckString(c.Name, "column name"); err != nil {
return base.Errf("%s: %w", name, err)
}
form := strings.ToUpper(strings.TrimSpace(c.Form))
forms[i] = form
if form == "" {
return base.Errf("%s: column %d (%s) has an empty form", name, i, c.Name)
}
if strings.HasSuffix(form, "A") {
if c.Text == nil {
return base.Errf("%s: character column %s needs Text", name, c.Name)
}
if c.Data != nil {
return base.Errf("%s: character column %s must not carry Data", name, c.Name)
}
if rows >= 0 && len(c.Text) != rows {
return base.Errf("%s: column %s has %d rows, want %d", name, c.Name, len(c.Text), rows)
}
rows = len(c.Text)
continue
}
if c.Data == nil {
return base.Errf("%s: numeric column %s needs Data", name, c.Name)
}
if c.Data.NDim() != 1 {
return base.Errf("%s: column %s must be a vector, got shape %s", name, c.Name, base.ShapeText(c.Data.Shape()))
}
if rows >= 0 && c.Data.Len() != rows {
return base.Errf("%s: column %s has %d rows, want %d", name, c.Name, c.Data.Len(), rows)
}
rows = c.Data.Len()
}
if rows == 0 {
return base.Errf("%s: tables need at least one row", name)
}
// Row image: for each column its encoded byte width (binary) or
// character width (ASCII), and an encoder closure. An encoder
// reports an error when the rendered value does not fit its field.
var rowBytes int
type encoder func(dst []byte, row int) error
var encoders []encoder
var asciiEncoders []encoder
asciiWidths := make([]int, len(cols))
for i := range cols {
c := &cols[i]
form := forms[i]
if ascii {
width, prec, code, perr := parseASCIISaveForm(form)
if perr != nil {
return base.Errf("%s: column %s: %w", name, c.Name, perr)
}
asciiWidths[i] = width
rowBytes += width
switch code {
case 'A':
text := c.Text
if text == nil {
return base.Errf("%s: character column %s needs Text", name, c.Name)
}
asciiEncoders = append(asciiEncoders, func(dst []byte, row int) error {
if len(text[row]) > width {
return base.Errf("text %q is %d characters, form %s allows %d", text[row], len(text[row]), form, width)
}
copy(dst, text[row])
for p := len(text[row]); p < width; p++ {
dst[p] = ' '
}
return nil
})
case 'I':
col := c.Data
if col == nil || col.Dtype() != core.Int {
return base.Errf("%s: form %s needs an int column", name, form)
}
ints := col.RawInts()
asciiEncoders = append(asciiEncoders, func(dst []byte, row int) error {
word := fmt.Sprintf("%d", ints[row])
if len(word) > width {
return base.Errf("value %d does not fit the width %d of form %s", ints[row], width, form)
}
copy(dst, word)
for p := len(word); p < width; p++ {
dst[p] = ' '
}
return nil
})
case 'E', 'F', 'D':
col := c.Data
if col == nil || (col.Dtype() != core.Float && col.Dtype() != core.Float32) {
return base.Errf("%s: form %s needs a float column", name, form)
}
scientific := form[0] == 'E' || form[0] == 'D'
// Keep every value inside the declared width: shrink
// the printed precision to what the column holds
// rather than letting the field overflow it.
overhead := 2 // sign room and decimal point
if scientific {
overhead = 8 // sign, one digit, point and the e+/-xx tail
}
if width <= overhead {
return base.Errf("%s: form %s is too narrow for any value", name, form)
}
effPrec := min(prec, width-overhead)
// The payload the column's dtype actually carries: a
// float32 array keeps its values in the float32 slice
// alone, so reading the float64 one first indexes a nil
// slice.
isFloat32 := col.Dtype() == core.Float32
floats := col.RawFloats()
floats32 := col.RawFloat32s()
asciiEncoders = append(asciiEncoders, func(dst []byte, row int) error {
var v float64
if isFloat32 {
v = float64(floats32[row])
} else {
v = floats[row]
}
var word string
if scientific {
word = fmt.Sprintf("%*.*e", width, effPrec, v)
} else {
word = fmt.Sprintf("%*.*f", width, effPrec, v)
}
if len(word) > width {
return base.Errf("value %g does not fit the width %d of form %s", v, width, form)
}
copy(dst, word)
for p := len(word); p < width; p++ {
dst[p] = ' '
}
return nil
})
}
continue
}
switch {
case strings.HasSuffix(form, "A"):
width, aerr := strconv.Atoi(strings.TrimSuffix(form, "A"))
if aerr != nil || width < 1 {
return base.Errf("%s: character form %q needs a positive repeat", name, c.Form)
}
text := c.Text
if text == nil {
return base.Errf("%s: character column %s needs Text", name, c.Name)
}
for r := range rows {
if len(text[r]) > width {
return base.Errf("%s: row %d of %s is %d characters, form %q allows %d",
name, r, c.Name, len(text[r]), form, width)
}
}
off := rowBytes
encoders = append(encoders, func(dst []byte, row int) error {
copy(dst[off:], text[row])
for p := len(text[row]); p < width; p++ {
dst[off+p] = ' '
}
return nil
})
rowBytes += width
case form == "D":
if c.Data.Dtype() != core.Float {
return base.Errf("%s: form D needs a float64 column, %s is %s", name, c.Name, c.Data.Dtype())
}
off := rowBytes
floats := c.Data.RawFloats()
encoders = append(encoders, func(dst []byte, row int) error {
binary.BigEndian.PutUint64(dst[off:], math.Float64bits(floats[row]))
return nil
})
rowBytes += 8
case form == "E":
if c.Data.Dtype() != core.Float32 {
return base.Errf("%s: form E needs a float32 column, %s is %s", name, c.Name, c.Data.Dtype())
}
off := rowBytes
floats32 := c.Data.RawFloat32s()
encoders = append(encoders, func(dst []byte, row int) error {
binary.BigEndian.PutUint32(dst[off:], math.Float32bits(floats32[row]))
return nil
})
rowBytes += 4
case form == "K" || form == "L":
if c.Data.Dtype() != core.Int {
return base.Errf("%s: form %s needs an int column, %s is %s", name, form, c.Name, c.Data.Dtype())
}
off := rowBytes
ints := c.Data.RawInts()
if form == "K" {
encoders = append(encoders, func(dst []byte, row int) error {
binary.BigEndian.PutUint64(dst[off:], uint64(ints[row]))
return nil
})
rowBytes += 8
} else {
encoders = append(encoders, func(dst []byte, row int) error {
dst[off] = 'F'
if ints[row] != 0 {
dst[off] = 'T'
}
return nil
})
rowBytes += 1
}
default:
return base.Errf("%s: unsupported form %q, want D, E, K, L or nA", name, c.Form)
}
}
if ascii {
// ASCII rows separate neighbouring columns by one space.
rowBytes += len(cols) - 1
}
// Extension header.
ext := "BINTABLE"
if ascii {
ext = "TABLE"
}
cards := []string{
fitsStringCardRaw("XTENSION", ext),
fitsIntCard("BITPIX", 8),
fitsIntCard("NAXIS", 2),
fitsIntCard("NAXIS1", rowBytes),
fitsIntCard("NAXIS2", rows),
fitsIntCard("PCOUNT", 0),
fitsIntCard("GCOUNT", 1),
fitsIntCard("TFIELDS", len(cols)),
}
colStart := 1
for i := range cols {
c := &cols[i]
n := strconv.Itoa(i + 1)
nameCard, err := fitsStringCard("TTYPE"+n, c.Name)
if err != nil {
return base.Errf("%s: %w", name, err)
}
cards = append(cards, nameCard)
if ascii {
cards = append(cards, fitsIntCard("TBCOL"+n, colStart))
colStart += asciiWidths[i] + 1
}
formCard, err := fitsStringCard("TFORM"+n, forms[i])
if err != nil {
return base.Errf("%s: %w", name, err)
}
cards = append(cards, formCard)
if c.Unit != "" {
unitCard, uerr := fitsStringCard("TUNIT"+n, c.Unit)
if uerr != nil {
return base.Errf("%s: %w", name, uerr)
}
cards = append(cards, unitCard)
}
}
userCards, err := fitsUserCards(headers)
if err != nil {
return base.Errf("%s: %w", name, err)
}
cards = append(cards, userCards...)
cards = append(cards, fitsEndCard())
// Primary HDU comes first: a zero-axis image header.
primary := []string{
fitsBoolCard("SIMPLE", true),
fitsIntCard("BITPIX", 8),
fitsIntCard("NAXIS", 0),
fitsBoolCard("EXTEND", true),
fitsEndCard(),
}
// Both headers pad to whole blocks and the body is rows*rowBytes, so
// the file size is known before a byte is written.
out := make([]byte, 0,
fitsBlockSize(len(primary))+fitsBlockSize(len(cards))+rows*rowBytes+2880)
out = fitsAppendCards(out, primary)
out = fitsAppendCards(out, cards)
body := make([]byte, rows*rowBytes)
if ascii {
for r := range rows {
p := 0
for i := range cols {
if i > 0 {
body[r*rowBytes+p] = ' '
p++
}
if err := asciiEncoders[i](body[r*rowBytes+p:], r); err != nil {
return base.Errf("%s: column %s row %d: %w", name, cols[i].Name, r, err)
}
p += asciiWidths[i]
}
}
} else {
for r := range rows {
for _, enc := range encoders {
if err := enc(body[r*rowBytes:], r); err != nil {
return base.Errf("%s: row %d: %w", name, r, err)
}
}
}
}
out = append(out, body...)
out = fitsAppendZeroPad(out)
return os.WriteFile(path, out, 0o644)
}
// parseASCIISaveForm splits an ASCII-table column form into its
// character width, fractional digits and type code: "10A" gives
// (10, 0, 'A'), "I7" gives (7, 0, 'I'), "F14.6" gives (14, 6, 'F'),
// "E12.4" and "D20.12" give their widths and digits with the
// scientific code.
func parseASCIISaveForm(form string) (int, int, byte, error) {
if form == "" {
return 0, 0, 0, base.Errf("empty form")
}
// "nA" carries its repeat before the code; every other form puts
// the width after it, optionally with a .precision tail.
if form[len(form)-1] == 'A' {
head := form[:len(form)-1]
if head == "" {
return 1, 0, 'A', nil
}
width, cerr := strconv.Atoi(head)
if cerr != nil || width < 1 {
return 0, 0, 0, base.Errf("form %q needs a positive width", form)
}
return width, 0, 'A', nil
}
code := form[0]
switch code {
case 'I', 'F', 'E', 'D':
default:
return 0, 0, 0, base.Errf("form %q has no known type code", form)
}
rest := form[1:]
before, after, ok := strings.Cut(rest, ".")
num := rest
if ok {
num = before
}
width, cerr := strconv.Atoi(num)
if cerr != nil || width < 1 {
return 0, 0, 0, base.Errf("form %q needs a positive width", form)
}
prec := 0
if ok {
p, perr := strconv.Atoi(after)
if perr != nil {
return 0, 0, 0, base.Errf("form %q has a malformed precision", form)
}
prec = p
}
return width, prec, code, nil
}
// fitsStringCardRaw renders a string card without the reserved-keyword
// check (XTENSION is reserved but the table writes it itself).
func fitsStringCardRaw(keyword, v string) string {
card, _ := fitsStringCard(keyword, v)
return card
}
// LoadFITSTable reads the first table extension (XTENSION 'BINTABLE'
// or 'TABLE') of a FITS file, skipping the primary HDU and any images
// before it. The result carries every column: numeric ones as arrays,
// character ones as string slices.
func LoadFITSTable(path string) (*FITSTable, error) {
data, err := os.ReadFile(path)
if err != nil {
return nil, base.Errf("LoadFITSTable: %w", err)
}
off := 0
for {
table, next, terr := parseFITSTableHDU(data, off)
if terr != nil {
return nil, base.Errf("LoadFITSTable: %w", terr)
}
if table != nil {
return table, nil
}
if next <= off {
return nil, base.Errf("LoadFITSTable: the file stalled at byte %d", off)
}
off = next
if off >= len(data) {
return nil, base.Errf("LoadFITSTable: no table extension found")
}
}
}
// parseFITSTableHDU parses one HDU starting at off. A nil table with
// a positive next means "an image HDU, carry on"; a table comes back
// fully populated.
func parseFITSTableHDU(data []byte, off int) (*FITSTable, int, error) {
if off+80 > len(data) {
return nil, 0, base.Errf("file ends inside a header at byte %d", off)
}
keyword := strings.TrimRight(string(data[off:off+8]), " ")
isTable := keyword == "XTENSION"
cards, headerLen, err := scanFITSCards(data, off)
if err != nil {
return nil, 0, err
}
dataAt := off + headerLen
// The keyword map answers lookups without the per-lookup scan the
// linear walk cost. A repeated keyword keeps its first value, which
// is what the scan it replaces returned, so a hostile header that
// repeats a structural card is read exactly as before.
first := make(map[string]string, len(cards))
for _, c := range cards {
if _, ok := first[c.key]; !ok {
first[c.key] = c.value
}
}
get := func(key string) string { return first[key] }
atoi := func(key string) (int, error) {
v := get(key)
n, cerr := strconv.Atoi(v)
if cerr != nil {
return 0, base.Errf("%s = %q is not an integer", key, v)
}
return n, nil
}
// atoiN is the numbered-keyword form: "NAXIS1" is prefix NAXIS and
// index 1, formatted into a stack buffer, with the same error text
// the concatenated key produced.
atoiN := func(prefix string, n int) (int, error) {
v := fitsLookupN(first, prefix, n)
parsed, cerr := strconv.Atoi(v)
if cerr != nil {
return 0, base.Errf("%s%d = %q is not an integer", prefix, n, v)
}
return parsed, nil
}
bitpix, berr := atoi("BITPIX")
if berr != nil {
return nil, 0, berr
}
naxis, aerr := atoi("NAXIS")
if aerr != nil {
return nil, 0, aerr
}
var dims []int
for j := 1; j <= naxis; j++ {
d, derr := atoiN("NAXIS", j)
if derr != nil {
return nil, 0, derr
}
dims = append(dims, d)
}
pcount := 0
if v := get("PCOUNT"); v != "" {
p, perr := atoi("PCOUNT")
if perr != nil {
return nil, 0, perr
}
pcount = p
}
gcount := 1
if v := get("GCOUNT"); v != "" {
g, gerr := atoi("GCOUNT")
if gerr != nil {
return nil, 0, gerr
}
gcount = g
}
if !isTable {
// An image HDU: skip its data block. The size is |BITPIX|/8
// bytes per element times GCOUNT*(PCOUNT + every NAXISn); positive
// BITPIX (the integer depths) counts the same as negative, and
// the axis extents multiply rather than add. Every factor is
// bounded against the bytes the file actually holds before it
// multiplies, the same discipline parseFITS applies, so a
// hostile product cannot wrap into a skip to a chosen offset.
width := 1
if bitpix < 0 {
width = -bitpix / 8
} else {
width = bitpix / 8
}
if width <= 0 {
return nil, 0, base.Errf("BITPIX = %d gives a non-positive element width", bitpix)
}
avail := max(int64(len(data)-dataAt), 0)
total := 0
if naxis > 0 {
prod := int64(1)
for _, d := range dims {
if d < 0 {
return nil, 0, base.Errf("image HDU declares the negative axis %d", d)
}
// A zero axis empties the data block whatever follows:
// the bound divides by the running product, so the
// division runs only while the product is live.
if prod > 0 && int64(d) > avail/int64(width)/prod {
return nil, 0, base.Errf("image HDU declares more data than the %d bytes the file holds", len(data)-dataAt)
}
prod *= int64(d)
}
if int64(pcount) < 0 || int64(pcount) > avail {
return nil, 0, base.Errf("image HDU declares a PCOUNT of %d against %d bytes", pcount, avail)
}
space := int64(pcount) + prod
switch {
case space == 0:
// A zero-sized group repeats into nothing whatever
// GCOUNT claims.
total = 0
case int64(gcount) > 1 && int64(gcount) > avail/space:
return nil, 0, base.Errf("image HDU declares more data than the %d bytes the file holds", len(data)-dataAt)
default:
total = int(int64(gcount) * space)
}
}
if total < 0 {
return nil, 0, base.Errf("image HDU declares a negative data size")
}
return nil, dataAt + fitsBlockSize(total*width), nil
}
ext := strings.Trim(get("XTENSION"), " ")
if ext != "BINTABLE" && ext != "TABLE" {
// An unknown extension: refuse rather than guess its size.
return nil, 0, base.Errf("unsupported extension %q", ext)
}
if naxis != 2 || len(dims) != 2 {
return nil, 0, base.Errf("%s extension must have NAXIS = 2, got %d", ext, naxis)
}
tfields, ferr := atoi("TFIELDS")
if ferr != nil {
return nil, 0, ferr
}
// A negative count would size the per-column slices below with a
// negative length, which panics; it is a header that lies, so it is
// refused by name.
if tfields < 0 {
return nil, 0, base.Errf("TFIELDS = %d is negative", tfields)
}
rows := dims[1]
if rows < 1 {
return nil, 0, base.Errf("table declares NAXIS2 = %d, want at least 1 row", rows)
}
// The count is bounded by what could back it, the way the NetCDF
// reader bounds its own: every column of either table kind occupies
// at least one byte of every row, so a TFIELDS past the data the
// file holds is a claim nothing can follow, and refusing it here
// keeps the per-column preallocations from being sized by the claim.
if left := int64(len(data) - dataAt); int64(tfields) > left/int64(rows) {
return nil, 0, base.Errf("TFIELDS = %d names more columns than the %d bytes of table data can hold", tfields, left)
}
table := &FITSTable{
Kind: ext,
Rows: rows,
Headers: map[string]string{},
}
type parsed struct {
name, unit, form string
tbcol int
}
var cols []parsed
for i := 1; i <= tfields; i++ {
col := parsed{
name: fitsLookupN(first, "TTYPE", i),
unit: fitsLookupN(first, "TUNIT", i),
form: strings.ToUpper(strings.TrimSpace(fitsLookupN(first, "TFORM", i))),
tbcol: 0,
}
if v := fitsLookupN(first, "TBCOL", i); v != "" {
tb, tberr := strconv.Atoi(strings.TrimSpace(v))
if tberr != nil {
return nil, 0, base.Errf("column %d: TBCOL %q is not an integer", i, v)
}
col.tbcol = tb
}
if col.form == "" {
return nil, 0, base.Errf("column %d has no TFORM", i)
}
cols = append(cols, col)
table.Names = append(table.Names, col.name)
table.Units = append(table.Units, col.unit)
}
headers := table.Headers
for _, c := range cards {
switch c.key {
case "XTENSION", "BITPIX", "NAXIS", "PCOUNT", "GCOUNT", "TFIELDS":
default:
if !fitsAxisKeyword(c.key) &&
!strings.HasPrefix(c.key, "TTYPE") &&
!strings.HasPrefix(c.key, "TFORM") &&
!strings.HasPrefix(c.key, "TUNIT") &&
!strings.HasPrefix(c.key, "TBCOL") {
headers[c.key] = c.value
}
}
}
if ext == "BINTABLE" {
// The data block starts on a 2880-byte block boundary, so a file
// that ends inside the header has no data at all. The row
// arithmetic below indexes from dataAt, which must be checked
// against the file before it is used.
if dataAt > len(data) {
return nil, 0, base.Errf("table data starts at byte %d of a %d-byte file", dataAt, len(data))
}
widths := make([]int, tfields)
forms := make([]string, tfields)
// Column offsets inside a row: prefix sums of the widths,
// computed once instead of once per row.
offs := make([]int, tfields)
rowBytes := 0
// The widest row any present data can hold: every width is
// bounded against it as it accumulates, so the prefix sum can
// never wrap past the guard below into a small positive value.
maxRow := (int64(len(data)) - int64(dataAt)) / int64(rows)
for i, col := range cols {
r, form, perr := parseTFORM(col.form)
if perr != nil {
return nil, 0, base.Errf("column %d: %w", i+1, perr)
}
if form == "A" {
// A character column's repeat is the string width:
// "16A" is one 16-character field, the layout every
// catalogue uses. The numeric repeat rule below would
// reject the library's own output.
if r < 1 {
return nil, 0, base.Errf("column %d: width %d is not positive", i+1, r)
}
forms[i] = "A"
widths[i] = r
} else {
if r != 1 {
return nil, 0, base.Errf("column %d: repeat %d is not supported, want a scalar", i+1, r)
}
widths[i] = tfSize(form)
// A zero width is an unknown code, or a variable-length
// descriptor ("P", "Q") the reader cannot size. Refusing
// it here keeps the row arithmetic below meaningful: a
// zero-width column would defeat the truncation guard and
// push the reads past the end of the file.
if widths[i] == 0 {
return nil, 0, base.Errf("column %d: unsupported TFORM %q", i+1, form)
}
forms[i] = form
}
if int64(widths[i]) > maxRow-int64(rowBytes) {
return nil, 0, base.Errf("column %d of width %d leaves the %d bytes of row the data holds", i+1, widths[i], maxRow)
}
offs[i] = rowBytes
rowBytes += widths[i]
}
// Bound before multiplying: a hostile NAXIS2 must not overflow
// the product (mirroring parseFITS's guard), it just means the
// declared data cannot fit the file.
if rowBytes > 0 && rows > (len(data)-dataAt)/rowBytes {
return nil, 0, base.Errf("table data is truncated")
}
for i := range cols {
form := forms[i]
if form == "A" {
table.Text = append(table.Text, fitsTextColumn(data, rows, dataAt+offs[i], rowBytes, widths[i]))
table.Columns = append(table.Columns, nil)
continue
}
var dt core.Dtype = core.Float
switch form {
case "B", "I", "J", "K", "L":
dt = core.Int
case "E":
dt = core.Float32
}
arr := core.New(dt, rows)
// The form decides the loop once per column, and the cell goes
// straight from the file's bytes into the array's payload: no
// cell is boxed into an any and the row body holds no type
// switch. colBase is the column's first byte, so a row's cell
// starts rowBytes further on.
colBase := dataAt + offs[i]
switch form {
case "D":
dst := arr.RawFloats()
for row := range rows {
dst[row] = math.Float64frombits(binary.BigEndian.Uint64(data[colBase+rowBytes*row:]))
}
case "E":
dst := arr.RawFloat32s()
for row := range rows {
dst[row] = math.Float32frombits(binary.BigEndian.Uint32(data[colBase+rowBytes*row:]))
}
case "K":
dst := arr.RawInts()
for row := range rows {
dst[row] = int64(binary.BigEndian.Uint64(data[colBase+rowBytes*row:]))
}
case "J":
dst := arr.RawInts()
for row := range rows {
dst[row] = int64(int32(binary.BigEndian.Uint32(data[colBase+rowBytes*row:])))
}
case "I":
dst := arr.RawInts()
for row := range rows {
dst[row] = int64(int16(binary.BigEndian.Uint16(data[colBase+rowBytes*row:])))
}
case "B":
dst := arr.RawInts()
for row := range rows {
dst[row] = int64(data[colBase+rowBytes*row])
}
case "L":
// A false logical leaves the zero the payload was
// allocated with.
dst := arr.RawInts()
for row := range rows {
if data[colBase+rowBytes*row] == 'T' {
dst[row] = 1
}
}
}
table.Columns = append(table.Columns, scaleColumn(table.Headers, i+1, arr))
table.Text = append(table.Text, nil)
}
return table, dataAt + fitsBlockSize(rows*rowBytes), nil
}
// ASCII table: NAXIS1-character text rows.
widths := make([]int, tfields)
starts := make([]int, tfields)
rowBytes := dims[0]
for i, col := range cols {
w, perr := parseASCIITFORM(col.form)
if perr != nil {
return nil, 0, base.Errf("column %d: %w", i+1, perr)
}
widths[i] = w
starts[i] = col.tbcol - 1
// The bound is written without the sum, which a hostile TBCOL
// would wrap past: starts + width must leave the row.
if starts[i] < 0 || w > rowBytes || starts[i] > rowBytes-w {
return nil, 0, base.Errf("column %d: TBCOL %d with width %d leaves the row", i+1, col.tbcol, w)
}
}
// Bound before multiplying, as in the binary branch above.
if rowBytes > 0 && rows > (len(data)-dataAt)/rowBytes {
return nil, 0, base.Errf("table data is truncated")
}
for i, col := range cols {
if strings.HasSuffix(col.form, "A") {
table.Text = append(table.Text, fitsTextColumn(data, rows, dataAt+starts[i], rowBytes, widths[i]))
table.Columns = append(table.Columns, nil)
continue
}
isInt := strings.HasPrefix(col.form, "I")
var dt core.Dtype = core.Float
if isInt {
dt = core.Int
}
arr := core.New(dt, rows)
// The cell is parsed where it lies: the trimmed field is a view of
// the row's bytes, not a copy, and the strconv call sees exactly
// the text the string form saw.
colBase := dataAt + starts[i]
for row := range rows {
p := colBase + rowBytes*row
field := bytes.TrimSpace(data[p : p+widths[i]])
if len(field) == 0 {
continue
}
if isInt {
v, cerr := strconv.ParseInt(fitsFieldString(field), 10, 64)
if cerr != nil {
return nil, 0, base.Errf("column %d row %d: %q is not an integer", i+1, row, string(field))
}
arr.RawInts()[row] = v
} else {
v, ok := fitsASCIIFloat(fitsFieldString(field))
if !ok {
return nil, 0, base.Errf("column %d row %d: %q is not a number", i+1, row, string(field))
}
arr.RawFloats()[row] = v
}
}
table.Columns = append(table.Columns, scaleColumn(table.Headers, i+1, arr))
table.Text = append(table.Text, nil)
}
return table, dataAt + fitsBlockSize(rows*rowBytes), nil
}
// parseTFORM splits a binary TFORM into its repeat count and type
// code: "3E" gives (3, "E"), "D" gives (1, "D"), "16A" gives (16, "A").
func parseTFORM(form string) (int, string, error) {
if form == "" {
return 0, "", base.Errf("empty TFORM")
}
repeat, code := 1, form
for i, r := range form {
if r < '0' || r > '9' {
code = form[i:]
if i > 0 {
v, aerr := strconv.Atoi(form[:i])
if aerr != nil {
// An overflowing repeat count must not silently
// fall back to 1: a hostile "99999999999999999999E"
// would be read as a scalar instead of refused.
return 0, "", base.Errf("TFORM %q: the repeat count does not fit an integer", form)
}
repeat = v
}
break
}
}
return repeat, code, nil
}
// fitsTextColumn builds a character column's strings from the
// fixed-width cells that start at colBase and repeat every rowStride
// bytes, trailing spaces trimmed. The trimmed cell bytes are copied
// into one slab and every string of the column aliases its own range
// of that slab, so the column costs one allocation whatever its row
// count. The caller's guards have already bounded colBase + rowStride*
// (rows-1) + width against the file's bytes.
//
// SAFETY: the slab is fully written before the first string is formed
// from it, the aliased ranges are disjoint, and nothing writes to the
// slab afterwards, so each string keeps exactly the bytes the copy
// left there for as long as it lives.
func fitsTextColumn(data []byte, rows, colBase, rowStride, width int) []string {
text := make([]string, rows)
total := 0
for row := range rows {
base := colBase + rowStride*row
p := base + width
for p > base && data[p-1] == ' ' {
p--
}
total += p - base
}
slab := make([]byte, total)
at := 0
for row := range rows {
base := colBase + rowStride*row
p := base + width
for p > base && data[p-1] == ' ' {
p--
}
n := p - base
copy(slab[at:at+n], data[base:p])
if n == 0 {
text[row] = ""
} else {
text[row] = unsafe.String(unsafe.SliceData(slab[at:at+n]), n)
}
at += n
}
return text
}
// scaleColumn applies the TSCALn/TZEROn affine map (physical = raw*
// scale + zero) to a decoded numeric column. Integer scaling that
// stays integral keeps the int dtype (the unsigned-integer convention
// TZERO = 32768/2147483648 lands here); any fractional or overflowing
// scaling promotes the column to float64, mirroring cfitsio. A
// malformed value is treated as unscaled: the raw storage values are
// the best available answer and an error here would lose the whole
// table over one keyword.
func scaleColumn(headers map[string]string, n int, arr *core.Array) *core.Array {
ss := fitsLookupN(headers, "TSCAL", n)
zs := fitsLookupN(headers, "TZERO", n)
if ss == "" && zs == "" {
return arr
}
scale, zero := 1.0, 0.0
if ss != "" {
if v, err := strconv.ParseFloat(strings.TrimSpace(ss), 64); err == nil {
scale = v
}
}
if zs != "" {
if v, err := strconv.ParseFloat(strings.TrimSpace(zs), 64); err == nil {
zero = v
}
}
if scale == 1 && zero == 0 {
return arr
}
switch arr.Dtype() {
case core.Float:
raw := arr.RawFloats()
for i := range raw {
raw[i] = raw[i]*scale + zero
}
return arr
case core.Float32:
// Scaled values leave the float32 guarantee; keep the full
// float64 result instead of rounding twice.
raw32 := arr.RawFloat32s()
out := core.New(core.Float, arr.Shape()...)
raw := out.RawFloats()
for i := range raw32 {
raw[i] = float64(raw32[i])*scale + zero
}
return out
default:
rawInts := arr.RawInts()
if scale == math.Trunc(scale) && zero == math.Trunc(zero) &&
math.Abs(scale) < 1e15 && math.Abs(zero) < 9e15 {
out := core.New(core.Int, arr.Shape()...)
dst := out.RawInts()
integral := true
for i, v := range rawInts {
sv := float64(v)*scale + zero
// The exact float64 bounds of the int64 range: MaxInt64
// is 9223372036854775807, which no float64 holds, so the
// first value past it is 2^63, while MinInt64 is exact.
// A conversion out of range is implementation-defined
// (amd64 answers MinInt64), which is the silent corruption
// this path exists to avoid.
if sv >= 9223372036854775808.0 || sv < -9223372036854775808.0 {
integral = false
break
}
dst[i] = int64(sv)
}
if integral {
return out
}
}
out := core.New(core.Float, arr.Shape()...)
raw := out.RawFloats()
for i, v := range rawInts {
raw[i] = float64(v)*scale + zero
}
return out
}
}
// tfSize returns the byte width of one element of a binary column.
func tfSize(code string) int {
switch code {
case "L", "B", "A":
return 1
case "I":
return 2
case "J", "E":
return 4
case "K", "D":
return 8
}
return 0
}
// parseASCIITFORM returns the character width of an ASCII-table
// column: "10A" gives 10, "I7" gives 7, "F14.6" gives 14, "E12.4"
// gives 12, "D20.12" gives 20.
func parseASCIITFORM(form string) (int, error) {
w, _, _, err := parseASCIISaveForm(strings.ToUpper(strings.TrimSpace(form)))
return w, err
}
// fitsASCIIFloat parses one ASCII-table cell. Beyond the forms
// strconv.ParseFloat accepts, the cells of a Fortran-written table
// carry the Dw.d exponent ("1.5D+03", the D column type this reader
// accepts) and the exponent-less form ("1.5+03"), where the sign after
// the mantissa introduces the exponent; both mean what the same text
// with an E means, and refusing them refused the whole table.
func fitsASCIIFloat(field string) (float64, bool) {
if v, err := strconv.ParseFloat(field, 64); err == nil {
return v, true
}
norm := field
switch i := strings.IndexAny(norm, "dD"); {
case i >= 0:
// A Fortran D exponent is the E exponent under another letter.
norm = norm[:i] + "E" + norm[i+1:]
case strings.IndexAny(norm, "eE") < 0:
// No exponent at all: the exponent-less Fortran form puts the
// exponent's sign directly behind the mantissa, "1.5+03".
if j := strings.IndexAny(norm[1:], "+-"); j >= 0 {
norm = norm[:j+1] + "E" + norm[j+1:]
}
}
if norm == field {
return 0, false
}
v, err := strconv.ParseFloat(norm, 64)
if err != nil {
return 0, false
}
return v, true
}
// fitsFieldString views one trimmed ASCII-table field as a string
// without copying it, the form strconv needs and the byte slice
// already holds.
//
// SAFETY: the view aliases the file's bytes for the length of one
// strconv call, which reads the string and retains nothing of it. The
// parse errors are formatted from a copy, so no view of the file
// outlives the slice it points into.
func fitsFieldString(b []byte) string {
return unsafe.String(unsafe.SliceData(b), len(b))
}
// fitsCheckString validates a human-supplied string field.
func fitsCheckString(v, what string) error {
if len(v) == 0 {
return base.Errf("%s must not be empty", what)
}
if len(v) > 68 {
return base.Errf("%s is %d characters, at most 68 fit a card", what, len(v))
}
return nil
}