// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package io import ( "bytes" "encoding/binary" "fmt" "math" "os" "strconv" "strings" "unsafe" "sourcedock.dev/petrbalvin/tensor/internal/base" "sourcedock.dev/petrbalvin/tensor/internal/core" ) // FITS table extensions. Catalogues, source lists and observation // logs live in the format's table HDUs: a binary table (XTENSION // 'BINTABLE') packs each column big-endian according to its TFORM // descriptor, an ASCII table (XTENSION 'TABLE') lays fixed-width text // columns into rows of NAXIS1 characters. Both share the header // vocabulary: TTYPEn names a column, TUNITn gives its unit, TFIELDS // counts them. // FITSTableColumn describes one column of a FITS table. Form carries // the TFORM-style descriptor: "D" (float64), "E" (float32), "K" // (int64), "L" (logical), "B" (unsigned byte), "I" (16-bit) or "J" // (32-bit) for a binary table, and "nA" (a character string of n // characters) for either kind; "Inw", "Fw.d", "Ew.d" or "Dw.d" lay out // an ASCII table column, where w is the column width in characters. A // numeric form reads its values from Data (Float answers D, Float32 // answers E, Int answers K, L, B, I and J); a character form reads // them from Text. type FITSTableColumn struct { Name string Unit string Form string Data *core.Array Text []string } // FITSTable is a parsed table extension. Names, Units, Columns and // Text run parallel to the file's column list: a numeric column holds // its values in Columns and nil in Text, a character column holds its // strings in Text and nil in Columns. type FITSTable struct { Kind string // "BINTABLE" or "TABLE" Names []string Units []string Columns []*core.Array Text [][]string Rows int Headers map[string]string } // SaveFITSTable writes a primary HDU followed by one table extension: // a binary table by default, an ASCII table when ascii is set. The // headers land in the extension's header beside the column cards. A // column with a numeric form but nil Data, a character form without // Text, mismatched row counts, an unknown form, an empty column list, // or a value that does not fit its ASCII form's width is an error. func SaveFITSTable(path string, ascii bool, cols []FITSTableColumn, headers map[string]string) error { const name = "SaveFITSTable" if len(cols) == 0 { return base.Errf("%s: at least one column is required", name) } rows := -1 forms := make([]string, len(cols)) for i := range cols { c := &cols[i] if err := fitsCheckString(c.Name, "column name"); err != nil { return base.Errf("%s: %w", name, err) } form := strings.ToUpper(strings.TrimSpace(c.Form)) forms[i] = form if form == "" { return base.Errf("%s: column %d (%s) has an empty form", name, i, c.Name) } if strings.HasSuffix(form, "A") { if c.Text == nil { return base.Errf("%s: character column %s needs Text", name, c.Name) } if c.Data != nil { return base.Errf("%s: character column %s must not carry Data", name, c.Name) } if rows >= 0 && len(c.Text) != rows { return base.Errf("%s: column %s has %d rows, want %d", name, c.Name, len(c.Text), rows) } rows = len(c.Text) continue } if c.Data == nil { return base.Errf("%s: numeric column %s needs Data", name, c.Name) } if c.Data.NDim() != 1 { return base.Errf("%s: column %s must be a vector, got shape %s", name, c.Name, base.ShapeText(c.Data.Shape())) } if rows >= 0 && c.Data.Len() != rows { return base.Errf("%s: column %s has %d rows, want %d", name, c.Name, c.Data.Len(), rows) } rows = c.Data.Len() } if rows == 0 { return base.Errf("%s: tables need at least one row", name) } // Row image: for each column its encoded byte width (binary) or // character width (ASCII), and an encoder closure. An encoder // reports an error when the rendered value does not fit its field. var rowBytes int type encoder func(dst []byte, row int) error var encoders []encoder var asciiEncoders []encoder asciiWidths := make([]int, len(cols)) for i := range cols { c := &cols[i] form := forms[i] if ascii { width, prec, code, perr := parseASCIISaveForm(form) if perr != nil { return base.Errf("%s: column %s: %w", name, c.Name, perr) } asciiWidths[i] = width rowBytes += width switch code { case 'A': text := c.Text if text == nil { return base.Errf("%s: character column %s needs Text", name, c.Name) } asciiEncoders = append(asciiEncoders, func(dst []byte, row int) error { if len(text[row]) > width { return base.Errf("text %q is %d characters, form %s allows %d", text[row], len(text[row]), form, width) } copy(dst, text[row]) for p := len(text[row]); p < width; p++ { dst[p] = ' ' } return nil }) case 'I': col := c.Data if col == nil || col.Dtype() != core.Int { return base.Errf("%s: form %s needs an int column", name, form) } ints := col.RawInts() asciiEncoders = append(asciiEncoders, func(dst []byte, row int) error { word := fmt.Sprintf("%d", ints[row]) if len(word) > width { return base.Errf("value %d does not fit the width %d of form %s", ints[row], width, form) } copy(dst, word) for p := len(word); p < width; p++ { dst[p] = ' ' } return nil }) case 'E', 'F', 'D': col := c.Data if col == nil || (col.Dtype() != core.Float && col.Dtype() != core.Float32) { return base.Errf("%s: form %s needs a float column", name, form) } scientific := form[0] == 'E' || form[0] == 'D' // Keep every value inside the declared width: shrink // the printed precision to what the column holds // rather than letting the field overflow it. overhead := 2 // sign room and decimal point if scientific { overhead = 8 // sign, one digit, point and the e+/-xx tail } if width <= overhead { return base.Errf("%s: form %s is too narrow for any value", name, form) } effPrec := min(prec, width-overhead) // The payload the column's dtype actually carries: a // float32 array keeps its values in the float32 slice // alone, so reading the float64 one first indexes a nil // slice. isFloat32 := col.Dtype() == core.Float32 floats := col.RawFloats() floats32 := col.RawFloat32s() asciiEncoders = append(asciiEncoders, func(dst []byte, row int) error { var v float64 if isFloat32 { v = float64(floats32[row]) } else { v = floats[row] } var word string if scientific { word = fmt.Sprintf("%*.*e", width, effPrec, v) } else { word = fmt.Sprintf("%*.*f", width, effPrec, v) } if len(word) > width { return base.Errf("value %g does not fit the width %d of form %s", v, width, form) } copy(dst, word) for p := len(word); p < width; p++ { dst[p] = ' ' } return nil }) } continue } switch { case strings.HasSuffix(form, "A"): width, aerr := strconv.Atoi(strings.TrimSuffix(form, "A")) if aerr != nil || width < 1 { return base.Errf("%s: character form %q needs a positive repeat", name, c.Form) } text := c.Text if text == nil { return base.Errf("%s: character column %s needs Text", name, c.Name) } for r := range rows { if len(text[r]) > width { return base.Errf("%s: row %d of %s is %d characters, form %q allows %d", name, r, c.Name, len(text[r]), form, width) } } off := rowBytes encoders = append(encoders, func(dst []byte, row int) error { copy(dst[off:], text[row]) for p := len(text[row]); p < width; p++ { dst[off+p] = ' ' } return nil }) rowBytes += width case form == "D": if c.Data.Dtype() != core.Float { return base.Errf("%s: form D needs a float64 column, %s is %s", name, c.Name, c.Data.Dtype()) } off := rowBytes floats := c.Data.RawFloats() encoders = append(encoders, func(dst []byte, row int) error { binary.BigEndian.PutUint64(dst[off:], math.Float64bits(floats[row])) return nil }) rowBytes += 8 case form == "E": if c.Data.Dtype() != core.Float32 { return base.Errf("%s: form E needs a float32 column, %s is %s", name, c.Name, c.Data.Dtype()) } off := rowBytes floats32 := c.Data.RawFloat32s() encoders = append(encoders, func(dst []byte, row int) error { binary.BigEndian.PutUint32(dst[off:], math.Float32bits(floats32[row])) return nil }) rowBytes += 4 case form == "K" || form == "L": if c.Data.Dtype() != core.Int { return base.Errf("%s: form %s needs an int column, %s is %s", name, form, c.Name, c.Data.Dtype()) } off := rowBytes ints := c.Data.RawInts() if form == "K" { encoders = append(encoders, func(dst []byte, row int) error { binary.BigEndian.PutUint64(dst[off:], uint64(ints[row])) return nil }) rowBytes += 8 } else { encoders = append(encoders, func(dst []byte, row int) error { dst[off] = 'F' if ints[row] != 0 { dst[off] = 'T' } return nil }) rowBytes += 1 } default: return base.Errf("%s: unsupported form %q, want D, E, K, L or nA", name, c.Form) } } if ascii { // ASCII rows separate neighbouring columns by one space. rowBytes += len(cols) - 1 } // Extension header. ext := "BINTABLE" if ascii { ext = "TABLE" } cards := []string{ fitsStringCardRaw("XTENSION", ext), fitsIntCard("BITPIX", 8), fitsIntCard("NAXIS", 2), fitsIntCard("NAXIS1", rowBytes), fitsIntCard("NAXIS2", rows), fitsIntCard("PCOUNT", 0), fitsIntCard("GCOUNT", 1), fitsIntCard("TFIELDS", len(cols)), } colStart := 1 for i := range cols { c := &cols[i] n := strconv.Itoa(i + 1) nameCard, err := fitsStringCard("TTYPE"+n, c.Name) if err != nil { return base.Errf("%s: %w", name, err) } cards = append(cards, nameCard) if ascii { cards = append(cards, fitsIntCard("TBCOL"+n, colStart)) colStart += asciiWidths[i] + 1 } formCard, err := fitsStringCard("TFORM"+n, forms[i]) if err != nil { return base.Errf("%s: %w", name, err) } cards = append(cards, formCard) if c.Unit != "" { unitCard, uerr := fitsStringCard("TUNIT"+n, c.Unit) if uerr != nil { return base.Errf("%s: %w", name, uerr) } cards = append(cards, unitCard) } } userCards, err := fitsUserCards(headers) if err != nil { return base.Errf("%s: %w", name, err) } cards = append(cards, userCards...) cards = append(cards, fitsEndCard()) // Primary HDU comes first: a zero-axis image header. primary := []string{ fitsBoolCard("SIMPLE", true), fitsIntCard("BITPIX", 8), fitsIntCard("NAXIS", 0), fitsBoolCard("EXTEND", true), fitsEndCard(), } // Both headers pad to whole blocks and the body is rows*rowBytes, so // the file size is known before a byte is written. out := make([]byte, 0, fitsBlockSize(len(primary))+fitsBlockSize(len(cards))+rows*rowBytes+2880) out = fitsAppendCards(out, primary) out = fitsAppendCards(out, cards) body := make([]byte, rows*rowBytes) if ascii { for r := range rows { p := 0 for i := range cols { if i > 0 { body[r*rowBytes+p] = ' ' p++ } if err := asciiEncoders[i](body[r*rowBytes+p:], r); err != nil { return base.Errf("%s: column %s row %d: %w", name, cols[i].Name, r, err) } p += asciiWidths[i] } } } else { for r := range rows { for _, enc := range encoders { if err := enc(body[r*rowBytes:], r); err != nil { return base.Errf("%s: row %d: %w", name, r, err) } } } } out = append(out, body...) out = fitsAppendZeroPad(out) return os.WriteFile(path, out, 0o644) } // parseASCIISaveForm splits an ASCII-table column form into its // character width, fractional digits and type code: "10A" gives // (10, 0, 'A'), "I7" gives (7, 0, 'I'), "F14.6" gives (14, 6, 'F'), // "E12.4" and "D20.12" give their widths and digits with the // scientific code. func parseASCIISaveForm(form string) (int, int, byte, error) { if form == "" { return 0, 0, 0, base.Errf("empty form") } // "nA" carries its repeat before the code; every other form puts // the width after it, optionally with a .precision tail. if form[len(form)-1] == 'A' { head := form[:len(form)-1] if head == "" { return 1, 0, 'A', nil } width, cerr := strconv.Atoi(head) if cerr != nil || width < 1 { return 0, 0, 0, base.Errf("form %q needs a positive width", form) } return width, 0, 'A', nil } code := form[0] switch code { case 'I', 'F', 'E', 'D': default: return 0, 0, 0, base.Errf("form %q has no known type code", form) } rest := form[1:] before, after, ok := strings.Cut(rest, ".") num := rest if ok { num = before } width, cerr := strconv.Atoi(num) if cerr != nil || width < 1 { return 0, 0, 0, base.Errf("form %q needs a positive width", form) } prec := 0 if ok { p, perr := strconv.Atoi(after) if perr != nil { return 0, 0, 0, base.Errf("form %q has a malformed precision", form) } prec = p } return width, prec, code, nil } // fitsStringCardRaw renders a string card without the reserved-keyword // check (XTENSION is reserved but the table writes it itself). func fitsStringCardRaw(keyword, v string) string { card, _ := fitsStringCard(keyword, v) return card } // LoadFITSTable reads the first table extension (XTENSION 'BINTABLE' // or 'TABLE') of a FITS file, skipping the primary HDU and any images // before it. The result carries every column: numeric ones as arrays, // character ones as string slices. func LoadFITSTable(path string) (*FITSTable, error) { data, err := os.ReadFile(path) if err != nil { return nil, base.Errf("LoadFITSTable: %w", err) } off := 0 for { table, next, terr := parseFITSTableHDU(data, off) if terr != nil { return nil, base.Errf("LoadFITSTable: %w", terr) } if table != nil { return table, nil } if next <= off { return nil, base.Errf("LoadFITSTable: the file stalled at byte %d", off) } off = next if off >= len(data) { return nil, base.Errf("LoadFITSTable: no table extension found") } } } // parseFITSTableHDU parses one HDU starting at off. A nil table with // a positive next means "an image HDU, carry on"; a table comes back // fully populated. func parseFITSTableHDU(data []byte, off int) (*FITSTable, int, error) { if off+80 > len(data) { return nil, 0, base.Errf("file ends inside a header at byte %d", off) } keyword := strings.TrimRight(string(data[off:off+8]), " ") isTable := keyword == "XTENSION" cards, headerLen, err := scanFITSCards(data, off) if err != nil { return nil, 0, err } dataAt := off + headerLen // The keyword map answers lookups without the per-lookup scan the // linear walk cost. A repeated keyword keeps its first value, which // is what the scan it replaces returned, so a hostile header that // repeats a structural card is read exactly as before. first := make(map[string]string, len(cards)) for _, c := range cards { if _, ok := first[c.key]; !ok { first[c.key] = c.value } } get := func(key string) string { return first[key] } atoi := func(key string) (int, error) { v := get(key) n, cerr := strconv.Atoi(v) if cerr != nil { return 0, base.Errf("%s = %q is not an integer", key, v) } return n, nil } // atoiN is the numbered-keyword form: "NAXIS1" is prefix NAXIS and // index 1, formatted into a stack buffer, with the same error text // the concatenated key produced. atoiN := func(prefix string, n int) (int, error) { v := fitsLookupN(first, prefix, n) parsed, cerr := strconv.Atoi(v) if cerr != nil { return 0, base.Errf("%s%d = %q is not an integer", prefix, n, v) } return parsed, nil } bitpix, berr := atoi("BITPIX") if berr != nil { return nil, 0, berr } naxis, aerr := atoi("NAXIS") if aerr != nil { return nil, 0, aerr } var dims []int for j := 1; j <= naxis; j++ { d, derr := atoiN("NAXIS", j) if derr != nil { return nil, 0, derr } dims = append(dims, d) } pcount := 0 if v := get("PCOUNT"); v != "" { p, perr := atoi("PCOUNT") if perr != nil { return nil, 0, perr } pcount = p } gcount := 1 if v := get("GCOUNT"); v != "" { g, gerr := atoi("GCOUNT") if gerr != nil { return nil, 0, gerr } gcount = g } if !isTable { // An image HDU: skip its data block. The size is |BITPIX|/8 // bytes per element times GCOUNT*(PCOUNT + every NAXISn); positive // BITPIX (the integer depths) counts the same as negative, and // the axis extents multiply rather than add. Every factor is // bounded against the bytes the file actually holds before it // multiplies, the same discipline parseFITS applies, so a // hostile product cannot wrap into a skip to a chosen offset. width := 1 if bitpix < 0 { width = -bitpix / 8 } else { width = bitpix / 8 } if width <= 0 { return nil, 0, base.Errf("BITPIX = %d gives a non-positive element width", bitpix) } avail := max(int64(len(data)-dataAt), 0) total := 0 if naxis > 0 { prod := int64(1) for _, d := range dims { if d < 0 { return nil, 0, base.Errf("image HDU declares the negative axis %d", d) } // A zero axis empties the data block whatever follows: // the bound divides by the running product, so the // division runs only while the product is live. if prod > 0 && int64(d) > avail/int64(width)/prod { return nil, 0, base.Errf("image HDU declares more data than the %d bytes the file holds", len(data)-dataAt) } prod *= int64(d) } if int64(pcount) < 0 || int64(pcount) > avail { return nil, 0, base.Errf("image HDU declares a PCOUNT of %d against %d bytes", pcount, avail) } space := int64(pcount) + prod switch { case space == 0: // A zero-sized group repeats into nothing whatever // GCOUNT claims. total = 0 case int64(gcount) > 1 && int64(gcount) > avail/space: return nil, 0, base.Errf("image HDU declares more data than the %d bytes the file holds", len(data)-dataAt) default: total = int(int64(gcount) * space) } } if total < 0 { return nil, 0, base.Errf("image HDU declares a negative data size") } return nil, dataAt + fitsBlockSize(total*width), nil } ext := strings.Trim(get("XTENSION"), " ") if ext != "BINTABLE" && ext != "TABLE" { // An unknown extension: refuse rather than guess its size. return nil, 0, base.Errf("unsupported extension %q", ext) } if naxis != 2 || len(dims) != 2 { return nil, 0, base.Errf("%s extension must have NAXIS = 2, got %d", ext, naxis) } tfields, ferr := atoi("TFIELDS") if ferr != nil { return nil, 0, ferr } // A negative count would size the per-column slices below with a // negative length, which panics; it is a header that lies, so it is // refused by name. if tfields < 0 { return nil, 0, base.Errf("TFIELDS = %d is negative", tfields) } rows := dims[1] if rows < 1 { return nil, 0, base.Errf("table declares NAXIS2 = %d, want at least 1 row", rows) } // The count is bounded by what could back it, the way the NetCDF // reader bounds its own: every column of either table kind occupies // at least one byte of every row, so a TFIELDS past the data the // file holds is a claim nothing can follow, and refusing it here // keeps the per-column preallocations from being sized by the claim. if left := int64(len(data) - dataAt); int64(tfields) > left/int64(rows) { return nil, 0, base.Errf("TFIELDS = %d names more columns than the %d bytes of table data can hold", tfields, left) } table := &FITSTable{ Kind: ext, Rows: rows, Headers: map[string]string{}, } type parsed struct { name, unit, form string tbcol int } var cols []parsed for i := 1; i <= tfields; i++ { col := parsed{ name: fitsLookupN(first, "TTYPE", i), unit: fitsLookupN(first, "TUNIT", i), form: strings.ToUpper(strings.TrimSpace(fitsLookupN(first, "TFORM", i))), tbcol: 0, } if v := fitsLookupN(first, "TBCOL", i); v != "" { tb, tberr := strconv.Atoi(strings.TrimSpace(v)) if tberr != nil { return nil, 0, base.Errf("column %d: TBCOL %q is not an integer", i, v) } col.tbcol = tb } if col.form == "" { return nil, 0, base.Errf("column %d has no TFORM", i) } cols = append(cols, col) table.Names = append(table.Names, col.name) table.Units = append(table.Units, col.unit) } headers := table.Headers for _, c := range cards { switch c.key { case "XTENSION", "BITPIX", "NAXIS", "PCOUNT", "GCOUNT", "TFIELDS": default: if !fitsAxisKeyword(c.key) && !strings.HasPrefix(c.key, "TTYPE") && !strings.HasPrefix(c.key, "TFORM") && !strings.HasPrefix(c.key, "TUNIT") && !strings.HasPrefix(c.key, "TBCOL") { headers[c.key] = c.value } } } if ext == "BINTABLE" { // The data block starts on a 2880-byte block boundary, so a file // that ends inside the header has no data at all. The row // arithmetic below indexes from dataAt, which must be checked // against the file before it is used. if dataAt > len(data) { return nil, 0, base.Errf("table data starts at byte %d of a %d-byte file", dataAt, len(data)) } widths := make([]int, tfields) forms := make([]string, tfields) // Column offsets inside a row: prefix sums of the widths, // computed once instead of once per row. offs := make([]int, tfields) rowBytes := 0 // The widest row any present data can hold: every width is // bounded against it as it accumulates, so the prefix sum can // never wrap past the guard below into a small positive value. maxRow := (int64(len(data)) - int64(dataAt)) / int64(rows) for i, col := range cols { r, form, perr := parseTFORM(col.form) if perr != nil { return nil, 0, base.Errf("column %d: %w", i+1, perr) } if form == "A" { // A character column's repeat is the string width: // "16A" is one 16-character field, the layout every // catalogue uses. The numeric repeat rule below would // reject the library's own output. if r < 1 { return nil, 0, base.Errf("column %d: width %d is not positive", i+1, r) } forms[i] = "A" widths[i] = r } else { if r != 1 { return nil, 0, base.Errf("column %d: repeat %d is not supported, want a scalar", i+1, r) } widths[i] = tfSize(form) // A zero width is an unknown code, or a variable-length // descriptor ("P", "Q") the reader cannot size. Refusing // it here keeps the row arithmetic below meaningful: a // zero-width column would defeat the truncation guard and // push the reads past the end of the file. if widths[i] == 0 { return nil, 0, base.Errf("column %d: unsupported TFORM %q", i+1, form) } forms[i] = form } if int64(widths[i]) > maxRow-int64(rowBytes) { return nil, 0, base.Errf("column %d of width %d leaves the %d bytes of row the data holds", i+1, widths[i], maxRow) } offs[i] = rowBytes rowBytes += widths[i] } // Bound before multiplying: a hostile NAXIS2 must not overflow // the product (mirroring parseFITS's guard), it just means the // declared data cannot fit the file. if rowBytes > 0 && rows > (len(data)-dataAt)/rowBytes { return nil, 0, base.Errf("table data is truncated") } for i := range cols { form := forms[i] if form == "A" { table.Text = append(table.Text, fitsTextColumn(data, rows, dataAt+offs[i], rowBytes, widths[i])) table.Columns = append(table.Columns, nil) continue } var dt core.Dtype = core.Float switch form { case "B", "I", "J", "K", "L": dt = core.Int case "E": dt = core.Float32 } arr := core.New(dt, rows) // The form decides the loop once per column, and the cell goes // straight from the file's bytes into the array's payload: no // cell is boxed into an any and the row body holds no type // switch. colBase is the column's first byte, so a row's cell // starts rowBytes further on. colBase := dataAt + offs[i] switch form { case "D": dst := arr.RawFloats() for row := range rows { dst[row] = math.Float64frombits(binary.BigEndian.Uint64(data[colBase+rowBytes*row:])) } case "E": dst := arr.RawFloat32s() for row := range rows { dst[row] = math.Float32frombits(binary.BigEndian.Uint32(data[colBase+rowBytes*row:])) } case "K": dst := arr.RawInts() for row := range rows { dst[row] = int64(binary.BigEndian.Uint64(data[colBase+rowBytes*row:])) } case "J": dst := arr.RawInts() for row := range rows { dst[row] = int64(int32(binary.BigEndian.Uint32(data[colBase+rowBytes*row:]))) } case "I": dst := arr.RawInts() for row := range rows { dst[row] = int64(int16(binary.BigEndian.Uint16(data[colBase+rowBytes*row:]))) } case "B": dst := arr.RawInts() for row := range rows { dst[row] = int64(data[colBase+rowBytes*row]) } case "L": // A false logical leaves the zero the payload was // allocated with. dst := arr.RawInts() for row := range rows { if data[colBase+rowBytes*row] == 'T' { dst[row] = 1 } } } table.Columns = append(table.Columns, scaleColumn(table.Headers, i+1, arr)) table.Text = append(table.Text, nil) } return table, dataAt + fitsBlockSize(rows*rowBytes), nil } // ASCII table: NAXIS1-character text rows. widths := make([]int, tfields) starts := make([]int, tfields) rowBytes := dims[0] for i, col := range cols { w, perr := parseASCIITFORM(col.form) if perr != nil { return nil, 0, base.Errf("column %d: %w", i+1, perr) } widths[i] = w starts[i] = col.tbcol - 1 // The bound is written without the sum, which a hostile TBCOL // would wrap past: starts + width must leave the row. if starts[i] < 0 || w > rowBytes || starts[i] > rowBytes-w { return nil, 0, base.Errf("column %d: TBCOL %d with width %d leaves the row", i+1, col.tbcol, w) } } // Bound before multiplying, as in the binary branch above. if rowBytes > 0 && rows > (len(data)-dataAt)/rowBytes { return nil, 0, base.Errf("table data is truncated") } for i, col := range cols { if strings.HasSuffix(col.form, "A") { table.Text = append(table.Text, fitsTextColumn(data, rows, dataAt+starts[i], rowBytes, widths[i])) table.Columns = append(table.Columns, nil) continue } isInt := strings.HasPrefix(col.form, "I") var dt core.Dtype = core.Float if isInt { dt = core.Int } arr := core.New(dt, rows) // The cell is parsed where it lies: the trimmed field is a view of // the row's bytes, not a copy, and the strconv call sees exactly // the text the string form saw. colBase := dataAt + starts[i] for row := range rows { p := colBase + rowBytes*row field := bytes.TrimSpace(data[p : p+widths[i]]) if len(field) == 0 { continue } if isInt { v, cerr := strconv.ParseInt(fitsFieldString(field), 10, 64) if cerr != nil { return nil, 0, base.Errf("column %d row %d: %q is not an integer", i+1, row, string(field)) } arr.RawInts()[row] = v } else { v, ok := fitsASCIIFloat(fitsFieldString(field)) if !ok { return nil, 0, base.Errf("column %d row %d: %q is not a number", i+1, row, string(field)) } arr.RawFloats()[row] = v } } table.Columns = append(table.Columns, scaleColumn(table.Headers, i+1, arr)) table.Text = append(table.Text, nil) } return table, dataAt + fitsBlockSize(rows*rowBytes), nil } // parseTFORM splits a binary TFORM into its repeat count and type // code: "3E" gives (3, "E"), "D" gives (1, "D"), "16A" gives (16, "A"). func parseTFORM(form string) (int, string, error) { if form == "" { return 0, "", base.Errf("empty TFORM") } repeat, code := 1, form for i, r := range form { if r < '0' || r > '9' { code = form[i:] if i > 0 { v, aerr := strconv.Atoi(form[:i]) if aerr != nil { // An overflowing repeat count must not silently // fall back to 1: a hostile "99999999999999999999E" // would be read as a scalar instead of refused. return 0, "", base.Errf("TFORM %q: the repeat count does not fit an integer", form) } repeat = v } break } } return repeat, code, nil } // fitsTextColumn builds a character column's strings from the // fixed-width cells that start at colBase and repeat every rowStride // bytes, trailing spaces trimmed. The trimmed cell bytes are copied // into one slab and every string of the column aliases its own range // of that slab, so the column costs one allocation whatever its row // count. The caller's guards have already bounded colBase + rowStride* // (rows-1) + width against the file's bytes. // // SAFETY: the slab is fully written before the first string is formed // from it, the aliased ranges are disjoint, and nothing writes to the // slab afterwards, so each string keeps exactly the bytes the copy // left there for as long as it lives. func fitsTextColumn(data []byte, rows, colBase, rowStride, width int) []string { text := make([]string, rows) total := 0 for row := range rows { base := colBase + rowStride*row p := base + width for p > base && data[p-1] == ' ' { p-- } total += p - base } slab := make([]byte, total) at := 0 for row := range rows { base := colBase + rowStride*row p := base + width for p > base && data[p-1] == ' ' { p-- } n := p - base copy(slab[at:at+n], data[base:p]) if n == 0 { text[row] = "" } else { text[row] = unsafe.String(unsafe.SliceData(slab[at:at+n]), n) } at += n } return text } // scaleColumn applies the TSCALn/TZEROn affine map (physical = raw* // scale + zero) to a decoded numeric column. Integer scaling that // stays integral keeps the int dtype (the unsigned-integer convention // TZERO = 32768/2147483648 lands here); any fractional or overflowing // scaling promotes the column to float64, mirroring cfitsio. A // malformed value is treated as unscaled: the raw storage values are // the best available answer and an error here would lose the whole // table over one keyword. func scaleColumn(headers map[string]string, n int, arr *core.Array) *core.Array { ss := fitsLookupN(headers, "TSCAL", n) zs := fitsLookupN(headers, "TZERO", n) if ss == "" && zs == "" { return arr } scale, zero := 1.0, 0.0 if ss != "" { if v, err := strconv.ParseFloat(strings.TrimSpace(ss), 64); err == nil { scale = v } } if zs != "" { if v, err := strconv.ParseFloat(strings.TrimSpace(zs), 64); err == nil { zero = v } } if scale == 1 && zero == 0 { return arr } switch arr.Dtype() { case core.Float: raw := arr.RawFloats() for i := range raw { raw[i] = raw[i]*scale + zero } return arr case core.Float32: // Scaled values leave the float32 guarantee; keep the full // float64 result instead of rounding twice. raw32 := arr.RawFloat32s() out := core.New(core.Float, arr.Shape()...) raw := out.RawFloats() for i := range raw32 { raw[i] = float64(raw32[i])*scale + zero } return out default: rawInts := arr.RawInts() if scale == math.Trunc(scale) && zero == math.Trunc(zero) && math.Abs(scale) < 1e15 && math.Abs(zero) < 9e15 { out := core.New(core.Int, arr.Shape()...) dst := out.RawInts() integral := true for i, v := range rawInts { sv := float64(v)*scale + zero // The exact float64 bounds of the int64 range: MaxInt64 // is 9223372036854775807, which no float64 holds, so the // first value past it is 2^63, while MinInt64 is exact. // A conversion out of range is implementation-defined // (amd64 answers MinInt64), which is the silent corruption // this path exists to avoid. if sv >= 9223372036854775808.0 || sv < -9223372036854775808.0 { integral = false break } dst[i] = int64(sv) } if integral { return out } } out := core.New(core.Float, arr.Shape()...) raw := out.RawFloats() for i, v := range rawInts { raw[i] = float64(v)*scale + zero } return out } } // tfSize returns the byte width of one element of a binary column. func tfSize(code string) int { switch code { case "L", "B", "A": return 1 case "I": return 2 case "J", "E": return 4 case "K", "D": return 8 } return 0 } // parseASCIITFORM returns the character width of an ASCII-table // column: "10A" gives 10, "I7" gives 7, "F14.6" gives 14, "E12.4" // gives 12, "D20.12" gives 20. func parseASCIITFORM(form string) (int, error) { w, _, _, err := parseASCIISaveForm(strings.ToUpper(strings.TrimSpace(form))) return w, err } // fitsASCIIFloat parses one ASCII-table cell. Beyond the forms // strconv.ParseFloat accepts, the cells of a Fortran-written table // carry the Dw.d exponent ("1.5D+03", the D column type this reader // accepts) and the exponent-less form ("1.5+03"), where the sign after // the mantissa introduces the exponent; both mean what the same text // with an E means, and refusing them refused the whole table. func fitsASCIIFloat(field string) (float64, bool) { if v, err := strconv.ParseFloat(field, 64); err == nil { return v, true } norm := field switch i := strings.IndexAny(norm, "dD"); { case i >= 0: // A Fortran D exponent is the E exponent under another letter. norm = norm[:i] + "E" + norm[i+1:] case strings.IndexAny(norm, "eE") < 0: // No exponent at all: the exponent-less Fortran form puts the // exponent's sign directly behind the mantissa, "1.5+03". if j := strings.IndexAny(norm[1:], "+-"); j >= 0 { norm = norm[:j+1] + "E" + norm[j+1:] } } if norm == field { return 0, false } v, err := strconv.ParseFloat(norm, 64) if err != nil { return 0, false } return v, true } // fitsFieldString views one trimmed ASCII-table field as a string // without copying it, the form strconv needs and the byte slice // already holds. // // SAFETY: the view aliases the file's bytes for the length of one // strconv call, which reads the string and retains nothing of it. The // parse errors are formatted from a copy, so no view of the file // outlives the slice it points into. func fitsFieldString(b []byte) string { return unsafe.String(unsafe.SliceData(b), len(b)) } // fitsCheckString validates a human-supplied string field. func fitsCheckString(v, what string) error { if len(v) == 0 { return base.Errf("%s must not be empty", what) } if len(v) > 68 { return base.Errf("%s is %d characters, at most 68 fit a card", what, len(v)) } return nil }