feat: initial release
Assisted-by: GLM 5.3 Flash
This commit is contained in:
@@ -0,0 +1,432 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package linalg
|
||||
|
||||
import (
|
||||
"math"
|
||||
"slices"
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
||||
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/engine"
|
||||
)
|
||||
|
||||
// Sparse symmetric eigensolver. The dense `Eigen` answers a full
|
||||
// spectrum by tridiagonalising the whole matrix, which costs O(n³)
|
||||
// and materialises n² entries; both are wasteful when the matrix is
|
||||
// sparse and only a few extreme eigenpairs are wanted. SpEigen runs
|
||||
// the Lanczos process instead, whose every step costs one
|
||||
// sparse-matrix-vector product, then reads the Ritz pairs off the
|
||||
// small tridiagonal matrix it builds.
|
||||
//
|
||||
// Lanczos with the plain three-term recurrence loses orthogonality
|
||||
// between the basis vectors as it converges, which shows up as
|
||||
// duplicated ("ghost") eigenvalues. The classical cure is full
|
||||
// reorthogonalisation, applied here twice per step: one pass removes
|
||||
// the loss, the second removes what the first reintroduces at rounding
|
||||
// level. That costs O(m·n) per step but keeps the Ritz values honest,
|
||||
// and m stays small because only a few eigenpairs are wanted.
|
||||
|
||||
// spEigenSeed is the fixed seed used when SpEigen gets no generator,
|
||||
// so an unseeded call is reproducible across runs. It is arbitrary but
|
||||
// fixed; any non-zero seed avoids the all-ones start vector, which is
|
||||
// orthogonal to every zero-sum eigenvector and would silently miss
|
||||
// them.
|
||||
const spEigenSeed = 20260912
|
||||
|
||||
// spEigenBlock is how many Lanczos steps to run past the number of
|
||||
// eigenpairs asked for. The extra steps buy convergence of the wanted
|
||||
// extremes: Ritz values settle well before Ritz vectors do, so the
|
||||
// budget is set by the vectors. A matrix of n ≤ k+spEigenBlock is
|
||||
// decomposed fully, which makes the answer exact rather than an
|
||||
// approximation; the partial path is what large n takes.
|
||||
const spEigenBlock = 40
|
||||
|
||||
// sparseCSR is a compressed-sparse-row view of a square real matrix,
|
||||
// built once per solve so the repeated matrix-vector products of the
|
||||
// Lanczos loop stream through contiguous value slices.
|
||||
type sparseCSR struct {
|
||||
rowStart []int
|
||||
colIdx []int
|
||||
vals []float64
|
||||
n int
|
||||
}
|
||||
|
||||
// cooToCSR converts a core.SparseCOO into CSR form, summing duplicate
|
||||
// coordinates the way COO semantics require. name names the calling
|
||||
// entry point in validation errors.
|
||||
func cooToCSR(s *core.SparseCOO, name string) (*sparseCSR, error) {
|
||||
ndim := len(s.Shape)
|
||||
nnz := s.Indices.Shape()[0]
|
||||
type entry struct {
|
||||
row, col int
|
||||
val float64
|
||||
}
|
||||
entries := make([]entry, nnz)
|
||||
ints := s.Indices.RawInts()
|
||||
for i := range nnz {
|
||||
row, col := 0, 0
|
||||
off := i * ndim
|
||||
for d := range ndim {
|
||||
idx := int(ints[off+d])
|
||||
if idx < 0 || idx >= s.Shape[d] {
|
||||
return nil, base.Errf("%s: index [%d]=%d out of range for dim %d of size %d",
|
||||
name, i, idx, d, s.Shape[d])
|
||||
}
|
||||
if d == 0 {
|
||||
row = idx
|
||||
}
|
||||
if d == 1 {
|
||||
col = idx
|
||||
}
|
||||
}
|
||||
entries[i] = entry{row: row, col: col, val: s.Values.FloatAt(i)}
|
||||
}
|
||||
// Sort by (row, col) so duplicates merge and each row is contiguous.
|
||||
slices.SortFunc(entries, func(a, b entry) int {
|
||||
if a.row != b.row {
|
||||
return a.row - b.row
|
||||
}
|
||||
return a.col - b.col
|
||||
})
|
||||
c := &sparseCSR{
|
||||
rowStart: make([]int, s.Shape[0]+1),
|
||||
colIdx: make([]int, 0, nnz),
|
||||
vals: make([]float64, 0, nnz),
|
||||
n: s.Shape[0],
|
||||
}
|
||||
// The duplicates are adjacent after the sort, so each run merges into
|
||||
// one accumulated value on the way into the compressed form.
|
||||
for p := 0; p < len(entries); {
|
||||
e := entries[p]
|
||||
v := e.val
|
||||
p++
|
||||
for p < len(entries) && entries[p].row == e.row && entries[p].col == e.col {
|
||||
v += entries[p].val
|
||||
p++
|
||||
}
|
||||
if v == 0 {
|
||||
continue
|
||||
}
|
||||
c.colIdx = append(c.colIdx, e.col)
|
||||
c.vals = append(c.vals, v)
|
||||
c.rowStart[e.row+1]++
|
||||
}
|
||||
for i := range c.n {
|
||||
c.rowStart[i+1] += c.rowStart[i]
|
||||
}
|
||||
return c, nil
|
||||
}
|
||||
|
||||
// symmetricCSR converts a core.SparseCOO into CSR form and verifies the
|
||||
// matrix is symmetric. Asymmetry beyond a scale-relative 1e-12 is
|
||||
// refused, matching the tolerance the dense `Eigen` applies.
|
||||
func symmetricCSR(s *core.SparseCOO, name string) (*sparseCSR, error) {
|
||||
c, err := cooToCSR(s, name)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := c.checkSymmetric(name); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return c, nil
|
||||
}
|
||||
|
||||
// checkSymmetric verifies that every stored entry has a matching
|
||||
// transposed entry of the same value, the property the Lanczos and
|
||||
// conjugate-gradient recurrences rely on. The tolerance is purely
|
||||
// relative to the largest magnitude present, deliberately without an
|
||||
// absolute floor: a floor would approve a matrix of small scale whose
|
||||
// asymmetry is a large fraction of it, and the recurrence would then
|
||||
// answer for a matrix that is neither A nor Aᵀ, whatever the wording of
|
||||
// the guard claims. Relative-only keeps the approval and the arithmetic
|
||||
// in agreement. A non-finite entry is refused in its own right: NaN
|
||||
// compares unequal to everything, so the mirror test alone would wave
|
||||
// it through as symmetric. The wording of the refusal is unchanged.
|
||||
func (c *sparseCSR) checkSymmetric(name string) error {
|
||||
scale := 0.0
|
||||
for _, v := range c.vals {
|
||||
if a := math.Abs(v); a > scale {
|
||||
scale = a
|
||||
}
|
||||
}
|
||||
tol := 1e-12 * scale
|
||||
for i := range c.n {
|
||||
for p := c.rowStart[i]; p < c.rowStart[i+1]; p++ {
|
||||
j := c.colIdx[p]
|
||||
v := c.vals[p]
|
||||
if math.IsNaN(v) || math.IsInf(v, 0) {
|
||||
return base.Errf("%s: entry [%d,%d] is not finite", name, i, j)
|
||||
}
|
||||
mirror, ok := c.at(j, i)
|
||||
if !ok || math.Abs(mirror-v) > tol {
|
||||
return base.Errf("%s: matrix is not symmetric within 1e-12 tolerance", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// at reads the entry (row, col), reporting whether it is stored. The
|
||||
// row's column indices are sorted, so the lookup is a binary search:
|
||||
// the symmetry check calls it once per stored entry, and a linear scan
|
||||
// made that check quadratic in the row length.
|
||||
func (c *sparseCSR) at(row, col int) (float64, bool) {
|
||||
lo, hi := c.rowStart[row], c.rowStart[row+1]
|
||||
for lo < hi {
|
||||
mid := int(uint(lo+hi) >> 1)
|
||||
if c.colIdx[mid] < col {
|
||||
lo = mid + 1
|
||||
} else {
|
||||
hi = mid
|
||||
}
|
||||
}
|
||||
if lo < c.rowStart[row+1] && c.colIdx[lo] == col {
|
||||
return c.vals[lo], true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// matVec computes y = A·x. Output rows are independent, so the row
|
||||
// range splits across workers once a worker's share of the stored
|
||||
// entries pays for the split, and runs on the calling goroutine below
|
||||
// that.
|
||||
func (c *sparseCSR) matVec(x, y []float64) {
|
||||
// The three structure slices are taken once: every worker's rows walk
|
||||
// them per stored entry, and the row loop holds no other state.
|
||||
rowStart := c.rowStart
|
||||
colIdx := c.colIdx
|
||||
vals := c.vals
|
||||
if sparseMatVecSplit(c.n, len(vals)) {
|
||||
engine.ParallelMin(c.n, 1, func(s, e int) {
|
||||
csrMatVecRange(s, e, rowStart, colIdx, vals, x, y)
|
||||
})
|
||||
return
|
||||
}
|
||||
csrMatVecRange(0, c.n, rowStart, colIdx, vals, x, y)
|
||||
}
|
||||
|
||||
// SpEigen returns the k eigenvalues of largest magnitude of a real
|
||||
// symmetric sparse matrix, each with its unit eigenvector. Values are
|
||||
// ordered by descending magnitude and vectors holds the matching
|
||||
// eigenvectors as columns, so column j goes with values[j].
|
||||
//
|
||||
// The matrix must be square, real and symmetric; complex sparse
|
||||
// matrices are refused, as they are everywhere else on the sparse
|
||||
// surface. Requesting k = n recovers the complete spectrum, though at
|
||||
// that point the dense `Eigen` is the cheaper path.
|
||||
//
|
||||
// The starting vector is drawn from gen; passing nil uses a fixed
|
||||
// seed, so the result is reproducible. Because Lanczos converges to
|
||||
// the extremes of the spectrum first, the eigenvalues returned are
|
||||
// approximations whose accuracy improves with the iteration budget,
|
||||
// not the exact answers `Eigen` computes.
|
||||
func SpEigen(s *core.SparseCOO, k int, gen *core.Generator) (values, vectors *core.Array, err error) {
|
||||
if s.Values.Dtype() == core.Complex {
|
||||
return nil, nil, base.Errf("SpEigen: complex sparse matrices are not supported")
|
||||
}
|
||||
if len(s.Shape) != 2 || s.Shape[0] != s.Shape[1] {
|
||||
return nil, nil, base.Errf("SpEigen: needs a square 2-D sparse matrix, got shape %v", s.Shape)
|
||||
}
|
||||
n := s.Shape[0]
|
||||
if n == 0 {
|
||||
return nil, nil, base.Errf("SpEigen: zero-sized matrix, got shape %v", s.Shape)
|
||||
}
|
||||
if k < 1 || k > n {
|
||||
return nil, nil, base.Errf("SpEigen: k must be in [1, %d], got %d", n, k)
|
||||
}
|
||||
c, err := symmetricCSR(s, "SpEigen")
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
// The projected tridiagonal carries the matrix's magnitude, and the
|
||||
// symmetric sweep's deflation floor is taken against the sum of its
|
||||
// squared entries: above about 1.3e154 that sum overflows to +Inf, so
|
||||
// every subdiagonal reads as negligible, the sweep deflates the whole
|
||||
// block at once and the raw Rayleigh quotients come back as the
|
||||
// spectrum. The matrix is moved into the window for the recurrence and
|
||||
// the Ritz values, which carry the scale, are moved back before they
|
||||
// are returned; the basis vectors are unit-length and therefore
|
||||
// scale-free.
|
||||
ws := windowScale(maxMagF64(c.vals))
|
||||
if ws != 1 {
|
||||
scaleFloats(c.vals, ws)
|
||||
}
|
||||
if gen == nil {
|
||||
gen = core.NewGenerator(spEigenSeed)
|
||||
}
|
||||
alphas, betas, basis := c.lanczos(k, gen)
|
||||
r := len(alphas)
|
||||
|
||||
// The small tridiagonal matrix whose eigenpairs are the Ritz
|
||||
// approximations. Its eigenvectors accumulate into the identity,
|
||||
// so the columns of tVec are the tridiagonal eigenvectors.
|
||||
tMat := make([]float64, r*r)
|
||||
for i := range r {
|
||||
tMat[i*r+i] = alphas[i]
|
||||
if i+1 < r {
|
||||
tMat[i*r+i+1] = betas[i]
|
||||
tMat[(i+1)*r+i] = betas[i]
|
||||
}
|
||||
}
|
||||
tVec := eye(r)
|
||||
if err := symmetricQr(tMat, tVec, r); err != nil {
|
||||
return nil, nil, base.Errf("SpEigen: %w", err)
|
||||
}
|
||||
|
||||
// Rank the Ritz pairs by descending magnitude and keep k.
|
||||
idx := make([]int, r)
|
||||
for i := range r {
|
||||
idx[i] = i
|
||||
}
|
||||
slices.SortFunc(idx, func(a, b int) int {
|
||||
da, db := math.Abs(tMat[a*r+a]), math.Abs(tMat[b*r+b])
|
||||
if da != db {
|
||||
if da > db {
|
||||
return -1
|
||||
}
|
||||
return 1
|
||||
}
|
||||
return a - b
|
||||
})
|
||||
sel := idx[:k]
|
||||
|
||||
outVals := make([]float64, k)
|
||||
outVecs := make([]float64, n*k)
|
||||
tmp := make([]float64, n)
|
||||
for j, si := range sel {
|
||||
outVals[j] = tMat[si*r+si]
|
||||
// Ritz vector: lift the tridiagonal eigenvector through the
|
||||
// Lanczos basis, v = Q·y.
|
||||
clear(tmp)
|
||||
for p := range r {
|
||||
y := tVec[p*r+si]
|
||||
if y == 0 {
|
||||
continue
|
||||
}
|
||||
row := basis[p*n : (p+1)*n]
|
||||
for i := range n {
|
||||
tmp[i] += y * row[i]
|
||||
}
|
||||
}
|
||||
normaliseF64(tmp)
|
||||
for i := range n {
|
||||
outVecs[i*k+j] = tmp[i]
|
||||
}
|
||||
}
|
||||
// Only the Ritz values carry the matrix's scale; the vectors do not.
|
||||
if ws != 1 {
|
||||
unscaleFloats(outVals, ws)
|
||||
}
|
||||
return floatsToArray(outVals, []int{k}), floatsToArray(outVecs, []int{n, k}), nil
|
||||
}
|
||||
|
||||
// lanczos runs the process, returning the diagonal alpha, the
|
||||
// off-diagonal beta and the basis vectors flattened as n-wide rows.
|
||||
// When the recurrence collapses, meaning the Krylov space is
|
||||
// exhausted, the block is closed off with a zero coupling and a fresh
|
||||
// direction restarts it: the tridiagonal matrix becomes
|
||||
// block-diagonal, and each block still yields valid Ritz pairs.
|
||||
func (c *sparseCSR) lanczos(k int, gen *core.Generator) (alphas, betas []float64, basis []float64) {
|
||||
n := c.n
|
||||
steps := min(n, max(2*k, k+spEigenBlock))
|
||||
w := make([]float64, n)
|
||||
q := make([]float64, n)
|
||||
// The budget bounds the recurrence: one basis row and at most one
|
||||
// coefficient per step, so all three are sized once rather than grown.
|
||||
alphas = make([]float64, 0, steps)
|
||||
betas = make([]float64, 0, steps)
|
||||
basis = make([]float64, 0, steps*n)
|
||||
scale := 0.0
|
||||
addScale := func(v float64) {
|
||||
if a := math.Abs(v); a > scale {
|
||||
scale = a
|
||||
}
|
||||
}
|
||||
start := func() {
|
||||
for i := range n {
|
||||
q[i] = gen.NormalUnit()
|
||||
}
|
||||
// Orthogonalise against every direction already taken so a
|
||||
// restarted block cannot re-enter an earlier one.
|
||||
for p := range len(alphas) {
|
||||
row := basis[p*n : (p+1)*n]
|
||||
d := dotF64(q, row)
|
||||
for i := range n {
|
||||
q[i] -= d * row[i]
|
||||
}
|
||||
}
|
||||
normaliseF64(q)
|
||||
}
|
||||
start()
|
||||
for range steps {
|
||||
basis = append(basis, q...)
|
||||
c.matVec(q, w)
|
||||
alpha := dotF64(q, w)
|
||||
alphas = append(alphas, alpha)
|
||||
addScale(alpha)
|
||||
// Strip the two previous directions out of w, the three-term
|
||||
// recurrence itself.
|
||||
for i := range n {
|
||||
w[i] -= alpha * q[i]
|
||||
}
|
||||
if len(alphas) > 1 {
|
||||
beta := betas[len(betas)-1]
|
||||
prev := basis[(len(alphas)-2)*n : (len(alphas)-1)*n]
|
||||
for i := range n {
|
||||
w[i] -= beta * prev[i]
|
||||
}
|
||||
}
|
||||
// Full reorthogonalisation, twice: one pass removes the
|
||||
// accumulated loss, the second what the first reintroduces.
|
||||
for range 2 {
|
||||
for p := range len(alphas) {
|
||||
row := basis[p*n : (p+1)*n]
|
||||
d := dotF64(w, row)
|
||||
for i := range n {
|
||||
w[i] -= d * row[i]
|
||||
}
|
||||
}
|
||||
}
|
||||
beta := norm2F64(w)
|
||||
// The exhaustion threshold is purely relative to the running
|
||||
// spectral scale: an absolute floor would treat a legitimate
|
||||
// tiny-scale matrix as one big deflated block and hand back
|
||||
// random-start Rayleigh quotients instead of the spectrum.
|
||||
if beta <= float64(n)*base.EpsF*scale {
|
||||
// The Krylov space is exhausted. Close the block with a
|
||||
// zero coupling and restart in a fresh direction, unless
|
||||
// the basis already spans the whole space.
|
||||
if len(alphas) >= n {
|
||||
break
|
||||
}
|
||||
betas = append(betas, 0)
|
||||
start()
|
||||
continue
|
||||
}
|
||||
betas = append(betas, beta)
|
||||
addScale(beta)
|
||||
for i := range n {
|
||||
q[i] = w[i] / beta
|
||||
}
|
||||
}
|
||||
// The last beta couples to a vector that was never formed, so it
|
||||
// does not belong on the tridiagonal diagonal-band.
|
||||
if len(betas) >= len(alphas) {
|
||||
betas = betas[:len(alphas)-1]
|
||||
}
|
||||
return alphas, betas, basis
|
||||
}
|
||||
|
||||
// normaliseF64 scales a vector to unit length in place, leaving a
|
||||
// zero vector untouched rather than dividing by zero.
|
||||
func normaliseF64(a []float64) {
|
||||
n := norm2F64(a)
|
||||
if n == 0 {
|
||||
return
|
||||
}
|
||||
for i := range a {
|
||||
a[i] /= n
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user