Files
tensor/linalg/sparseigen.go
T

433 lines
14 KiB
Go
Raw Normal View History

2026-09-03 10:00:00 +02:00
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package linalg
import (
"math"
"slices"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
"sourcedock.dev/petrbalvin/tensor/internal/engine"
)
// Sparse symmetric eigensolver. The dense `Eigen` answers a full
// spectrum by tridiagonalising the whole matrix, which costs O(n³)
// and materialises n² entries; both are wasteful when the matrix is
// sparse and only a few extreme eigenpairs are wanted. SpEigen runs
// the Lanczos process instead, whose every step costs one
// sparse-matrix-vector product, then reads the Ritz pairs off the
// small tridiagonal matrix it builds.
//
// Lanczos with the plain three-term recurrence loses orthogonality
// between the basis vectors as it converges, which shows up as
// duplicated ("ghost") eigenvalues. The classical cure is full
// reorthogonalisation, applied here twice per step: one pass removes
// the loss, the second removes what the first reintroduces at rounding
// level. That costs O(m·n) per step but keeps the Ritz values honest,
// and m stays small because only a few eigenpairs are wanted.
// spEigenSeed is the fixed seed used when SpEigen gets no generator,
// so an unseeded call is reproducible across runs. It is arbitrary but
// fixed; any non-zero seed avoids the all-ones start vector, which is
// orthogonal to every zero-sum eigenvector and would silently miss
// them.
const spEigenSeed = 20260912
// spEigenBlock is how many Lanczos steps to run past the number of
// eigenpairs asked for. The extra steps buy convergence of the wanted
// extremes: Ritz values settle well before Ritz vectors do, so the
// budget is set by the vectors. A matrix of n ≤ k+spEigenBlock is
// decomposed fully, which makes the answer exact rather than an
// approximation; the partial path is what large n takes.
const spEigenBlock = 40
// sparseCSR is a compressed-sparse-row view of a square real matrix,
// built once per solve so the repeated matrix-vector products of the
// Lanczos loop stream through contiguous value slices.
type sparseCSR struct {
rowStart []int
colIdx []int
vals []float64
n int
}
// cooToCSR converts a core.SparseCOO into CSR form, summing duplicate
// coordinates the way COO semantics require. name names the calling
// entry point in validation errors.
func cooToCSR(s *core.SparseCOO, name string) (*sparseCSR, error) {
ndim := len(s.Shape)
nnz := s.Indices.Shape()[0]
type entry struct {
row, col int
val float64
}
entries := make([]entry, nnz)
ints := s.Indices.RawInts()
for i := range nnz {
row, col := 0, 0
off := i * ndim
for d := range ndim {
idx := int(ints[off+d])
if idx < 0 || idx >= s.Shape[d] {
return nil, base.Errf("%s: index [%d]=%d out of range for dim %d of size %d",
name, i, idx, d, s.Shape[d])
}
if d == 0 {
row = idx
}
if d == 1 {
col = idx
}
}
entries[i] = entry{row: row, col: col, val: s.Values.FloatAt(i)}
}
// Sort by (row, col) so duplicates merge and each row is contiguous.
slices.SortFunc(entries, func(a, b entry) int {
if a.row != b.row {
return a.row - b.row
}
return a.col - b.col
})
c := &sparseCSR{
rowStart: make([]int, s.Shape[0]+1),
colIdx: make([]int, 0, nnz),
vals: make([]float64, 0, nnz),
n: s.Shape[0],
}
// The duplicates are adjacent after the sort, so each run merges into
// one accumulated value on the way into the compressed form.
for p := 0; p < len(entries); {
e := entries[p]
v := e.val
p++
for p < len(entries) && entries[p].row == e.row && entries[p].col == e.col {
v += entries[p].val
p++
}
if v == 0 {
continue
}
c.colIdx = append(c.colIdx, e.col)
c.vals = append(c.vals, v)
c.rowStart[e.row+1]++
}
for i := range c.n {
c.rowStart[i+1] += c.rowStart[i]
}
return c, nil
}
// symmetricCSR converts a core.SparseCOO into CSR form and verifies the
// matrix is symmetric. Asymmetry beyond a scale-relative 1e-12 is
// refused, matching the tolerance the dense `Eigen` applies.
func symmetricCSR(s *core.SparseCOO, name string) (*sparseCSR, error) {
c, err := cooToCSR(s, name)
if err != nil {
return nil, err
}
if err := c.checkSymmetric(name); err != nil {
return nil, err
}
return c, nil
}
// checkSymmetric verifies that every stored entry has a matching
// transposed entry of the same value, the property the Lanczos and
// conjugate-gradient recurrences rely on. The tolerance is purely
// relative to the largest magnitude present, deliberately without an
// absolute floor: a floor would approve a matrix of small scale whose
// asymmetry is a large fraction of it, and the recurrence would then
// answer for a matrix that is neither A nor Aᵀ, whatever the wording of
// the guard claims. Relative-only keeps the approval and the arithmetic
// in agreement. A non-finite entry is refused in its own right: NaN
// compares unequal to everything, so the mirror test alone would wave
// it through as symmetric. The wording of the refusal is unchanged.
func (c *sparseCSR) checkSymmetric(name string) error {
scale := 0.0
for _, v := range c.vals {
if a := math.Abs(v); a > scale {
scale = a
}
}
tol := 1e-12 * scale
for i := range c.n {
for p := c.rowStart[i]; p < c.rowStart[i+1]; p++ {
j := c.colIdx[p]
v := c.vals[p]
if math.IsNaN(v) || math.IsInf(v, 0) {
return base.Errf("%s: entry [%d,%d] is not finite", name, i, j)
}
mirror, ok := c.at(j, i)
if !ok || math.Abs(mirror-v) > tol {
return base.Errf("%s: matrix is not symmetric within 1e-12 tolerance", name)
}
}
}
return nil
}
// at reads the entry (row, col), reporting whether it is stored. The
// row's column indices are sorted, so the lookup is a binary search:
// the symmetry check calls it once per stored entry, and a linear scan
// made that check quadratic in the row length.
func (c *sparseCSR) at(row, col int) (float64, bool) {
lo, hi := c.rowStart[row], c.rowStart[row+1]
for lo < hi {
mid := int(uint(lo+hi) >> 1)
if c.colIdx[mid] < col {
lo = mid + 1
} else {
hi = mid
}
}
if lo < c.rowStart[row+1] && c.colIdx[lo] == col {
return c.vals[lo], true
}
return 0, false
}
// matVec computes y = A·x. Output rows are independent, so the row
// range splits across workers once a worker's share of the stored
// entries pays for the split, and runs on the calling goroutine below
// that.
func (c *sparseCSR) matVec(x, y []float64) {
// The three structure slices are taken once: every worker's rows walk
// them per stored entry, and the row loop holds no other state.
rowStart := c.rowStart
colIdx := c.colIdx
vals := c.vals
if sparseMatVecSplit(c.n, len(vals)) {
engine.ParallelMin(c.n, 1, func(s, e int) {
csrMatVecRange(s, e, rowStart, colIdx, vals, x, y)
})
return
}
csrMatVecRange(0, c.n, rowStart, colIdx, vals, x, y)
}
// SpEigen returns the k eigenvalues of largest magnitude of a real
// symmetric sparse matrix, each with its unit eigenvector. Values are
// ordered by descending magnitude and vectors holds the matching
// eigenvectors as columns, so column j goes with values[j].
//
// The matrix must be square, real and symmetric; complex sparse
// matrices are refused, as they are everywhere else on the sparse
// surface. Requesting k = n recovers the complete spectrum, though at
// that point the dense `Eigen` is the cheaper path.
//
// The starting vector is drawn from gen; passing nil uses a fixed
// seed, so the result is reproducible. Because Lanczos converges to
// the extremes of the spectrum first, the eigenvalues returned are
// approximations whose accuracy improves with the iteration budget,
// not the exact answers `Eigen` computes.
func SpEigen(s *core.SparseCOO, k int, gen *core.Generator) (values, vectors *core.Array, err error) {
if s.Values.Dtype() == core.Complex {
return nil, nil, base.Errf("SpEigen: complex sparse matrices are not supported")
}
if len(s.Shape) != 2 || s.Shape[0] != s.Shape[1] {
return nil, nil, base.Errf("SpEigen: needs a square 2-D sparse matrix, got shape %v", s.Shape)
}
n := s.Shape[0]
if n == 0 {
return nil, nil, base.Errf("SpEigen: zero-sized matrix, got shape %v", s.Shape)
}
if k < 1 || k > n {
return nil, nil, base.Errf("SpEigen: k must be in [1, %d], got %d", n, k)
}
c, err := symmetricCSR(s, "SpEigen")
if err != nil {
return nil, nil, err
}
// The projected tridiagonal carries the matrix's magnitude, and the
// symmetric sweep's deflation floor is taken against the sum of its
// squared entries: above about 1.3e154 that sum overflows to +Inf, so
// every subdiagonal reads as negligible, the sweep deflates the whole
// block at once and the raw Rayleigh quotients come back as the
// spectrum. The matrix is moved into the window for the recurrence and
// the Ritz values, which carry the scale, are moved back before they
// are returned; the basis vectors are unit-length and therefore
// scale-free.
ws := windowScale(maxMagF64(c.vals))
if ws != 1 {
scaleFloats(c.vals, ws)
}
if gen == nil {
gen = core.NewGenerator(spEigenSeed)
}
alphas, betas, basis := c.lanczos(k, gen)
r := len(alphas)
// The small tridiagonal matrix whose eigenpairs are the Ritz
// approximations. Its eigenvectors accumulate into the identity,
// so the columns of tVec are the tridiagonal eigenvectors.
tMat := make([]float64, r*r)
for i := range r {
tMat[i*r+i] = alphas[i]
if i+1 < r {
tMat[i*r+i+1] = betas[i]
tMat[(i+1)*r+i] = betas[i]
}
}
tVec := eye(r)
if err := symmetricQr(tMat, tVec, r); err != nil {
return nil, nil, base.Errf("SpEigen: %w", err)
}
// Rank the Ritz pairs by descending magnitude and keep k.
idx := make([]int, r)
for i := range r {
idx[i] = i
}
slices.SortFunc(idx, func(a, b int) int {
da, db := math.Abs(tMat[a*r+a]), math.Abs(tMat[b*r+b])
if da != db {
if da > db {
return -1
}
return 1
}
return a - b
})
sel := idx[:k]
outVals := make([]float64, k)
outVecs := make([]float64, n*k)
tmp := make([]float64, n)
for j, si := range sel {
outVals[j] = tMat[si*r+si]
// Ritz vector: lift the tridiagonal eigenvector through the
// Lanczos basis, v = Q·y.
clear(tmp)
for p := range r {
y := tVec[p*r+si]
if y == 0 {
continue
}
row := basis[p*n : (p+1)*n]
for i := range n {
tmp[i] += y * row[i]
}
}
normaliseF64(tmp)
for i := range n {
outVecs[i*k+j] = tmp[i]
}
}
// Only the Ritz values carry the matrix's scale; the vectors do not.
if ws != 1 {
unscaleFloats(outVals, ws)
}
return floatsToArray(outVals, []int{k}), floatsToArray(outVecs, []int{n, k}), nil
}
// lanczos runs the process, returning the diagonal alpha, the
// off-diagonal beta and the basis vectors flattened as n-wide rows.
// When the recurrence collapses, meaning the Krylov space is
// exhausted, the block is closed off with a zero coupling and a fresh
// direction restarts it: the tridiagonal matrix becomes
// block-diagonal, and each block still yields valid Ritz pairs.
func (c *sparseCSR) lanczos(k int, gen *core.Generator) (alphas, betas []float64, basis []float64) {
n := c.n
steps := min(n, max(2*k, k+spEigenBlock))
w := make([]float64, n)
q := make([]float64, n)
// The budget bounds the recurrence: one basis row and at most one
// coefficient per step, so all three are sized once rather than grown.
alphas = make([]float64, 0, steps)
betas = make([]float64, 0, steps)
basis = make([]float64, 0, steps*n)
scale := 0.0
addScale := func(v float64) {
if a := math.Abs(v); a > scale {
scale = a
}
}
start := func() {
for i := range n {
q[i] = gen.NormalUnit()
}
// Orthogonalise against every direction already taken so a
// restarted block cannot re-enter an earlier one.
for p := range len(alphas) {
row := basis[p*n : (p+1)*n]
d := dotF64(q, row)
for i := range n {
q[i] -= d * row[i]
}
}
normaliseF64(q)
}
start()
for range steps {
basis = append(basis, q...)
c.matVec(q, w)
alpha := dotF64(q, w)
alphas = append(alphas, alpha)
addScale(alpha)
// Strip the two previous directions out of w, the three-term
// recurrence itself.
for i := range n {
w[i] -= alpha * q[i]
}
if len(alphas) > 1 {
beta := betas[len(betas)-1]
prev := basis[(len(alphas)-2)*n : (len(alphas)-1)*n]
for i := range n {
w[i] -= beta * prev[i]
}
}
// Full reorthogonalisation, twice: one pass removes the
// accumulated loss, the second what the first reintroduces.
for range 2 {
for p := range len(alphas) {
row := basis[p*n : (p+1)*n]
d := dotF64(w, row)
for i := range n {
w[i] -= d * row[i]
}
}
}
beta := norm2F64(w)
// The exhaustion threshold is purely relative to the running
// spectral scale: an absolute floor would treat a legitimate
// tiny-scale matrix as one big deflated block and hand back
// random-start Rayleigh quotients instead of the spectrum.
if beta <= float64(n)*base.EpsF*scale {
// The Krylov space is exhausted. Close the block with a
// zero coupling and restart in a fresh direction, unless
// the basis already spans the whole space.
if len(alphas) >= n {
break
}
betas = append(betas, 0)
start()
continue
}
betas = append(betas, beta)
addScale(beta)
for i := range n {
q[i] = w[i] / beta
}
}
// The last beta couples to a vector that was never formed, so it
// does not belong on the tridiagonal diagonal-band.
if len(betas) >= len(alphas) {
betas = betas[:len(alphas)-1]
}
return alphas, betas, basis
}
// normaliseF64 scales a vector to unit length in place, leaving a
// zero vector untouched rather than dividing by zero.
func normaliseF64(a []float64) {
n := norm2F64(a)
if n == 0 {
return
}
for i := range a {
a[i] /= n
}
}