433 lines
14 KiB
Go
433 lines
14 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|||
|
|
// SPDX-License-Identifier: MIT
|
||
|
|
|
||
|
|
package linalg
|
||
|
|
|
||
|
|
import (
|
||
|
|
"math"
|
||
|
|
"slices"
|
||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
||
|
|
|
||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/engine"
|
||
|
|
)
|
||
|
|
|
||
|
|
// Sparse symmetric eigensolver. The dense `Eigen` answers a full
|
||
|
|
// spectrum by tridiagonalising the whole matrix, which costs O(n³)
|
||
|
|
// and materialises n² entries; both are wasteful when the matrix is
|
||
|
|
// sparse and only a few extreme eigenpairs are wanted. SpEigen runs
|
||
|
|
// the Lanczos process instead, whose every step costs one
|
||
|
|
// sparse-matrix-vector product, then reads the Ritz pairs off the
|
||
|
|
// small tridiagonal matrix it builds.
|
||
|
|
//
|
||
|
|
// Lanczos with the plain three-term recurrence loses orthogonality
|
||
|
|
// between the basis vectors as it converges, which shows up as
|
||
|
|
// duplicated ("ghost") eigenvalues. The classical cure is full
|
||
|
|
// reorthogonalisation, applied here twice per step: one pass removes
|
||
|
|
// the loss, the second removes what the first reintroduces at rounding
|
||
|
|
// level. That costs O(m·n) per step but keeps the Ritz values honest,
|
||
|
|
// and m stays small because only a few eigenpairs are wanted.
|
||
|
|
|
||
|
|
// spEigenSeed is the fixed seed used when SpEigen gets no generator,
|
||
|
|
// so an unseeded call is reproducible across runs. It is arbitrary but
|
||
|
|
// fixed; any non-zero seed avoids the all-ones start vector, which is
|
||
|
|
// orthogonal to every zero-sum eigenvector and would silently miss
|
||
|
|
// them.
|
||
|
|
const spEigenSeed = 20260912
|
||
|
|
|
||
|
|
// spEigenBlock is how many Lanczos steps to run past the number of
|
||
|
|
// eigenpairs asked for. The extra steps buy convergence of the wanted
|
||
|
|
// extremes: Ritz values settle well before Ritz vectors do, so the
|
||
|
|
// budget is set by the vectors. A matrix of n ≤ k+spEigenBlock is
|
||
|
|
// decomposed fully, which makes the answer exact rather than an
|
||
|
|
// approximation; the partial path is what large n takes.
|
||
|
|
const spEigenBlock = 40
|
||
|
|
|
||
|
|
// sparseCSR is a compressed-sparse-row view of a square real matrix,
|
||
|
|
// built once per solve so the repeated matrix-vector products of the
|
||
|
|
// Lanczos loop stream through contiguous value slices.
|
||
|
|
type sparseCSR struct {
|
||
|
|
rowStart []int
|
||
|
|
colIdx []int
|
||
|
|
vals []float64
|
||
|
|
n int
|
||
|
|
}
|
||
|
|
|
||
|
|
// cooToCSR converts a core.SparseCOO into CSR form, summing duplicate
|
||
|
|
// coordinates the way COO semantics require. name names the calling
|
||
|
|
// entry point in validation errors.
|
||
|
|
func cooToCSR(s *core.SparseCOO, name string) (*sparseCSR, error) {
|
||
|
|
ndim := len(s.Shape)
|
||
|
|
nnz := s.Indices.Shape()[0]
|
||
|
|
type entry struct {
|
||
|
|
row, col int
|
||
|
|
val float64
|
||
|
|
}
|
||
|
|
entries := make([]entry, nnz)
|
||
|
|
ints := s.Indices.RawInts()
|
||
|
|
for i := range nnz {
|
||
|
|
row, col := 0, 0
|
||
|
|
off := i * ndim
|
||
|
|
for d := range ndim {
|
||
|
|
idx := int(ints[off+d])
|
||
|
|
if idx < 0 || idx >= s.Shape[d] {
|
||
|
|
return nil, base.Errf("%s: index [%d]=%d out of range for dim %d of size %d",
|
||
|
|
name, i, idx, d, s.Shape[d])
|
||
|
|
}
|
||
|
|
if d == 0 {
|
||
|
|
row = idx
|
||
|
|
}
|
||
|
|
if d == 1 {
|
||
|
|
col = idx
|
||
|
|
}
|
||
|
|
}
|
||
|
|
entries[i] = entry{row: row, col: col, val: s.Values.FloatAt(i)}
|
||
|
|
}
|
||
|
|
// Sort by (row, col) so duplicates merge and each row is contiguous.
|
||
|
|
slices.SortFunc(entries, func(a, b entry) int {
|
||
|
|
if a.row != b.row {
|
||
|
|
return a.row - b.row
|
||
|
|
}
|
||
|
|
return a.col - b.col
|
||
|
|
})
|
||
|
|
c := &sparseCSR{
|
||
|
|
rowStart: make([]int, s.Shape[0]+1),
|
||
|
|
colIdx: make([]int, 0, nnz),
|
||
|
|
vals: make([]float64, 0, nnz),
|
||
|
|
n: s.Shape[0],
|
||
|
|
}
|
||
|
|
// The duplicates are adjacent after the sort, so each run merges into
|
||
|
|
// one accumulated value on the way into the compressed form.
|
||
|
|
for p := 0; p < len(entries); {
|
||
|
|
e := entries[p]
|
||
|
|
v := e.val
|
||
|
|
p++
|
||
|
|
for p < len(entries) && entries[p].row == e.row && entries[p].col == e.col {
|
||
|
|
v += entries[p].val
|
||
|
|
p++
|
||
|
|
}
|
||
|
|
if v == 0 {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
c.colIdx = append(c.colIdx, e.col)
|
||
|
|
c.vals = append(c.vals, v)
|
||
|
|
c.rowStart[e.row+1]++
|
||
|
|
}
|
||
|
|
for i := range c.n {
|
||
|
|
c.rowStart[i+1] += c.rowStart[i]
|
||
|
|
}
|
||
|
|
return c, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// symmetricCSR converts a core.SparseCOO into CSR form and verifies the
|
||
|
|
// matrix is symmetric. Asymmetry beyond a scale-relative 1e-12 is
|
||
|
|
// refused, matching the tolerance the dense `Eigen` applies.
|
||
|
|
func symmetricCSR(s *core.SparseCOO, name string) (*sparseCSR, error) {
|
||
|
|
c, err := cooToCSR(s, name)
|
||
|
|
if err != nil {
|
||
|
|
return nil, err
|
||
|
|
}
|
||
|
|
if err := c.checkSymmetric(name); err != nil {
|
||
|
|
return nil, err
|
||
|
|
}
|
||
|
|
return c, nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// checkSymmetric verifies that every stored entry has a matching
|
||
|
|
// transposed entry of the same value, the property the Lanczos and
|
||
|
|
// conjugate-gradient recurrences rely on. The tolerance is purely
|
||
|
|
// relative to the largest magnitude present, deliberately without an
|
||
|
|
// absolute floor: a floor would approve a matrix of small scale whose
|
||
|
|
// asymmetry is a large fraction of it, and the recurrence would then
|
||
|
|
// answer for a matrix that is neither A nor Aᵀ, whatever the wording of
|
||
|
|
// the guard claims. Relative-only keeps the approval and the arithmetic
|
||
|
|
// in agreement. A non-finite entry is refused in its own right: NaN
|
||
|
|
// compares unequal to everything, so the mirror test alone would wave
|
||
|
|
// it through as symmetric. The wording of the refusal is unchanged.
|
||
|
|
func (c *sparseCSR) checkSymmetric(name string) error {
|
||
|
|
scale := 0.0
|
||
|
|
for _, v := range c.vals {
|
||
|
|
if a := math.Abs(v); a > scale {
|
||
|
|
scale = a
|
||
|
|
}
|
||
|
|
}
|
||
|
|
tol := 1e-12 * scale
|
||
|
|
for i := range c.n {
|
||
|
|
for p := c.rowStart[i]; p < c.rowStart[i+1]; p++ {
|
||
|
|
j := c.colIdx[p]
|
||
|
|
v := c.vals[p]
|
||
|
|
if math.IsNaN(v) || math.IsInf(v, 0) {
|
||
|
|
return base.Errf("%s: entry [%d,%d] is not finite", name, i, j)
|
||
|
|
}
|
||
|
|
mirror, ok := c.at(j, i)
|
||
|
|
if !ok || math.Abs(mirror-v) > tol {
|
||
|
|
return base.Errf("%s: matrix is not symmetric within 1e-12 tolerance", name)
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// at reads the entry (row, col), reporting whether it is stored. The
|
||
|
|
// row's column indices are sorted, so the lookup is a binary search:
|
||
|
|
// the symmetry check calls it once per stored entry, and a linear scan
|
||
|
|
// made that check quadratic in the row length.
|
||
|
|
func (c *sparseCSR) at(row, col int) (float64, bool) {
|
||
|
|
lo, hi := c.rowStart[row], c.rowStart[row+1]
|
||
|
|
for lo < hi {
|
||
|
|
mid := int(uint(lo+hi) >> 1)
|
||
|
|
if c.colIdx[mid] < col {
|
||
|
|
lo = mid + 1
|
||
|
|
} else {
|
||
|
|
hi = mid
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if lo < c.rowStart[row+1] && c.colIdx[lo] == col {
|
||
|
|
return c.vals[lo], true
|
||
|
|
}
|
||
|
|
return 0, false
|
||
|
|
}
|
||
|
|
|
||
|
|
// matVec computes y = A·x. Output rows are independent, so the row
|
||
|
|
// range splits across workers once a worker's share of the stored
|
||
|
|
// entries pays for the split, and runs on the calling goroutine below
|
||
|
|
// that.
|
||
|
|
func (c *sparseCSR) matVec(x, y []float64) {
|
||
|
|
// The three structure slices are taken once: every worker's rows walk
|
||
|
|
// them per stored entry, and the row loop holds no other state.
|
||
|
|
rowStart := c.rowStart
|
||
|
|
colIdx := c.colIdx
|
||
|
|
vals := c.vals
|
||
|
|
if sparseMatVecSplit(c.n, len(vals)) {
|
||
|
|
engine.ParallelMin(c.n, 1, func(s, e int) {
|
||
|
|
csrMatVecRange(s, e, rowStart, colIdx, vals, x, y)
|
||
|
|
})
|
||
|
|
return
|
||
|
|
}
|
||
|
|
csrMatVecRange(0, c.n, rowStart, colIdx, vals, x, y)
|
||
|
|
}
|
||
|
|
|
||
|
|
// SpEigen returns the k eigenvalues of largest magnitude of a real
|
||
|
|
// symmetric sparse matrix, each with its unit eigenvector. Values are
|
||
|
|
// ordered by descending magnitude and vectors holds the matching
|
||
|
|
// eigenvectors as columns, so column j goes with values[j].
|
||
|
|
//
|
||
|
|
// The matrix must be square, real and symmetric; complex sparse
|
||
|
|
// matrices are refused, as they are everywhere else on the sparse
|
||
|
|
// surface. Requesting k = n recovers the complete spectrum, though at
|
||
|
|
// that point the dense `Eigen` is the cheaper path.
|
||
|
|
//
|
||
|
|
// The starting vector is drawn from gen; passing nil uses a fixed
|
||
|
|
// seed, so the result is reproducible. Because Lanczos converges to
|
||
|
|
// the extremes of the spectrum first, the eigenvalues returned are
|
||
|
|
// approximations whose accuracy improves with the iteration budget,
|
||
|
|
// not the exact answers `Eigen` computes.
|
||
|
|
func SpEigen(s *core.SparseCOO, k int, gen *core.Generator) (values, vectors *core.Array, err error) {
|
||
|
|
if s.Values.Dtype() == core.Complex {
|
||
|
|
return nil, nil, base.Errf("SpEigen: complex sparse matrices are not supported")
|
||
|
|
}
|
||
|
|
if len(s.Shape) != 2 || s.Shape[0] != s.Shape[1] {
|
||
|
|
return nil, nil, base.Errf("SpEigen: needs a square 2-D sparse matrix, got shape %v", s.Shape)
|
||
|
|
}
|
||
|
|
n := s.Shape[0]
|
||
|
|
if n == 0 {
|
||
|
|
return nil, nil, base.Errf("SpEigen: zero-sized matrix, got shape %v", s.Shape)
|
||
|
|
}
|
||
|
|
if k < 1 || k > n {
|
||
|
|
return nil, nil, base.Errf("SpEigen: k must be in [1, %d], got %d", n, k)
|
||
|
|
}
|
||
|
|
c, err := symmetricCSR(s, "SpEigen")
|
||
|
|
if err != nil {
|
||
|
|
return nil, nil, err
|
||
|
|
}
|
||
|
|
// The projected tridiagonal carries the matrix's magnitude, and the
|
||
|
|
// symmetric sweep's deflation floor is taken against the sum of its
|
||
|
|
// squared entries: above about 1.3e154 that sum overflows to +Inf, so
|
||
|
|
// every subdiagonal reads as negligible, the sweep deflates the whole
|
||
|
|
// block at once and the raw Rayleigh quotients come back as the
|
||
|
|
// spectrum. The matrix is moved into the window for the recurrence and
|
||
|
|
// the Ritz values, which carry the scale, are moved back before they
|
||
|
|
// are returned; the basis vectors are unit-length and therefore
|
||
|
|
// scale-free.
|
||
|
|
ws := windowScale(maxMagF64(c.vals))
|
||
|
|
if ws != 1 {
|
||
|
|
scaleFloats(c.vals, ws)
|
||
|
|
}
|
||
|
|
if gen == nil {
|
||
|
|
gen = core.NewGenerator(spEigenSeed)
|
||
|
|
}
|
||
|
|
alphas, betas, basis := c.lanczos(k, gen)
|
||
|
|
r := len(alphas)
|
||
|
|
|
||
|
|
// The small tridiagonal matrix whose eigenpairs are the Ritz
|
||
|
|
// approximations. Its eigenvectors accumulate into the identity,
|
||
|
|
// so the columns of tVec are the tridiagonal eigenvectors.
|
||
|
|
tMat := make([]float64, r*r)
|
||
|
|
for i := range r {
|
||
|
|
tMat[i*r+i] = alphas[i]
|
||
|
|
if i+1 < r {
|
||
|
|
tMat[i*r+i+1] = betas[i]
|
||
|
|
tMat[(i+1)*r+i] = betas[i]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
tVec := eye(r)
|
||
|
|
if err := symmetricQr(tMat, tVec, r); err != nil {
|
||
|
|
return nil, nil, base.Errf("SpEigen: %w", err)
|
||
|
|
}
|
||
|
|
|
||
|
|
// Rank the Ritz pairs by descending magnitude and keep k.
|
||
|
|
idx := make([]int, r)
|
||
|
|
for i := range r {
|
||
|
|
idx[i] = i
|
||
|
|
}
|
||
|
|
slices.SortFunc(idx, func(a, b int) int {
|
||
|
|
da, db := math.Abs(tMat[a*r+a]), math.Abs(tMat[b*r+b])
|
||
|
|
if da != db {
|
||
|
|
if da > db {
|
||
|
|
return -1
|
||
|
|
}
|
||
|
|
return 1
|
||
|
|
}
|
||
|
|
return a - b
|
||
|
|
})
|
||
|
|
sel := idx[:k]
|
||
|
|
|
||
|
|
outVals := make([]float64, k)
|
||
|
|
outVecs := make([]float64, n*k)
|
||
|
|
tmp := make([]float64, n)
|
||
|
|
for j, si := range sel {
|
||
|
|
outVals[j] = tMat[si*r+si]
|
||
|
|
// Ritz vector: lift the tridiagonal eigenvector through the
|
||
|
|
// Lanczos basis, v = Q·y.
|
||
|
|
clear(tmp)
|
||
|
|
for p := range r {
|
||
|
|
y := tVec[p*r+si]
|
||
|
|
if y == 0 {
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
row := basis[p*n : (p+1)*n]
|
||
|
|
for i := range n {
|
||
|
|
tmp[i] += y * row[i]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
normaliseF64(tmp)
|
||
|
|
for i := range n {
|
||
|
|
outVecs[i*k+j] = tmp[i]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Only the Ritz values carry the matrix's scale; the vectors do not.
|
||
|
|
if ws != 1 {
|
||
|
|
unscaleFloats(outVals, ws)
|
||
|
|
}
|
||
|
|
return floatsToArray(outVals, []int{k}), floatsToArray(outVecs, []int{n, k}), nil
|
||
|
|
}
|
||
|
|
|
||
|
|
// lanczos runs the process, returning the diagonal alpha, the
|
||
|
|
// off-diagonal beta and the basis vectors flattened as n-wide rows.
|
||
|
|
// When the recurrence collapses, meaning the Krylov space is
|
||
|
|
// exhausted, the block is closed off with a zero coupling and a fresh
|
||
|
|
// direction restarts it: the tridiagonal matrix becomes
|
||
|
|
// block-diagonal, and each block still yields valid Ritz pairs.
|
||
|
|
func (c *sparseCSR) lanczos(k int, gen *core.Generator) (alphas, betas []float64, basis []float64) {
|
||
|
|
n := c.n
|
||
|
|
steps := min(n, max(2*k, k+spEigenBlock))
|
||
|
|
w := make([]float64, n)
|
||
|
|
q := make([]float64, n)
|
||
|
|
// The budget bounds the recurrence: one basis row and at most one
|
||
|
|
// coefficient per step, so all three are sized once rather than grown.
|
||
|
|
alphas = make([]float64, 0, steps)
|
||
|
|
betas = make([]float64, 0, steps)
|
||
|
|
basis = make([]float64, 0, steps*n)
|
||
|
|
scale := 0.0
|
||
|
|
addScale := func(v float64) {
|
||
|
|
if a := math.Abs(v); a > scale {
|
||
|
|
scale = a
|
||
|
|
}
|
||
|
|
}
|
||
|
|
start := func() {
|
||
|
|
for i := range n {
|
||
|
|
q[i] = gen.NormalUnit()
|
||
|
|
}
|
||
|
|
// Orthogonalise against every direction already taken so a
|
||
|
|
// restarted block cannot re-enter an earlier one.
|
||
|
|
for p := range len(alphas) {
|
||
|
|
row := basis[p*n : (p+1)*n]
|
||
|
|
d := dotF64(q, row)
|
||
|
|
for i := range n {
|
||
|
|
q[i] -= d * row[i]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
normaliseF64(q)
|
||
|
|
}
|
||
|
|
start()
|
||
|
|
for range steps {
|
||
|
|
basis = append(basis, q...)
|
||
|
|
c.matVec(q, w)
|
||
|
|
alpha := dotF64(q, w)
|
||
|
|
alphas = append(alphas, alpha)
|
||
|
|
addScale(alpha)
|
||
|
|
// Strip the two previous directions out of w, the three-term
|
||
|
|
// recurrence itself.
|
||
|
|
for i := range n {
|
||
|
|
w[i] -= alpha * q[i]
|
||
|
|
}
|
||
|
|
if len(alphas) > 1 {
|
||
|
|
beta := betas[len(betas)-1]
|
||
|
|
prev := basis[(len(alphas)-2)*n : (len(alphas)-1)*n]
|
||
|
|
for i := range n {
|
||
|
|
w[i] -= beta * prev[i]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Full reorthogonalisation, twice: one pass removes the
|
||
|
|
// accumulated loss, the second what the first reintroduces.
|
||
|
|
for range 2 {
|
||
|
|
for p := range len(alphas) {
|
||
|
|
row := basis[p*n : (p+1)*n]
|
||
|
|
d := dotF64(w, row)
|
||
|
|
for i := range n {
|
||
|
|
w[i] -= d * row[i]
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
beta := norm2F64(w)
|
||
|
|
// The exhaustion threshold is purely relative to the running
|
||
|
|
// spectral scale: an absolute floor would treat a legitimate
|
||
|
|
// tiny-scale matrix as one big deflated block and hand back
|
||
|
|
// random-start Rayleigh quotients instead of the spectrum.
|
||
|
|
if beta <= float64(n)*base.EpsF*scale {
|
||
|
|
// The Krylov space is exhausted. Close the block with a
|
||
|
|
// zero coupling and restart in a fresh direction, unless
|
||
|
|
// the basis already spans the whole space.
|
||
|
|
if len(alphas) >= n {
|
||
|
|
break
|
||
|
|
}
|
||
|
|
betas = append(betas, 0)
|
||
|
|
start()
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
betas = append(betas, beta)
|
||
|
|
addScale(beta)
|
||
|
|
for i := range n {
|
||
|
|
q[i] = w[i] / beta
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// The last beta couples to a vector that was never formed, so it
|
||
|
|
// does not belong on the tridiagonal diagonal-band.
|
||
|
|
if len(betas) >= len(alphas) {
|
||
|
|
betas = betas[:len(alphas)-1]
|
||
|
|
}
|
||
|
|
return alphas, betas, basis
|
||
|
|
}
|
||
|
|
|
||
|
|
// normaliseF64 scales a vector to unit length in place, leaving a
|
||
|
|
// zero vector untouched rather than dividing by zero.
|
||
|
|
func normaliseF64(a []float64) {
|
||
|
|
n := norm2F64(a)
|
||
|
|
if n == 0 {
|
||
|
|
return
|
||
|
|
}
|
||
|
|
for i := range a {
|
||
|
|
a[i] /= n
|
||
|
|
}
|
||
|
|
}
|