// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: MIT package linalg import ( "math" "slices" "sourcedock.dev/petrbalvin/tensor/internal/base" "sourcedock.dev/petrbalvin/tensor/internal/core" "sourcedock.dev/petrbalvin/tensor/internal/engine" ) // Sparse symmetric eigensolver. The dense `Eigen` answers a full // spectrum by tridiagonalising the whole matrix, which costs O(n³) // and materialises n² entries; both are wasteful when the matrix is // sparse and only a few extreme eigenpairs are wanted. SpEigen runs // the Lanczos process instead, whose every step costs one // sparse-matrix-vector product, then reads the Ritz pairs off the // small tridiagonal matrix it builds. // // Lanczos with the plain three-term recurrence loses orthogonality // between the basis vectors as it converges, which shows up as // duplicated ("ghost") eigenvalues. The classical cure is full // reorthogonalisation, applied here twice per step: one pass removes // the loss, the second removes what the first reintroduces at rounding // level. That costs O(m·n) per step but keeps the Ritz values honest, // and m stays small because only a few eigenpairs are wanted. // spEigenSeed is the fixed seed used when SpEigen gets no generator, // so an unseeded call is reproducible across runs. It is arbitrary but // fixed; any non-zero seed avoids the all-ones start vector, which is // orthogonal to every zero-sum eigenvector and would silently miss // them. const spEigenSeed = 20260912 // spEigenBlock is how many Lanczos steps to run past the number of // eigenpairs asked for. The extra steps buy convergence of the wanted // extremes: Ritz values settle well before Ritz vectors do, so the // budget is set by the vectors. A matrix of n ≤ k+spEigenBlock is // decomposed fully, which makes the answer exact rather than an // approximation; the partial path is what large n takes. const spEigenBlock = 40 // sparseCSR is a compressed-sparse-row view of a square real matrix, // built once per solve so the repeated matrix-vector products of the // Lanczos loop stream through contiguous value slices. type sparseCSR struct { rowStart []int colIdx []int vals []float64 n int } // cooToCSR converts a core.SparseCOO into CSR form, summing duplicate // coordinates the way COO semantics require. name names the calling // entry point in validation errors. func cooToCSR(s *core.SparseCOO, name string) (*sparseCSR, error) { ndim := len(s.Shape) nnz := s.Indices.Shape()[0] type entry struct { row, col int val float64 } entries := make([]entry, nnz) ints := s.Indices.RawInts() for i := range nnz { row, col := 0, 0 off := i * ndim for d := range ndim { idx := int(ints[off+d]) if idx < 0 || idx >= s.Shape[d] { return nil, base.Errf("%s: index [%d]=%d out of range for dim %d of size %d", name, i, idx, d, s.Shape[d]) } if d == 0 { row = idx } if d == 1 { col = idx } } entries[i] = entry{row: row, col: col, val: s.Values.FloatAt(i)} } // Sort by (row, col) so duplicates merge and each row is contiguous. slices.SortFunc(entries, func(a, b entry) int { if a.row != b.row { return a.row - b.row } return a.col - b.col }) c := &sparseCSR{ rowStart: make([]int, s.Shape[0]+1), colIdx: make([]int, 0, nnz), vals: make([]float64, 0, nnz), n: s.Shape[0], } // The duplicates are adjacent after the sort, so each run merges into // one accumulated value on the way into the compressed form. for p := 0; p < len(entries); { e := entries[p] v := e.val p++ for p < len(entries) && entries[p].row == e.row && entries[p].col == e.col { v += entries[p].val p++ } if v == 0 { continue } c.colIdx = append(c.colIdx, e.col) c.vals = append(c.vals, v) c.rowStart[e.row+1]++ } for i := range c.n { c.rowStart[i+1] += c.rowStart[i] } return c, nil } // symmetricCSR converts a core.SparseCOO into CSR form and verifies the // matrix is symmetric. Asymmetry beyond a scale-relative 1e-12 is // refused, matching the tolerance the dense `Eigen` applies. func symmetricCSR(s *core.SparseCOO, name string) (*sparseCSR, error) { c, err := cooToCSR(s, name) if err != nil { return nil, err } if err := c.checkSymmetric(name); err != nil { return nil, err } return c, nil } // checkSymmetric verifies that every stored entry has a matching // transposed entry of the same value, the property the Lanczos and // conjugate-gradient recurrences rely on. The tolerance is purely // relative to the largest magnitude present, deliberately without an // absolute floor: a floor would approve a matrix of small scale whose // asymmetry is a large fraction of it, and the recurrence would then // answer for a matrix that is neither A nor Aᵀ, whatever the wording of // the guard claims. Relative-only keeps the approval and the arithmetic // in agreement. A non-finite entry is refused in its own right: NaN // compares unequal to everything, so the mirror test alone would wave // it through as symmetric. The wording of the refusal is unchanged. func (c *sparseCSR) checkSymmetric(name string) error { scale := 0.0 for _, v := range c.vals { if a := math.Abs(v); a > scale { scale = a } } tol := 1e-12 * scale for i := range c.n { for p := c.rowStart[i]; p < c.rowStart[i+1]; p++ { j := c.colIdx[p] v := c.vals[p] if math.IsNaN(v) || math.IsInf(v, 0) { return base.Errf("%s: entry [%d,%d] is not finite", name, i, j) } mirror, ok := c.at(j, i) if !ok || math.Abs(mirror-v) > tol { return base.Errf("%s: matrix is not symmetric within 1e-12 tolerance", name) } } } return nil } // at reads the entry (row, col), reporting whether it is stored. The // row's column indices are sorted, so the lookup is a binary search: // the symmetry check calls it once per stored entry, and a linear scan // made that check quadratic in the row length. func (c *sparseCSR) at(row, col int) (float64, bool) { lo, hi := c.rowStart[row], c.rowStart[row+1] for lo < hi { mid := int(uint(lo+hi) >> 1) if c.colIdx[mid] < col { lo = mid + 1 } else { hi = mid } } if lo < c.rowStart[row+1] && c.colIdx[lo] == col { return c.vals[lo], true } return 0, false } // matVec computes y = A·x. Output rows are independent, so the row // range splits across workers once a worker's share of the stored // entries pays for the split, and runs on the calling goroutine below // that. func (c *sparseCSR) matVec(x, y []float64) { // The three structure slices are taken once: every worker's rows walk // them per stored entry, and the row loop holds no other state. rowStart := c.rowStart colIdx := c.colIdx vals := c.vals if sparseMatVecSplit(c.n, len(vals)) { engine.ParallelMin(c.n, 1, func(s, e int) { csrMatVecRange(s, e, rowStart, colIdx, vals, x, y) }) return } csrMatVecRange(0, c.n, rowStart, colIdx, vals, x, y) } // SpEigen returns the k eigenvalues of largest magnitude of a real // symmetric sparse matrix, each with its unit eigenvector. Values are // ordered by descending magnitude and vectors holds the matching // eigenvectors as columns, so column j goes with values[j]. // // The matrix must be square, real and symmetric; complex sparse // matrices are refused, as they are everywhere else on the sparse // surface. Requesting k = n recovers the complete spectrum, though at // that point the dense `Eigen` is the cheaper path. // // The starting vector is drawn from gen; passing nil uses a fixed // seed, so the result is reproducible. Because Lanczos converges to // the extremes of the spectrum first, the eigenvalues returned are // approximations whose accuracy improves with the iteration budget, // not the exact answers `Eigen` computes. func SpEigen(s *core.SparseCOO, k int, gen *core.Generator) (values, vectors *core.Array, err error) { if s.Values.Dtype() == core.Complex { return nil, nil, base.Errf("SpEigen: complex sparse matrices are not supported") } if len(s.Shape) != 2 || s.Shape[0] != s.Shape[1] { return nil, nil, base.Errf("SpEigen: needs a square 2-D sparse matrix, got shape %v", s.Shape) } n := s.Shape[0] if n == 0 { return nil, nil, base.Errf("SpEigen: zero-sized matrix, got shape %v", s.Shape) } if k < 1 || k > n { return nil, nil, base.Errf("SpEigen: k must be in [1, %d], got %d", n, k) } c, err := symmetricCSR(s, "SpEigen") if err != nil { return nil, nil, err } // The projected tridiagonal carries the matrix's magnitude, and the // symmetric sweep's deflation floor is taken against the sum of its // squared entries: above about 1.3e154 that sum overflows to +Inf, so // every subdiagonal reads as negligible, the sweep deflates the whole // block at once and the raw Rayleigh quotients come back as the // spectrum. The matrix is moved into the window for the recurrence and // the Ritz values, which carry the scale, are moved back before they // are returned; the basis vectors are unit-length and therefore // scale-free. ws := windowScale(maxMagF64(c.vals)) if ws != 1 { scaleFloats(c.vals, ws) } if gen == nil { gen = core.NewGenerator(spEigenSeed) } alphas, betas, basis := c.lanczos(k, gen) r := len(alphas) // The small tridiagonal matrix whose eigenpairs are the Ritz // approximations. Its eigenvectors accumulate into the identity, // so the columns of tVec are the tridiagonal eigenvectors. tMat := make([]float64, r*r) for i := range r { tMat[i*r+i] = alphas[i] if i+1 < r { tMat[i*r+i+1] = betas[i] tMat[(i+1)*r+i] = betas[i] } } tVec := eye(r) if err := symmetricQr(tMat, tVec, r); err != nil { return nil, nil, base.Errf("SpEigen: %w", err) } // Rank the Ritz pairs by descending magnitude and keep k. idx := make([]int, r) for i := range r { idx[i] = i } slices.SortFunc(idx, func(a, b int) int { da, db := math.Abs(tMat[a*r+a]), math.Abs(tMat[b*r+b]) if da != db { if da > db { return -1 } return 1 } return a - b }) sel := idx[:k] outVals := make([]float64, k) outVecs := make([]float64, n*k) tmp := make([]float64, n) for j, si := range sel { outVals[j] = tMat[si*r+si] // Ritz vector: lift the tridiagonal eigenvector through the // Lanczos basis, v = Q·y. clear(tmp) for p := range r { y := tVec[p*r+si] if y == 0 { continue } row := basis[p*n : (p+1)*n] for i := range n { tmp[i] += y * row[i] } } normaliseF64(tmp) for i := range n { outVecs[i*k+j] = tmp[i] } } // Only the Ritz values carry the matrix's scale; the vectors do not. if ws != 1 { unscaleFloats(outVals, ws) } return floatsToArray(outVals, []int{k}), floatsToArray(outVecs, []int{n, k}), nil } // lanczos runs the process, returning the diagonal alpha, the // off-diagonal beta and the basis vectors flattened as n-wide rows. // When the recurrence collapses, meaning the Krylov space is // exhausted, the block is closed off with a zero coupling and a fresh // direction restarts it: the tridiagonal matrix becomes // block-diagonal, and each block still yields valid Ritz pairs. func (c *sparseCSR) lanczos(k int, gen *core.Generator) (alphas, betas []float64, basis []float64) { n := c.n steps := min(n, max(2*k, k+spEigenBlock)) w := make([]float64, n) q := make([]float64, n) // The budget bounds the recurrence: one basis row and at most one // coefficient per step, so all three are sized once rather than grown. alphas = make([]float64, 0, steps) betas = make([]float64, 0, steps) basis = make([]float64, 0, steps*n) scale := 0.0 addScale := func(v float64) { if a := math.Abs(v); a > scale { scale = a } } start := func() { for i := range n { q[i] = gen.NormalUnit() } // Orthogonalise against every direction already taken so a // restarted block cannot re-enter an earlier one. for p := range len(alphas) { row := basis[p*n : (p+1)*n] d := dotF64(q, row) for i := range n { q[i] -= d * row[i] } } normaliseF64(q) } start() for range steps { basis = append(basis, q...) c.matVec(q, w) alpha := dotF64(q, w) alphas = append(alphas, alpha) addScale(alpha) // Strip the two previous directions out of w, the three-term // recurrence itself. for i := range n { w[i] -= alpha * q[i] } if len(alphas) > 1 { beta := betas[len(betas)-1] prev := basis[(len(alphas)-2)*n : (len(alphas)-1)*n] for i := range n { w[i] -= beta * prev[i] } } // Full reorthogonalisation, twice: one pass removes the // accumulated loss, the second what the first reintroduces. for range 2 { for p := range len(alphas) { row := basis[p*n : (p+1)*n] d := dotF64(w, row) for i := range n { w[i] -= d * row[i] } } } beta := norm2F64(w) // The exhaustion threshold is purely relative to the running // spectral scale: an absolute floor would treat a legitimate // tiny-scale matrix as one big deflated block and hand back // random-start Rayleigh quotients instead of the spectrum. if beta <= float64(n)*base.EpsF*scale { // The Krylov space is exhausted. Close the block with a // zero coupling and restart in a fresh direction, unless // the basis already spans the whole space. if len(alphas) >= n { break } betas = append(betas, 0) start() continue } betas = append(betas, beta) addScale(beta) for i := range n { q[i] = w[i] / beta } } // The last beta couples to a vector that was never formed, so it // does not belong on the tridiagonal diagonal-band. if len(betas) >= len(alphas) { betas = betas[:len(alphas)-1] } return alphas, betas, basis } // normaliseF64 scales a vector to unit length in place, leaving a // zero vector untouched rather than dividing by zero. func normaliseF64(a []float64) { n := norm2F64(a) if n == 0 { return } for i := range a { a[i] /= n } }