Files
tensor/stats/pca.go
T
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

414 lines
12 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package stats
import (
"cmp"
"math"
"slices"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
// Principal component analysis: the rotation of a cloud of
// observations onto the axes along which it actually spreads, with the
// variances of those axes, the coordinates of every observation on
// them, and the whitening transform that flattens the cloud to unit
// covariance. The covariance comes from the package's own
// CovarianceMatrix; its eigendecomposition is computed here, because
// the package's linear algebra has so far only needed LU solves, and a
// symmetric eigensolver is carried as a cyclic Jacobi iteration:
// unconditionally convergent on symmetric matrices, quadratically so
// near the answer, and exactly the right instrument for the small
// dense covariance matrices a PCA runs on.
// PCAResult carries the decomposition of an observation array.
type PCAResult struct {
// Mean holds the column means of the observations the fit ran on.
Mean []float64
// Loadings is the (p, p) rotation: entry (j, k) is the loading of
// variable j on component k, columns ordered by falling explained
// variance, orthonormal as columns. Every column's largest-magnitude
// loading is positive, the first index winning a tie, so the axes
// have a fixed orientation and two runs on the same data agree
// sign for sign.
Loadings *core.Array
// Scores is the (n, p) array of the observations' coordinates on
// the components: the centred observations times the loadings.
Scores *core.Array
// ExplainedVariance holds each component's eigenvalue of the
// covariance, in the loadings' order; ExplainedVarianceRatio
// divides it by the total variance, so the ratios sum to one.
ExplainedVariance []float64
ExplainedVarianceRatio []float64
// Whitening and Unwhitening are the (p, p) transforms between the
// centred observations and the unit-covariance representation:
// whitening maps a centred row x to x·Whitening, unwhitening maps
// it back. Both are nil when the covariance is rank deficient,
// where no whitening transform exists.
Whitening *core.Array
Unwhitening *core.Array
}
// PCA decomposes the observations in a, an (n, p) array whose rows are
// observations and columns variables, onto their principal
// components. At least two observations are needed, the input must be
// real and finite, and data with no variance at all has no
// decomposition to report. A rank-deficient covariance decomposes
// normally, with zero variances on the empty components; only the
// whitening transforms, which would divide by those variances, are
// withheld.
func PCA(a *core.Array) (*PCAResult, error) {
const name = "PCA"
covArr, err := CovarianceMatrix(a)
if err != nil {
return nil, base.Errf("%s: %w", name, err)
}
n, p := a.Shape()[0], a.Shape()[1]
// The covariance is read once through its payload where it has one;
// the values are the ones FloatAt returns.
covVals := rawFloats(covArr)
cov := make([][]float64, p)
total := 0.0
for i := range p {
cov[i] = make([]float64, p)
if covVals != nil {
copy(cov[i], covVals[i*p:i*p+p])
} else {
for j := range p {
cov[i][j] = covArr.FloatAt(i*p + j)
}
}
total += cov[i][i]
}
if total == 0 {
return nil, base.Errf("%s: the observations have no variance", name)
}
values, vectors := jacobiEigen(cov)
// A covariance is positive semidefinite by construction, so a
// negative eigenvalue beyond a rounding-scale fraction of the
// largest means the decomposition cannot be trusted; anything
// within it is rounding and clamps to an exactly empty component.
worst := 0.0
for _, v := range values {
worst = max(worst, math.Abs(v))
}
for k, v := range values {
if v < 0 {
if v < -1e-9*worst {
return nil, base.Errf("%s: the covariance decomposed to the negative eigenvalue %g, not a covariance", name, v)
}
values[k] = 0
}
}
// Column means, then the scores of every observation.
means := make([]float64, p)
rows := rawFloats(a)
for j := range p {
s := 0.0
if rows != nil {
for i := range n {
s += rows[i*p+j]
}
} else {
for i := range n {
s += a.FloatAt(i*p + j)
}
}
means[j] = s / float64(n)
}
scores := make([]float64, 0, n*p)
for i := range n {
for k := range p {
s := 0.0
vk := vectors[k]
for j := range p {
var xj float64
if rows != nil {
xj = rows[i*p+j]
} else {
xj = a.FloatAt(i*p + j)
}
s += (xj - means[j]) * vk[j]
}
scores = append(scores, s)
}
}
// The loadings land in the documented (variable, component)
// layout: entry (j, k) is component k's loading on variable j.
loadings := make([]float64, 0, p*p)
for j := range p {
for k := range p {
loadings = append(loadings, vectors[k][j])
}
}
out := &PCAResult{
Mean: means,
Loadings: floatsToArray(loadings, []int{p, p}),
Scores: floatsToArray(scores, []int{n, p}),
ExplainedVariance: values,
ExplainedVarianceRatio: make([]float64, p),
}
for k := range p {
out.ExplainedVarianceRatio[k] = values[k] / total
}
if values[p-1] > 0 {
whitening := make([]float64, 0, p*p)
unwhitening := make([]float64, 0, p*p)
for i := range p {
for k := range p {
whitening = append(whitening, vectors[k][i]/math.Sqrt(values[k]))
}
}
for k := range p {
for j := range p {
unwhitening = append(unwhitening, math.Sqrt(values[k])*vectors[k][j])
}
}
out.Whitening = floatsToArray(whitening, []int{p, p})
out.Unwhitening = floatsToArray(unwhitening, []int{p, p})
}
return out, nil
}
// Whiten maps the observations in x, an (n, p) array on the fit's own
// variables, to their unit-covariance representation: the centred
// observations expressed on the components and scaled by each
// component's standard deviation. The result has the identity as its
// covariance, up to the fit's rank: a rank-deficient fit refuses to
// whiten rather than divide by an empty component's variance.
func (r *PCAResult) Whiten(x *core.Array) (*core.Array, error) {
const name = "Whiten"
if r == nil {
return nil, base.Errf("%s: no fit to whiten with", name)
}
if r.Whitening == nil {
return nil, base.Errf("%s: the fit's covariance is rank deficient and no whitening transform exists", name)
}
if x.NDim() != 2 {
return nil, base.Errf("%s: the observations must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
}
if x.Dtype() == core.Complex {
return nil, base.Errf("%s: complex observations are not supported", name)
}
if err := checkFinite(name, "the observations", x); err != nil {
return nil, err
}
n, p := x.Shape()[0], len(r.Mean)
if x.Shape()[1] != p {
return nil, base.Errf("%s: the observations have %d columns, the fit %d", name, x.Shape()[1], p)
}
rows := rawFloats(x)
out := core.New(core.Float, n, p)
vals := out.RawFloats()
// The transform is read once into a plain slice: the values are the
// ones FloatAt returns.
transform := make([]float64, p*p)
if fs := rawFloats(r.Whitening); fs != nil {
copy(transform, fs)
} else {
for i := range transform {
transform[i] = r.Whitening.FloatAt(i)
}
}
for i := range n {
for k := range p {
s := 0.0
for j := range p {
var xj float64
if rows != nil {
xj = rows[i*p+j]
} else {
xj = x.FloatAt(i*p + j)
}
s += (xj - r.Mean[j]) * transform[j*p+k]
}
vals[i*p+k] = s
}
}
return out, nil
}
// Unwhiten inverts Whiten: it maps a unit-covariance representation
// back to the observations' own coordinates, the observations
// themselves recovered exactly to rounding.
func (r *PCAResult) Unwhiten(z *core.Array) (*core.Array, error) {
const name = "Unwhiten"
if r == nil {
return nil, base.Errf("%s: no fit to unwhiten with", name)
}
if r.Unwhitening == nil {
return nil, base.Errf("%s: the fit's covariance is rank deficient and no unwhitening transform exists", name)
}
if z.NDim() != 2 {
return nil, base.Errf("%s: the whitened observations must be rank 2, got shape %s", name, base.ShapeText(z.Shape()))
}
if z.Dtype() == core.Complex {
return nil, base.Errf("%s: complex observations are not supported", name)
}
if err := checkFinite(name, "the whitened observations", z); err != nil {
return nil, err
}
n, p := z.Shape()[0], len(r.Mean)
if z.Shape()[1] != p {
return nil, base.Errf("%s: the whitened observations have %d columns, the fit %d", name, z.Shape()[1], p)
}
rows := rawFloats(z)
out := core.New(core.Float, n, p)
vals := out.RawFloats()
// The transform is read once into a plain slice: the values are the
// ones FloatAt returns.
transform := make([]float64, p*p)
if fs := rawFloats(r.Unwhitening); fs != nil {
copy(transform, fs)
} else {
for i := range transform {
transform[i] = r.Unwhitening.FloatAt(i)
}
}
for i := range n {
for j := range p {
s := 0.0
for k := range p {
var zk float64
if rows != nil {
zk = rows[i*p+k]
} else {
zk = z.FloatAt(i*p + k)
}
s += zk * transform[k*p+j]
}
vals[i*p+j] = r.Mean[j] + s
}
}
return out, nil
}
// jacobiEigen eigendecomposes the symmetric matrix a by the cyclic
// Jacobi iteration: every sweep applies the plane rotation that zeroes
// each off-diagonal entry in turn, accumulating the rotations into the
// eigenvector matrix. The iteration is unconditionally convergent for
// symmetric matrices and quadratically convergent near the answer, so
// a handful of sweeps carries any small dense matrix to machine
// precision. The eigenvalues come back in falling order with the
// matching orthonormal eigenvectors as columns, each column oriented
// so its largest-magnitude entry is positive, the first index winning
// a tie.
func jacobiEigen(a [][]float64) (values []float64, vectors [][]float64) {
p := len(a)
m := make([][]float64, p)
for i := range p {
m[i] = append([]float64(nil), a[i]...)
}
v := make([][]float64, p)
for i := range p {
v[i] = make([]float64, p)
v[i][i] = 1
}
scale := 0.0
for i := range p {
for j := i; j < p; j++ {
scale += m[i][j] * m[i][j]
}
}
for range 100 {
off := 0.0
for i := range p {
for j := i + 1; j < p; j++ {
off += m[i][j] * m[i][j]
}
}
if math.Sqrt(off) <= 1e-14*math.Sqrt(scale) {
break
}
for q := 1; q < p; q++ {
for i := 0; i < q; i++ {
aiq := m[i][q]
if aiq == 0 {
continue
}
theta := (m[q][q] - m[i][i]) / (2 * aiq)
sign := 1.0
if theta < 0 {
sign = -1
}
t := sign / (math.Abs(theta) + math.Sqrt(theta*theta+1))
c := 1 / math.Sqrt(t*t+1)
s := t * c
tangent := s / (1 + c)
aii := m[i][i]
aqq := m[q][q]
m[i][i] = aii - t*aiq
m[q][q] = aqq + t*aiq
m[i][q] = 0
m[q][i] = 0
for k := range p {
if k == i || k == q {
continue
}
aki := m[k][i]
akq := m[k][q]
m[k][i] = aki - s*(akq+tangent*aki)
m[i][k] = m[k][i]
m[k][q] = akq + s*(aki-tangent*akq)
m[q][k] = m[k][q]
}
for k := range p {
vki := v[k][i]
vkq := v[k][q]
v[k][i] = vki - s*(vkq+tangent*vki)
v[k][q] = vkq + s*(vki-tangent*vkq)
}
}
}
}
values = make([]float64, p)
for i := range p {
values[i] = m[i][i]
}
// Falling order, ties left in their original order.
order := make([]int, p)
for i := range p {
order[i] = i
}
slices.SortStableFunc(order, func(a, b int) int {
return cmp.Compare(values[b], values[a])
})
sortedValues := make([]float64, p)
sortedVectors := make([][]float64, p)
for k := range p {
sortedValues[k] = values[order[k]]
sortedVectors[k] = make([]float64, p)
for i := range p {
sortedVectors[k][i] = v[i][order[k]]
}
}
fixEigenSigns(sortedVectors)
return sortedValues, sortedVectors
}
// fixEigenSigns orients every eigenvector, held as a row of v, so its
// largest-magnitude entry is positive, the first index winning a tie.
// An eigendecomposition is defined only up to each eigenvector's sign,
// and a decomposition that reports different signs on the same input
// twice would be no decomposition at all.
func fixEigenSigns(v [][]float64) {
for k := range v {
worst := 0.0
index := 0
for i := range v[k] {
if a := math.Abs(v[k][i]); a > worst {
worst = a
index = i
}
}
if v[k][index] < 0 {
for i := range v[k] {
v[k][i] = -v[k][i]
}
}
}
}