414 lines
12 KiB
Go
414 lines
12 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
package stats
|
|
|
|
import (
|
|
"cmp"
|
|
"math"
|
|
"slices"
|
|
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
|
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
|
)
|
|
|
|
// Principal component analysis: the rotation of a cloud of
|
|
// observations onto the axes along which it actually spreads, with the
|
|
// variances of those axes, the coordinates of every observation on
|
|
// them, and the whitening transform that flattens the cloud to unit
|
|
// covariance. The covariance comes from the package's own
|
|
// CovarianceMatrix; its eigendecomposition is computed here, because
|
|
// the package's linear algebra has so far only needed LU solves, and a
|
|
// symmetric eigensolver is carried as a cyclic Jacobi iteration:
|
|
// unconditionally convergent on symmetric matrices, quadratically so
|
|
// near the answer, and exactly the right instrument for the small
|
|
// dense covariance matrices a PCA runs on.
|
|
|
|
// PCAResult carries the decomposition of an observation array.
|
|
type PCAResult struct {
|
|
// Mean holds the column means of the observations the fit ran on.
|
|
Mean []float64
|
|
// Loadings is the (p, p) rotation: entry (j, k) is the loading of
|
|
// variable j on component k, columns ordered by falling explained
|
|
// variance, orthonormal as columns. Every column's largest-magnitude
|
|
// loading is positive, the first index winning a tie, so the axes
|
|
// have a fixed orientation and two runs on the same data agree
|
|
// sign for sign.
|
|
Loadings *core.Array
|
|
// Scores is the (n, p) array of the observations' coordinates on
|
|
// the components: the centred observations times the loadings.
|
|
Scores *core.Array
|
|
// ExplainedVariance holds each component's eigenvalue of the
|
|
// covariance, in the loadings' order; ExplainedVarianceRatio
|
|
// divides it by the total variance, so the ratios sum to one.
|
|
ExplainedVariance []float64
|
|
ExplainedVarianceRatio []float64
|
|
// Whitening and Unwhitening are the (p, p) transforms between the
|
|
// centred observations and the unit-covariance representation:
|
|
// whitening maps a centred row x to x·Whitening, unwhitening maps
|
|
// it back. Both are nil when the covariance is rank deficient,
|
|
// where no whitening transform exists.
|
|
Whitening *core.Array
|
|
Unwhitening *core.Array
|
|
}
|
|
|
|
// PCA decomposes the observations in a, an (n, p) array whose rows are
|
|
// observations and columns variables, onto their principal
|
|
// components. At least two observations are needed, the input must be
|
|
// real and finite, and data with no variance at all has no
|
|
// decomposition to report. A rank-deficient covariance decomposes
|
|
// normally, with zero variances on the empty components; only the
|
|
// whitening transforms, which would divide by those variances, are
|
|
// withheld.
|
|
func PCA(a *core.Array) (*PCAResult, error) {
|
|
const name = "PCA"
|
|
covArr, err := CovarianceMatrix(a)
|
|
if err != nil {
|
|
return nil, base.Errf("%s: %w", name, err)
|
|
}
|
|
n, p := a.Shape()[0], a.Shape()[1]
|
|
// The covariance is read once through its payload where it has one;
|
|
// the values are the ones FloatAt returns.
|
|
covVals := rawFloats(covArr)
|
|
cov := make([][]float64, p)
|
|
total := 0.0
|
|
for i := range p {
|
|
cov[i] = make([]float64, p)
|
|
if covVals != nil {
|
|
copy(cov[i], covVals[i*p:i*p+p])
|
|
} else {
|
|
for j := range p {
|
|
cov[i][j] = covArr.FloatAt(i*p + j)
|
|
}
|
|
}
|
|
total += cov[i][i]
|
|
}
|
|
if total == 0 {
|
|
return nil, base.Errf("%s: the observations have no variance", name)
|
|
}
|
|
values, vectors := jacobiEigen(cov)
|
|
// A covariance is positive semidefinite by construction, so a
|
|
// negative eigenvalue beyond a rounding-scale fraction of the
|
|
// largest means the decomposition cannot be trusted; anything
|
|
// within it is rounding and clamps to an exactly empty component.
|
|
worst := 0.0
|
|
for _, v := range values {
|
|
worst = max(worst, math.Abs(v))
|
|
}
|
|
for k, v := range values {
|
|
if v < 0 {
|
|
if v < -1e-9*worst {
|
|
return nil, base.Errf("%s: the covariance decomposed to the negative eigenvalue %g, not a covariance", name, v)
|
|
}
|
|
values[k] = 0
|
|
}
|
|
}
|
|
// Column means, then the scores of every observation.
|
|
means := make([]float64, p)
|
|
rows := rawFloats(a)
|
|
for j := range p {
|
|
s := 0.0
|
|
if rows != nil {
|
|
for i := range n {
|
|
s += rows[i*p+j]
|
|
}
|
|
} else {
|
|
for i := range n {
|
|
s += a.FloatAt(i*p + j)
|
|
}
|
|
}
|
|
means[j] = s / float64(n)
|
|
}
|
|
scores := make([]float64, 0, n*p)
|
|
for i := range n {
|
|
for k := range p {
|
|
s := 0.0
|
|
vk := vectors[k]
|
|
for j := range p {
|
|
var xj float64
|
|
if rows != nil {
|
|
xj = rows[i*p+j]
|
|
} else {
|
|
xj = a.FloatAt(i*p + j)
|
|
}
|
|
s += (xj - means[j]) * vk[j]
|
|
}
|
|
scores = append(scores, s)
|
|
}
|
|
}
|
|
// The loadings land in the documented (variable, component)
|
|
// layout: entry (j, k) is component k's loading on variable j.
|
|
loadings := make([]float64, 0, p*p)
|
|
for j := range p {
|
|
for k := range p {
|
|
loadings = append(loadings, vectors[k][j])
|
|
}
|
|
}
|
|
out := &PCAResult{
|
|
Mean: means,
|
|
Loadings: floatsToArray(loadings, []int{p, p}),
|
|
Scores: floatsToArray(scores, []int{n, p}),
|
|
ExplainedVariance: values,
|
|
ExplainedVarianceRatio: make([]float64, p),
|
|
}
|
|
for k := range p {
|
|
out.ExplainedVarianceRatio[k] = values[k] / total
|
|
}
|
|
if values[p-1] > 0 {
|
|
whitening := make([]float64, 0, p*p)
|
|
unwhitening := make([]float64, 0, p*p)
|
|
for i := range p {
|
|
for k := range p {
|
|
whitening = append(whitening, vectors[k][i]/math.Sqrt(values[k]))
|
|
}
|
|
}
|
|
for k := range p {
|
|
for j := range p {
|
|
unwhitening = append(unwhitening, math.Sqrt(values[k])*vectors[k][j])
|
|
}
|
|
}
|
|
out.Whitening = floatsToArray(whitening, []int{p, p})
|
|
out.Unwhitening = floatsToArray(unwhitening, []int{p, p})
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// Whiten maps the observations in x, an (n, p) array on the fit's own
|
|
// variables, to their unit-covariance representation: the centred
|
|
// observations expressed on the components and scaled by each
|
|
// component's standard deviation. The result has the identity as its
|
|
// covariance, up to the fit's rank: a rank-deficient fit refuses to
|
|
// whiten rather than divide by an empty component's variance.
|
|
func (r *PCAResult) Whiten(x *core.Array) (*core.Array, error) {
|
|
const name = "Whiten"
|
|
if r == nil {
|
|
return nil, base.Errf("%s: no fit to whiten with", name)
|
|
}
|
|
if r.Whitening == nil {
|
|
return nil, base.Errf("%s: the fit's covariance is rank deficient and no whitening transform exists", name)
|
|
}
|
|
if x.NDim() != 2 {
|
|
return nil, base.Errf("%s: the observations must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
|
|
}
|
|
if x.Dtype() == core.Complex {
|
|
return nil, base.Errf("%s: complex observations are not supported", name)
|
|
}
|
|
if err := checkFinite(name, "the observations", x); err != nil {
|
|
return nil, err
|
|
}
|
|
n, p := x.Shape()[0], len(r.Mean)
|
|
if x.Shape()[1] != p {
|
|
return nil, base.Errf("%s: the observations have %d columns, the fit %d", name, x.Shape()[1], p)
|
|
}
|
|
rows := rawFloats(x)
|
|
out := core.New(core.Float, n, p)
|
|
vals := out.RawFloats()
|
|
// The transform is read once into a plain slice: the values are the
|
|
// ones FloatAt returns.
|
|
transform := make([]float64, p*p)
|
|
if fs := rawFloats(r.Whitening); fs != nil {
|
|
copy(transform, fs)
|
|
} else {
|
|
for i := range transform {
|
|
transform[i] = r.Whitening.FloatAt(i)
|
|
}
|
|
}
|
|
for i := range n {
|
|
for k := range p {
|
|
s := 0.0
|
|
for j := range p {
|
|
var xj float64
|
|
if rows != nil {
|
|
xj = rows[i*p+j]
|
|
} else {
|
|
xj = x.FloatAt(i*p + j)
|
|
}
|
|
s += (xj - r.Mean[j]) * transform[j*p+k]
|
|
}
|
|
vals[i*p+k] = s
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// Unwhiten inverts Whiten: it maps a unit-covariance representation
|
|
// back to the observations' own coordinates, the observations
|
|
// themselves recovered exactly to rounding.
|
|
func (r *PCAResult) Unwhiten(z *core.Array) (*core.Array, error) {
|
|
const name = "Unwhiten"
|
|
if r == nil {
|
|
return nil, base.Errf("%s: no fit to unwhiten with", name)
|
|
}
|
|
if r.Unwhitening == nil {
|
|
return nil, base.Errf("%s: the fit's covariance is rank deficient and no unwhitening transform exists", name)
|
|
}
|
|
if z.NDim() != 2 {
|
|
return nil, base.Errf("%s: the whitened observations must be rank 2, got shape %s", name, base.ShapeText(z.Shape()))
|
|
}
|
|
if z.Dtype() == core.Complex {
|
|
return nil, base.Errf("%s: complex observations are not supported", name)
|
|
}
|
|
if err := checkFinite(name, "the whitened observations", z); err != nil {
|
|
return nil, err
|
|
}
|
|
n, p := z.Shape()[0], len(r.Mean)
|
|
if z.Shape()[1] != p {
|
|
return nil, base.Errf("%s: the whitened observations have %d columns, the fit %d", name, z.Shape()[1], p)
|
|
}
|
|
rows := rawFloats(z)
|
|
out := core.New(core.Float, n, p)
|
|
vals := out.RawFloats()
|
|
// The transform is read once into a plain slice: the values are the
|
|
// ones FloatAt returns.
|
|
transform := make([]float64, p*p)
|
|
if fs := rawFloats(r.Unwhitening); fs != nil {
|
|
copy(transform, fs)
|
|
} else {
|
|
for i := range transform {
|
|
transform[i] = r.Unwhitening.FloatAt(i)
|
|
}
|
|
}
|
|
for i := range n {
|
|
for j := range p {
|
|
s := 0.0
|
|
for k := range p {
|
|
var zk float64
|
|
if rows != nil {
|
|
zk = rows[i*p+k]
|
|
} else {
|
|
zk = z.FloatAt(i*p + k)
|
|
}
|
|
s += zk * transform[k*p+j]
|
|
}
|
|
vals[i*p+j] = r.Mean[j] + s
|
|
}
|
|
}
|
|
return out, nil
|
|
}
|
|
|
|
// jacobiEigen eigendecomposes the symmetric matrix a by the cyclic
|
|
// Jacobi iteration: every sweep applies the plane rotation that zeroes
|
|
// each off-diagonal entry in turn, accumulating the rotations into the
|
|
// eigenvector matrix. The iteration is unconditionally convergent for
|
|
// symmetric matrices and quadratically convergent near the answer, so
|
|
// a handful of sweeps carries any small dense matrix to machine
|
|
// precision. The eigenvalues come back in falling order with the
|
|
// matching orthonormal eigenvectors as columns, each column oriented
|
|
// so its largest-magnitude entry is positive, the first index winning
|
|
// a tie.
|
|
func jacobiEigen(a [][]float64) (values []float64, vectors [][]float64) {
|
|
p := len(a)
|
|
m := make([][]float64, p)
|
|
for i := range p {
|
|
m[i] = append([]float64(nil), a[i]...)
|
|
}
|
|
v := make([][]float64, p)
|
|
for i := range p {
|
|
v[i] = make([]float64, p)
|
|
v[i][i] = 1
|
|
}
|
|
scale := 0.0
|
|
for i := range p {
|
|
for j := i; j < p; j++ {
|
|
scale += m[i][j] * m[i][j]
|
|
}
|
|
}
|
|
for range 100 {
|
|
off := 0.0
|
|
for i := range p {
|
|
for j := i + 1; j < p; j++ {
|
|
off += m[i][j] * m[i][j]
|
|
}
|
|
}
|
|
if math.Sqrt(off) <= 1e-14*math.Sqrt(scale) {
|
|
break
|
|
}
|
|
for q := 1; q < p; q++ {
|
|
for i := 0; i < q; i++ {
|
|
aiq := m[i][q]
|
|
if aiq == 0 {
|
|
continue
|
|
}
|
|
theta := (m[q][q] - m[i][i]) / (2 * aiq)
|
|
sign := 1.0
|
|
if theta < 0 {
|
|
sign = -1
|
|
}
|
|
t := sign / (math.Abs(theta) + math.Sqrt(theta*theta+1))
|
|
c := 1 / math.Sqrt(t*t+1)
|
|
s := t * c
|
|
tangent := s / (1 + c)
|
|
aii := m[i][i]
|
|
aqq := m[q][q]
|
|
m[i][i] = aii - t*aiq
|
|
m[q][q] = aqq + t*aiq
|
|
m[i][q] = 0
|
|
m[q][i] = 0
|
|
for k := range p {
|
|
if k == i || k == q {
|
|
continue
|
|
}
|
|
aki := m[k][i]
|
|
akq := m[k][q]
|
|
m[k][i] = aki - s*(akq+tangent*aki)
|
|
m[i][k] = m[k][i]
|
|
m[k][q] = akq + s*(aki-tangent*akq)
|
|
m[q][k] = m[k][q]
|
|
}
|
|
for k := range p {
|
|
vki := v[k][i]
|
|
vkq := v[k][q]
|
|
v[k][i] = vki - s*(vkq+tangent*vki)
|
|
v[k][q] = vkq + s*(vki-tangent*vkq)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
values = make([]float64, p)
|
|
for i := range p {
|
|
values[i] = m[i][i]
|
|
}
|
|
// Falling order, ties left in their original order.
|
|
order := make([]int, p)
|
|
for i := range p {
|
|
order[i] = i
|
|
}
|
|
slices.SortStableFunc(order, func(a, b int) int {
|
|
return cmp.Compare(values[b], values[a])
|
|
})
|
|
sortedValues := make([]float64, p)
|
|
sortedVectors := make([][]float64, p)
|
|
for k := range p {
|
|
sortedValues[k] = values[order[k]]
|
|
sortedVectors[k] = make([]float64, p)
|
|
for i := range p {
|
|
sortedVectors[k][i] = v[i][order[k]]
|
|
}
|
|
}
|
|
fixEigenSigns(sortedVectors)
|
|
return sortedValues, sortedVectors
|
|
}
|
|
|
|
// fixEigenSigns orients every eigenvector, held as a row of v, so its
|
|
// largest-magnitude entry is positive, the first index winning a tie.
|
|
// An eigendecomposition is defined only up to each eigenvector's sign,
|
|
// and a decomposition that reports different signs on the same input
|
|
// twice would be no decomposition at all.
|
|
func fixEigenSigns(v [][]float64) {
|
|
for k := range v {
|
|
worst := 0.0
|
|
index := 0
|
|
for i := range v[k] {
|
|
if a := math.Abs(v[k][i]); a > worst {
|
|
worst = a
|
|
index = i
|
|
}
|
|
}
|
|
if v[k][index] < 0 {
|
|
for i := range v[k] {
|
|
v[k][i] = -v[k][i]
|
|
}
|
|
}
|
|
}
|
|
}
|