feat: initial release
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+413
@@ -0,0 +1,413 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package stats
|
||||
|
||||
import (
|
||||
"cmp"
|
||||
"math"
|
||||
"slices"
|
||||
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
||||
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
||||
)
|
||||
|
||||
// Principal component analysis: the rotation of a cloud of
|
||||
// observations onto the axes along which it actually spreads, with the
|
||||
// variances of those axes, the coordinates of every observation on
|
||||
// them, and the whitening transform that flattens the cloud to unit
|
||||
// covariance. The covariance comes from the package's own
|
||||
// CovarianceMatrix; its eigendecomposition is computed here, because
|
||||
// the package's linear algebra has so far only needed LU solves, and a
|
||||
// symmetric eigensolver is carried as a cyclic Jacobi iteration:
|
||||
// unconditionally convergent on symmetric matrices, quadratically so
|
||||
// near the answer, and exactly the right instrument for the small
|
||||
// dense covariance matrices a PCA runs on.
|
||||
|
||||
// PCAResult carries the decomposition of an observation array.
|
||||
type PCAResult struct {
|
||||
// Mean holds the column means of the observations the fit ran on.
|
||||
Mean []float64
|
||||
// Loadings is the (p, p) rotation: entry (j, k) is the loading of
|
||||
// variable j on component k, columns ordered by falling explained
|
||||
// variance, orthonormal as columns. Every column's largest-magnitude
|
||||
// loading is positive, the first index winning a tie, so the axes
|
||||
// have a fixed orientation and two runs on the same data agree
|
||||
// sign for sign.
|
||||
Loadings *core.Array
|
||||
// Scores is the (n, p) array of the observations' coordinates on
|
||||
// the components: the centred observations times the loadings.
|
||||
Scores *core.Array
|
||||
// ExplainedVariance holds each component's eigenvalue of the
|
||||
// covariance, in the loadings' order; ExplainedVarianceRatio
|
||||
// divides it by the total variance, so the ratios sum to one.
|
||||
ExplainedVariance []float64
|
||||
ExplainedVarianceRatio []float64
|
||||
// Whitening and Unwhitening are the (p, p) transforms between the
|
||||
// centred observations and the unit-covariance representation:
|
||||
// whitening maps a centred row x to x·Whitening, unwhitening maps
|
||||
// it back. Both are nil when the covariance is rank deficient,
|
||||
// where no whitening transform exists.
|
||||
Whitening *core.Array
|
||||
Unwhitening *core.Array
|
||||
}
|
||||
|
||||
// PCA decomposes the observations in a, an (n, p) array whose rows are
|
||||
// observations and columns variables, onto their principal
|
||||
// components. At least two observations are needed, the input must be
|
||||
// real and finite, and data with no variance at all has no
|
||||
// decomposition to report. A rank-deficient covariance decomposes
|
||||
// normally, with zero variances on the empty components; only the
|
||||
// whitening transforms, which would divide by those variances, are
|
||||
// withheld.
|
||||
func PCA(a *core.Array) (*PCAResult, error) {
|
||||
const name = "PCA"
|
||||
covArr, err := CovarianceMatrix(a)
|
||||
if err != nil {
|
||||
return nil, base.Errf("%s: %w", name, err)
|
||||
}
|
||||
n, p := a.Shape()[0], a.Shape()[1]
|
||||
// The covariance is read once through its payload where it has one;
|
||||
// the values are the ones FloatAt returns.
|
||||
covVals := rawFloats(covArr)
|
||||
cov := make([][]float64, p)
|
||||
total := 0.0
|
||||
for i := range p {
|
||||
cov[i] = make([]float64, p)
|
||||
if covVals != nil {
|
||||
copy(cov[i], covVals[i*p:i*p+p])
|
||||
} else {
|
||||
for j := range p {
|
||||
cov[i][j] = covArr.FloatAt(i*p + j)
|
||||
}
|
||||
}
|
||||
total += cov[i][i]
|
||||
}
|
||||
if total == 0 {
|
||||
return nil, base.Errf("%s: the observations have no variance", name)
|
||||
}
|
||||
values, vectors := jacobiEigen(cov)
|
||||
// A covariance is positive semidefinite by construction, so a
|
||||
// negative eigenvalue beyond a rounding-scale fraction of the
|
||||
// largest means the decomposition cannot be trusted; anything
|
||||
// within it is rounding and clamps to an exactly empty component.
|
||||
worst := 0.0
|
||||
for _, v := range values {
|
||||
worst = max(worst, math.Abs(v))
|
||||
}
|
||||
for k, v := range values {
|
||||
if v < 0 {
|
||||
if v < -1e-9*worst {
|
||||
return nil, base.Errf("%s: the covariance decomposed to the negative eigenvalue %g, not a covariance", name, v)
|
||||
}
|
||||
values[k] = 0
|
||||
}
|
||||
}
|
||||
// Column means, then the scores of every observation.
|
||||
means := make([]float64, p)
|
||||
rows := rawFloats(a)
|
||||
for j := range p {
|
||||
s := 0.0
|
||||
if rows != nil {
|
||||
for i := range n {
|
||||
s += rows[i*p+j]
|
||||
}
|
||||
} else {
|
||||
for i := range n {
|
||||
s += a.FloatAt(i*p + j)
|
||||
}
|
||||
}
|
||||
means[j] = s / float64(n)
|
||||
}
|
||||
scores := make([]float64, 0, n*p)
|
||||
for i := range n {
|
||||
for k := range p {
|
||||
s := 0.0
|
||||
vk := vectors[k]
|
||||
for j := range p {
|
||||
var xj float64
|
||||
if rows != nil {
|
||||
xj = rows[i*p+j]
|
||||
} else {
|
||||
xj = a.FloatAt(i*p + j)
|
||||
}
|
||||
s += (xj - means[j]) * vk[j]
|
||||
}
|
||||
scores = append(scores, s)
|
||||
}
|
||||
}
|
||||
// The loadings land in the documented (variable, component)
|
||||
// layout: entry (j, k) is component k's loading on variable j.
|
||||
loadings := make([]float64, 0, p*p)
|
||||
for j := range p {
|
||||
for k := range p {
|
||||
loadings = append(loadings, vectors[k][j])
|
||||
}
|
||||
}
|
||||
out := &PCAResult{
|
||||
Mean: means,
|
||||
Loadings: floatsToArray(loadings, []int{p, p}),
|
||||
Scores: floatsToArray(scores, []int{n, p}),
|
||||
ExplainedVariance: values,
|
||||
ExplainedVarianceRatio: make([]float64, p),
|
||||
}
|
||||
for k := range p {
|
||||
out.ExplainedVarianceRatio[k] = values[k] / total
|
||||
}
|
||||
if values[p-1] > 0 {
|
||||
whitening := make([]float64, 0, p*p)
|
||||
unwhitening := make([]float64, 0, p*p)
|
||||
for i := range p {
|
||||
for k := range p {
|
||||
whitening = append(whitening, vectors[k][i]/math.Sqrt(values[k]))
|
||||
}
|
||||
}
|
||||
for k := range p {
|
||||
for j := range p {
|
||||
unwhitening = append(unwhitening, math.Sqrt(values[k])*vectors[k][j])
|
||||
}
|
||||
}
|
||||
out.Whitening = floatsToArray(whitening, []int{p, p})
|
||||
out.Unwhitening = floatsToArray(unwhitening, []int{p, p})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Whiten maps the observations in x, an (n, p) array on the fit's own
|
||||
// variables, to their unit-covariance representation: the centred
|
||||
// observations expressed on the components and scaled by each
|
||||
// component's standard deviation. The result has the identity as its
|
||||
// covariance, up to the fit's rank: a rank-deficient fit refuses to
|
||||
// whiten rather than divide by an empty component's variance.
|
||||
func (r *PCAResult) Whiten(x *core.Array) (*core.Array, error) {
|
||||
const name = "Whiten"
|
||||
if r == nil {
|
||||
return nil, base.Errf("%s: no fit to whiten with", name)
|
||||
}
|
||||
if r.Whitening == nil {
|
||||
return nil, base.Errf("%s: the fit's covariance is rank deficient and no whitening transform exists", name)
|
||||
}
|
||||
if x.NDim() != 2 {
|
||||
return nil, base.Errf("%s: the observations must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
|
||||
}
|
||||
if x.Dtype() == core.Complex {
|
||||
return nil, base.Errf("%s: complex observations are not supported", name)
|
||||
}
|
||||
if err := checkFinite(name, "the observations", x); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
n, p := x.Shape()[0], len(r.Mean)
|
||||
if x.Shape()[1] != p {
|
||||
return nil, base.Errf("%s: the observations have %d columns, the fit %d", name, x.Shape()[1], p)
|
||||
}
|
||||
rows := rawFloats(x)
|
||||
out := core.New(core.Float, n, p)
|
||||
vals := out.RawFloats()
|
||||
// The transform is read once into a plain slice: the values are the
|
||||
// ones FloatAt returns.
|
||||
transform := make([]float64, p*p)
|
||||
if fs := rawFloats(r.Whitening); fs != nil {
|
||||
copy(transform, fs)
|
||||
} else {
|
||||
for i := range transform {
|
||||
transform[i] = r.Whitening.FloatAt(i)
|
||||
}
|
||||
}
|
||||
for i := range n {
|
||||
for k := range p {
|
||||
s := 0.0
|
||||
for j := range p {
|
||||
var xj float64
|
||||
if rows != nil {
|
||||
xj = rows[i*p+j]
|
||||
} else {
|
||||
xj = x.FloatAt(i*p + j)
|
||||
}
|
||||
s += (xj - r.Mean[j]) * transform[j*p+k]
|
||||
}
|
||||
vals[i*p+k] = s
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Unwhiten inverts Whiten: it maps a unit-covariance representation
|
||||
// back to the observations' own coordinates, the observations
|
||||
// themselves recovered exactly to rounding.
|
||||
func (r *PCAResult) Unwhiten(z *core.Array) (*core.Array, error) {
|
||||
const name = "Unwhiten"
|
||||
if r == nil {
|
||||
return nil, base.Errf("%s: no fit to unwhiten with", name)
|
||||
}
|
||||
if r.Unwhitening == nil {
|
||||
return nil, base.Errf("%s: the fit's covariance is rank deficient and no unwhitening transform exists", name)
|
||||
}
|
||||
if z.NDim() != 2 {
|
||||
return nil, base.Errf("%s: the whitened observations must be rank 2, got shape %s", name, base.ShapeText(z.Shape()))
|
||||
}
|
||||
if z.Dtype() == core.Complex {
|
||||
return nil, base.Errf("%s: complex observations are not supported", name)
|
||||
}
|
||||
if err := checkFinite(name, "the whitened observations", z); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
n, p := z.Shape()[0], len(r.Mean)
|
||||
if z.Shape()[1] != p {
|
||||
return nil, base.Errf("%s: the whitened observations have %d columns, the fit %d", name, z.Shape()[1], p)
|
||||
}
|
||||
rows := rawFloats(z)
|
||||
out := core.New(core.Float, n, p)
|
||||
vals := out.RawFloats()
|
||||
// The transform is read once into a plain slice: the values are the
|
||||
// ones FloatAt returns.
|
||||
transform := make([]float64, p*p)
|
||||
if fs := rawFloats(r.Unwhitening); fs != nil {
|
||||
copy(transform, fs)
|
||||
} else {
|
||||
for i := range transform {
|
||||
transform[i] = r.Unwhitening.FloatAt(i)
|
||||
}
|
||||
}
|
||||
for i := range n {
|
||||
for j := range p {
|
||||
s := 0.0
|
||||
for k := range p {
|
||||
var zk float64
|
||||
if rows != nil {
|
||||
zk = rows[i*p+k]
|
||||
} else {
|
||||
zk = z.FloatAt(i*p + k)
|
||||
}
|
||||
s += zk * transform[k*p+j]
|
||||
}
|
||||
vals[i*p+j] = r.Mean[j] + s
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// jacobiEigen eigendecomposes the symmetric matrix a by the cyclic
|
||||
// Jacobi iteration: every sweep applies the plane rotation that zeroes
|
||||
// each off-diagonal entry in turn, accumulating the rotations into the
|
||||
// eigenvector matrix. The iteration is unconditionally convergent for
|
||||
// symmetric matrices and quadratically convergent near the answer, so
|
||||
// a handful of sweeps carries any small dense matrix to machine
|
||||
// precision. The eigenvalues come back in falling order with the
|
||||
// matching orthonormal eigenvectors as columns, each column oriented
|
||||
// so its largest-magnitude entry is positive, the first index winning
|
||||
// a tie.
|
||||
func jacobiEigen(a [][]float64) (values []float64, vectors [][]float64) {
|
||||
p := len(a)
|
||||
m := make([][]float64, p)
|
||||
for i := range p {
|
||||
m[i] = append([]float64(nil), a[i]...)
|
||||
}
|
||||
v := make([][]float64, p)
|
||||
for i := range p {
|
||||
v[i] = make([]float64, p)
|
||||
v[i][i] = 1
|
||||
}
|
||||
scale := 0.0
|
||||
for i := range p {
|
||||
for j := i; j < p; j++ {
|
||||
scale += m[i][j] * m[i][j]
|
||||
}
|
||||
}
|
||||
for range 100 {
|
||||
off := 0.0
|
||||
for i := range p {
|
||||
for j := i + 1; j < p; j++ {
|
||||
off += m[i][j] * m[i][j]
|
||||
}
|
||||
}
|
||||
if math.Sqrt(off) <= 1e-14*math.Sqrt(scale) {
|
||||
break
|
||||
}
|
||||
for q := 1; q < p; q++ {
|
||||
for i := 0; i < q; i++ {
|
||||
aiq := m[i][q]
|
||||
if aiq == 0 {
|
||||
continue
|
||||
}
|
||||
theta := (m[q][q] - m[i][i]) / (2 * aiq)
|
||||
sign := 1.0
|
||||
if theta < 0 {
|
||||
sign = -1
|
||||
}
|
||||
t := sign / (math.Abs(theta) + math.Sqrt(theta*theta+1))
|
||||
c := 1 / math.Sqrt(t*t+1)
|
||||
s := t * c
|
||||
tangent := s / (1 + c)
|
||||
aii := m[i][i]
|
||||
aqq := m[q][q]
|
||||
m[i][i] = aii - t*aiq
|
||||
m[q][q] = aqq + t*aiq
|
||||
m[i][q] = 0
|
||||
m[q][i] = 0
|
||||
for k := range p {
|
||||
if k == i || k == q {
|
||||
continue
|
||||
}
|
||||
aki := m[k][i]
|
||||
akq := m[k][q]
|
||||
m[k][i] = aki - s*(akq+tangent*aki)
|
||||
m[i][k] = m[k][i]
|
||||
m[k][q] = akq + s*(aki-tangent*akq)
|
||||
m[q][k] = m[k][q]
|
||||
}
|
||||
for k := range p {
|
||||
vki := v[k][i]
|
||||
vkq := v[k][q]
|
||||
v[k][i] = vki - s*(vkq+tangent*vki)
|
||||
v[k][q] = vkq + s*(vki-tangent*vkq)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
values = make([]float64, p)
|
||||
for i := range p {
|
||||
values[i] = m[i][i]
|
||||
}
|
||||
// Falling order, ties left in their original order.
|
||||
order := make([]int, p)
|
||||
for i := range p {
|
||||
order[i] = i
|
||||
}
|
||||
slices.SortStableFunc(order, func(a, b int) int {
|
||||
return cmp.Compare(values[b], values[a])
|
||||
})
|
||||
sortedValues := make([]float64, p)
|
||||
sortedVectors := make([][]float64, p)
|
||||
for k := range p {
|
||||
sortedValues[k] = values[order[k]]
|
||||
sortedVectors[k] = make([]float64, p)
|
||||
for i := range p {
|
||||
sortedVectors[k][i] = v[i][order[k]]
|
||||
}
|
||||
}
|
||||
fixEigenSigns(sortedVectors)
|
||||
return sortedValues, sortedVectors
|
||||
}
|
||||
|
||||
// fixEigenSigns orients every eigenvector, held as a row of v, so its
|
||||
// largest-magnitude entry is positive, the first index winning a tie.
|
||||
// An eigendecomposition is defined only up to each eigenvector's sign,
|
||||
// and a decomposition that reports different signs on the same input
|
||||
// twice would be no decomposition at all.
|
||||
func fixEigenSigns(v [][]float64) {
|
||||
for k := range v {
|
||||
worst := 0.0
|
||||
index := 0
|
||||
for i := range v[k] {
|
||||
if a := math.Abs(v[k][i]); a > worst {
|
||||
worst = a
|
||||
index = i
|
||||
}
|
||||
}
|
||||
if v[k][index] < 0 {
|
||||
for i := range v[k] {
|
||||
v[k][i] = -v[k][i]
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user