// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: MIT package stats import ( "cmp" "math" "slices" "sourcedock.dev/petrbalvin/tensor/internal/base" "sourcedock.dev/petrbalvin/tensor/internal/core" ) // Principal component analysis: the rotation of a cloud of // observations onto the axes along which it actually spreads, with the // variances of those axes, the coordinates of every observation on // them, and the whitening transform that flattens the cloud to unit // covariance. The covariance comes from the package's own // CovarianceMatrix; its eigendecomposition is computed here, because // the package's linear algebra has so far only needed LU solves, and a // symmetric eigensolver is carried as a cyclic Jacobi iteration: // unconditionally convergent on symmetric matrices, quadratically so // near the answer, and exactly the right instrument for the small // dense covariance matrices a PCA runs on. // PCAResult carries the decomposition of an observation array. type PCAResult struct { // Mean holds the column means of the observations the fit ran on. Mean []float64 // Loadings is the (p, p) rotation: entry (j, k) is the loading of // variable j on component k, columns ordered by falling explained // variance, orthonormal as columns. Every column's largest-magnitude // loading is positive, the first index winning a tie, so the axes // have a fixed orientation and two runs on the same data agree // sign for sign. Loadings *core.Array // Scores is the (n, p) array of the observations' coordinates on // the components: the centred observations times the loadings. Scores *core.Array // ExplainedVariance holds each component's eigenvalue of the // covariance, in the loadings' order; ExplainedVarianceRatio // divides it by the total variance, so the ratios sum to one. ExplainedVariance []float64 ExplainedVarianceRatio []float64 // Whitening and Unwhitening are the (p, p) transforms between the // centred observations and the unit-covariance representation: // whitening maps a centred row x to x·Whitening, unwhitening maps // it back. Both are nil when the covariance is rank deficient, // where no whitening transform exists. Whitening *core.Array Unwhitening *core.Array } // PCA decomposes the observations in a, an (n, p) array whose rows are // observations and columns variables, onto their principal // components. At least two observations are needed, the input must be // real and finite, and data with no variance at all has no // decomposition to report. A rank-deficient covariance decomposes // normally, with zero variances on the empty components; only the // whitening transforms, which would divide by those variances, are // withheld. func PCA(a *core.Array) (*PCAResult, error) { const name = "PCA" covArr, err := CovarianceMatrix(a) if err != nil { return nil, base.Errf("%s: %w", name, err) } n, p := a.Shape()[0], a.Shape()[1] // The covariance is read once through its payload where it has one; // the values are the ones FloatAt returns. covVals := rawFloats(covArr) cov := make([][]float64, p) total := 0.0 for i := range p { cov[i] = make([]float64, p) if covVals != nil { copy(cov[i], covVals[i*p:i*p+p]) } else { for j := range p { cov[i][j] = covArr.FloatAt(i*p + j) } } total += cov[i][i] } if total == 0 { return nil, base.Errf("%s: the observations have no variance", name) } values, vectors := jacobiEigen(cov) // A covariance is positive semidefinite by construction, so a // negative eigenvalue beyond a rounding-scale fraction of the // largest means the decomposition cannot be trusted; anything // within it is rounding and clamps to an exactly empty component. worst := 0.0 for _, v := range values { worst = max(worst, math.Abs(v)) } for k, v := range values { if v < 0 { if v < -1e-9*worst { return nil, base.Errf("%s: the covariance decomposed to the negative eigenvalue %g, not a covariance", name, v) } values[k] = 0 } } // Column means, then the scores of every observation. means := make([]float64, p) rows := rawFloats(a) for j := range p { s := 0.0 if rows != nil { for i := range n { s += rows[i*p+j] } } else { for i := range n { s += a.FloatAt(i*p + j) } } means[j] = s / float64(n) } scores := make([]float64, 0, n*p) for i := range n { for k := range p { s := 0.0 vk := vectors[k] for j := range p { var xj float64 if rows != nil { xj = rows[i*p+j] } else { xj = a.FloatAt(i*p + j) } s += (xj - means[j]) * vk[j] } scores = append(scores, s) } } // The loadings land in the documented (variable, component) // layout: entry (j, k) is component k's loading on variable j. loadings := make([]float64, 0, p*p) for j := range p { for k := range p { loadings = append(loadings, vectors[k][j]) } } out := &PCAResult{ Mean: means, Loadings: floatsToArray(loadings, []int{p, p}), Scores: floatsToArray(scores, []int{n, p}), ExplainedVariance: values, ExplainedVarianceRatio: make([]float64, p), } for k := range p { out.ExplainedVarianceRatio[k] = values[k] / total } if values[p-1] > 0 { whitening := make([]float64, 0, p*p) unwhitening := make([]float64, 0, p*p) for i := range p { for k := range p { whitening = append(whitening, vectors[k][i]/math.Sqrt(values[k])) } } for k := range p { for j := range p { unwhitening = append(unwhitening, math.Sqrt(values[k])*vectors[k][j]) } } out.Whitening = floatsToArray(whitening, []int{p, p}) out.Unwhitening = floatsToArray(unwhitening, []int{p, p}) } return out, nil } // Whiten maps the observations in x, an (n, p) array on the fit's own // variables, to their unit-covariance representation: the centred // observations expressed on the components and scaled by each // component's standard deviation. The result has the identity as its // covariance, up to the fit's rank: a rank-deficient fit refuses to // whiten rather than divide by an empty component's variance. func (r *PCAResult) Whiten(x *core.Array) (*core.Array, error) { const name = "Whiten" if r == nil { return nil, base.Errf("%s: no fit to whiten with", name) } if r.Whitening == nil { return nil, base.Errf("%s: the fit's covariance is rank deficient and no whitening transform exists", name) } if x.NDim() != 2 { return nil, base.Errf("%s: the observations must be rank 2, got shape %s", name, base.ShapeText(x.Shape())) } if x.Dtype() == core.Complex { return nil, base.Errf("%s: complex observations are not supported", name) } if err := checkFinite(name, "the observations", x); err != nil { return nil, err } n, p := x.Shape()[0], len(r.Mean) if x.Shape()[1] != p { return nil, base.Errf("%s: the observations have %d columns, the fit %d", name, x.Shape()[1], p) } rows := rawFloats(x) out := core.New(core.Float, n, p) vals := out.RawFloats() // The transform is read once into a plain slice: the values are the // ones FloatAt returns. transform := make([]float64, p*p) if fs := rawFloats(r.Whitening); fs != nil { copy(transform, fs) } else { for i := range transform { transform[i] = r.Whitening.FloatAt(i) } } for i := range n { for k := range p { s := 0.0 for j := range p { var xj float64 if rows != nil { xj = rows[i*p+j] } else { xj = x.FloatAt(i*p + j) } s += (xj - r.Mean[j]) * transform[j*p+k] } vals[i*p+k] = s } } return out, nil } // Unwhiten inverts Whiten: it maps a unit-covariance representation // back to the observations' own coordinates, the observations // themselves recovered exactly to rounding. func (r *PCAResult) Unwhiten(z *core.Array) (*core.Array, error) { const name = "Unwhiten" if r == nil { return nil, base.Errf("%s: no fit to unwhiten with", name) } if r.Unwhitening == nil { return nil, base.Errf("%s: the fit's covariance is rank deficient and no unwhitening transform exists", name) } if z.NDim() != 2 { return nil, base.Errf("%s: the whitened observations must be rank 2, got shape %s", name, base.ShapeText(z.Shape())) } if z.Dtype() == core.Complex { return nil, base.Errf("%s: complex observations are not supported", name) } if err := checkFinite(name, "the whitened observations", z); err != nil { return nil, err } n, p := z.Shape()[0], len(r.Mean) if z.Shape()[1] != p { return nil, base.Errf("%s: the whitened observations have %d columns, the fit %d", name, z.Shape()[1], p) } rows := rawFloats(z) out := core.New(core.Float, n, p) vals := out.RawFloats() // The transform is read once into a plain slice: the values are the // ones FloatAt returns. transform := make([]float64, p*p) if fs := rawFloats(r.Unwhitening); fs != nil { copy(transform, fs) } else { for i := range transform { transform[i] = r.Unwhitening.FloatAt(i) } } for i := range n { for j := range p { s := 0.0 for k := range p { var zk float64 if rows != nil { zk = rows[i*p+k] } else { zk = z.FloatAt(i*p + k) } s += zk * transform[k*p+j] } vals[i*p+j] = r.Mean[j] + s } } return out, nil } // jacobiEigen eigendecomposes the symmetric matrix a by the cyclic // Jacobi iteration: every sweep applies the plane rotation that zeroes // each off-diagonal entry in turn, accumulating the rotations into the // eigenvector matrix. The iteration is unconditionally convergent for // symmetric matrices and quadratically convergent near the answer, so // a handful of sweeps carries any small dense matrix to machine // precision. The eigenvalues come back in falling order with the // matching orthonormal eigenvectors as columns, each column oriented // so its largest-magnitude entry is positive, the first index winning // a tie. func jacobiEigen(a [][]float64) (values []float64, vectors [][]float64) { p := len(a) m := make([][]float64, p) for i := range p { m[i] = append([]float64(nil), a[i]...) } v := make([][]float64, p) for i := range p { v[i] = make([]float64, p) v[i][i] = 1 } scale := 0.0 for i := range p { for j := i; j < p; j++ { scale += m[i][j] * m[i][j] } } for range 100 { off := 0.0 for i := range p { for j := i + 1; j < p; j++ { off += m[i][j] * m[i][j] } } if math.Sqrt(off) <= 1e-14*math.Sqrt(scale) { break } for q := 1; q < p; q++ { for i := 0; i < q; i++ { aiq := m[i][q] if aiq == 0 { continue } theta := (m[q][q] - m[i][i]) / (2 * aiq) sign := 1.0 if theta < 0 { sign = -1 } t := sign / (math.Abs(theta) + math.Sqrt(theta*theta+1)) c := 1 / math.Sqrt(t*t+1) s := t * c tangent := s / (1 + c) aii := m[i][i] aqq := m[q][q] m[i][i] = aii - t*aiq m[q][q] = aqq + t*aiq m[i][q] = 0 m[q][i] = 0 for k := range p { if k == i || k == q { continue } aki := m[k][i] akq := m[k][q] m[k][i] = aki - s*(akq+tangent*aki) m[i][k] = m[k][i] m[k][q] = akq + s*(aki-tangent*akq) m[q][k] = m[k][q] } for k := range p { vki := v[k][i] vkq := v[k][q] v[k][i] = vki - s*(vkq+tangent*vki) v[k][q] = vkq + s*(vki-tangent*vkq) } } } } values = make([]float64, p) for i := range p { values[i] = m[i][i] } // Falling order, ties left in their original order. order := make([]int, p) for i := range p { order[i] = i } slices.SortStableFunc(order, func(a, b int) int { return cmp.Compare(values[b], values[a]) }) sortedValues := make([]float64, p) sortedVectors := make([][]float64, p) for k := range p { sortedValues[k] = values[order[k]] sortedVectors[k] = make([]float64, p) for i := range p { sortedVectors[k][i] = v[i][order[k]] } } fixEigenSigns(sortedVectors) return sortedValues, sortedVectors } // fixEigenSigns orients every eigenvector, held as a row of v, so its // largest-magnitude entry is positive, the first index winning a tie. // An eigendecomposition is defined only up to each eigenvector's sign, // and a decomposition that reports different signs on the same input // twice would be no decomposition at all. func fixEigenSigns(v [][]float64) { for k := range v { worst := 0.0 index := 0 for i := range v[k] { if a := math.Abs(v[k][i]); a > worst { worst = a index = i } } if v[k][index] < 0 { for i := range v[k] { v[k][i] = -v[k][i] } } } }