// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: MIT package stats import ( "math" "sourcedock.dev/petrbalvin/tensor/internal/base" "sourcedock.dev/petrbalvin/tensor/internal/core" ) // Gaussian-process regression: a prior over functions fixed by a // covariance kernel, conditioned exactly on the observations through // the Cholesky factor of the Gram matrix, the same house route the // multivariate normal takes. The marginal log likelihood is exposed as // a plain function of the hyperparameters: the dependency graph gives // stats no edge to optim, so the fitting loop stays with the caller, // and the house minimiser (optim.Minimise) drives // MarginalLogLikelihood from outside the package over the two or three // hyperparameters a kernel carries. // Kernel is the covariance function of a Gaussian process: it returns // the prior covariance k(x, y) of the process at one pair of input // points, the points of equal dimension. Every house kernel carries a // unit amplitude: k(x, x) = 1, and the constructors below enforce the // parameter contract, a finite positive scale (and period where one // belongs), so a Kernel value is always safe to evaluate. type Kernel interface { // Covariance returns k(x, y) for two points of equal length. Covariance(x, y []float64) float64 } // gpKernelScale validates one finite positive kernel parameter, the // contract every constructor states. func gpKernelScale(name, label string, v float64) error { if !(v > 0) || math.IsInf(v, 0) { return base.Errf("%s: %s must be finite and positive, got %g", name, label, v) } return nil } // squaredExponential is the smooth limit kernel // k(r) = exp(-r²/(2·lengthScale²)), infinitely differentiable, the // default prior for functions believed smooth. type squaredExponential struct { lengthScale float64 } // SquaredExponentialKernel returns the squared-exponential (RBF) // kernel of unit amplitude over the Euclidean distance of the input // points, with the given length scale. func SquaredExponentialKernel(lengthScale float64) (Kernel, error) { const name = "SquaredExponentialKernel" if err := gpKernelScale(name, "the length scale", lengthScale); err != nil { return nil, err } return squaredExponential{lengthScale: lengthScale}, nil } // Covariance implements Kernel. func (k squaredExponential) Covariance(x, y []float64) float64 { r2 := sqDistance(x, y) return math.Exp(-r2 / (2 * k.lengthScale * k.lengthScale)) } // matern32 is the once differentiable Matérn with ν = 3/2, // k(r) = (1 + √3·r/lengthScale)·exp(-√3·r/lengthScale), the spectral // density of which is S(ω) = 4a³/(a² + ω²)² for a = √3/lengthScale: // the closed form above and that spectral form are a Fourier pair, // k(r) = (1/π)∫₀^∞ S(ω)·cos(ω·r) dω, the defining equation the tests // hold the closed form against. type matern32 struct { lengthScale float64 } // Matern32Kernel returns the Matérn kernel with ν = 3/2 and the given // length scale, of unit amplitude over the Euclidean distance. func Matern32Kernel(lengthScale float64) (Kernel, error) { const name = "Matern32Kernel" if err := gpKernelScale(name, "the length scale", lengthScale); err != nil { return nil, err } return matern32{lengthScale: lengthScale}, nil } // Covariance implements Kernel. func (k matern32) Covariance(x, y []float64) float64 { a := math.Sqrt(3) * math.Sqrt(sqDistance(x, y)) / k.lengthScale return (1 + a) * math.Exp(-a) } // matern52 is the twice differentiable Matérn with ν = 5/2, // k(r) = (1 + √5·r/lengthScale + 5r²/(3·lengthScale²))·exp(-√5·r/lengthScale), // the roughness the incidence of real fields usually lands between the // two lower Matérns and the smooth exponential square. type matern52 struct { lengthScale float64 } // Matern52Kernel returns the Matérn kernel with ν = 5/2 and the given // length scale, of unit amplitude over the Euclidean distance. func Matern52Kernel(lengthScale float64) (Kernel, error) { const name = "Matern52Kernel" if err := gpKernelScale(name, "the length scale", lengthScale); err != nil { return nil, err } return matern52{lengthScale: lengthScale}, nil } // Covariance implements Kernel. func (k matern52) Covariance(x, y []float64) float64 { r := math.Sqrt(sqDistance(x, y)) a := math.Sqrt(5) * r / k.lengthScale return (1 + a + a*a/3) * math.Exp(-a) } // periodic is the periodic kernel // k(r) = exp(-2·sin²(π·r/period)/lengthScale²), the prior over // functions that repeat exactly with the given period, the length // scale setting how sharply neighbouring periods decorrelate. type periodic struct { lengthScale float64 period float64 } // PeriodicKernel returns the periodic kernel with the given length // scale and period, of unit amplitude over the Euclidean distance. func PeriodicKernel(lengthScale, period float64) (Kernel, error) { const name = "PeriodicKernel" if err := gpKernelScale(name, "the length scale", lengthScale); err != nil { return nil, err } if err := gpKernelScale(name, "the period", period); err != nil { return nil, err } return periodic{lengthScale: lengthScale, period: period}, nil } // Covariance implements Kernel. func (k periodic) Covariance(x, y []float64) float64 { r := math.Sqrt(sqDistance(x, y)) s := math.Sin(math.Pi * r / k.period) return math.Exp(-2 * s * s / (k.lengthScale * k.lengthScale)) } // GaussianProcessResult carries the posterior of a Gaussian process at // the test points. type GaussianProcessResult struct { // Mean is the posterior mean function evaluated at the test // points. Mean []float64 // Covariance is the posterior covariance between the test // points, m-by-m row-major in the order the test points were // given, the full uncertainty the posterior carries. Covariance []float64 // Variance is the diagonal of Covariance, the marginal posterior // variance per test point, clamped at zero: rounding can push a // training-point diagonal a hair below zero where the truth is // zero, and a negative variance reports nothing. Variance []float64 // LogLikelihood is the log marginal likelihood of the training // observations under the prior, the objective a hyperparameter // fit maximises (see MarginalLogLikelihood). LogLikelihood float64 } // GaussianProcessRegression conditions the prior defined by the // kernel on the n noisy observations y over the n training rows of // trainX (n rows, d columns), and evaluates the posterior mean and // covariance at the m rows of testX. The noise variance sits on the // Gram diagonal, K = k(X, X) + noiseVariance·I: zero is a legitimate // noiseless fit, and with it the posterior interpolates the training // data exactly and has zero variance there, while duplicated training // rows name the singular row in an error. The posterior algebra is the // Cholesky route: the Gram matrix factors through the multivariate // normal machinery of mvn.go, the weights alpha solve the triangular // systems against it, and the predictive covariance subtracts the // forward-solved cross covariances from the prior. // // The inputs must be real and finite, the training and test designs // of equal width, y of the training length, and the noise variance // finite and non-negative. func GaussianProcessRegression(kernel Kernel, trainX, trainY *core.Array, noiseVariance float64, testX *core.Array) (*GaussianProcessResult, error) { const name = "GaussianProcessRegression" l, alpha, train, n, d, err := gpFit(name, kernel, trainX, trainY, noiseVariance) if err != nil { return nil, err } test, m, testD, err := gpReadDesign(name, "the test design", testX) if err != nil { return nil, err } if testD != d { return nil, base.Errf("%s: the training design is %d wide but the test design %d", name, d, testD) } out := &GaussianProcessResult{ Mean: make([]float64, m), Covariance: make([]float64, m*m), Variance: make([]float64, m), } // The cross covariances A: A[i][j] = k(test i, train j), and the // forward-solved V = L⁻¹Aᵀ whose rows pair off against each other // in the predictive covariance. cross := make([][]float64, m) solved := make([][]float64, m) for i := range m { cross[i] = make([]float64, n) for j := range n { cross[i][j] = kernel.Covariance(test[i*d:i*d+d], train[j*d:j*d+d]) } solved[i] = gpForwardSolve(l, cross[i]) } for i := range m { total := 0.0 for j := range n { total += cross[i][j] * alpha[j] } out.Mean[i] = total for j := i; j < m; j++ { // The prior term is the kernel's own diagonal entry, // k(x*, x*), which the unit-amplitude house kernels fix at // 1 but a custom kernel may set otherwise. prior := kernel.Covariance(test[j*d:j*d+d], test[j*d:j*d+d]) if j > i { prior = kernel.Covariance(test[i*d:i*d+d], test[j*d:j*d+d]) } cov := prior for r := range n { cov -= solved[i][r] * solved[j][r] } out.Covariance[i*m+j] = cov out.Covariance[j*m+i] = cov } diag := out.Covariance[i*m+i] if diag < 0 { diag = 0 } out.Variance[i] = diag } out.LogLikelihood = gpLogLikelihood(l, alpha, trainY, n) return out, nil } // MarginalLogLikelihood returns the log marginal likelihood of the // observations y under the prior the kernel defines over the training // design: the evidence of the hyperparameters, integrating the // training responses against their multivariate normal prior. It is // the objective a hyperparameter fit maximises; the dependency graph // gives stats no edge to optim, so the house minimiser drives this // function from outside the package, over the kernel's scale (and // period) and the noise variance. func MarginalLogLikelihood(kernel Kernel, trainX, trainY *core.Array, noiseVariance float64) (float64, error) { const name = "MarginalLogLikelihood" l, alpha, _, n, _, err := gpFit(name, kernel, trainX, trainY, noiseVariance) if err != nil { return 0, err } return gpLogLikelihood(l, alpha, trainY, n), nil } // gpFit validates the training inputs and returns the Cholesky factor // of the regularised Gram matrix, the solved weights alpha and the // flattened training design. func gpFit(name string, kernel Kernel, trainX, trainY *core.Array, noiseVariance float64) ([][]float64, []float64, []float64, int, int, error) { if kernel == nil { return nil, nil, nil, 0, 0, base.Errf("%s: the kernel is nil", name) } if math.IsNaN(noiseVariance) || math.IsInf(noiseVariance, 0) || noiseVariance < 0 { return nil, nil, nil, 0, 0, base.Errf("%s: the noise variance must be finite and non-negative, got %g", name, noiseVariance) } train, n, d, err := gpReadDesign(name, "the training design", trainX) if err != nil { return nil, nil, nil, 0, 0, err } if err := gpReadResponse(name, trainY, n); err != nil { return nil, nil, nil, 0, 0, err } // The Gram matrix, computed on the lower triangle and mirrored // exactly, so the factorisation's symmetry check reads two // bit-identical halves. It is written straight into one flat // row-major slice, the layout the factorisation reads, instead of // through a row-of-rows form another pass would flatten. gram := make([]float64, n*n) for i := range n { for j := range i + 1 { v := kernel.Covariance(train[i*d:i*d+d], train[j*d:j*d+d]) if i == j { v += noiseVariance } gram[i*n+j] = v gram[j*n+i] = v } } l, err := mvnCholeskyFlat(name, gram, n) if err != nil { return nil, nil, nil, 0, 0, base.Errf("%s: %w", name, err) } y := make([]float64, n) fy := rawFloats(trainY) for i := range n { if fy != nil { y[i] = fy[i] } else { y[i] = trainY.FloatAt(i) } } alpha := gpBackSolve(l, gpForwardSolve(l, y)) return l, alpha, train, n, d, nil } // gpLogLikelihood assembles the log marginal likelihood from the // factor and the solved weights: // -½·yᵀ·alpha - Σ ln Lᵢᵢ - (n/2)·ln 2π. func gpLogLikelihood(l [][]float64, alpha []float64, trainY *core.Array, n int) float64 { fy := rawFloats(trainY) total := 0.0 for i := range n { var yv float64 if fy != nil { yv = fy[i] } else { yv = trainY.FloatAt(i) } total -= 0.5 * yv * alpha[i] total -= math.Log(l[i][i]) } return total - 0.5*float64(n)*math.Log(2*math.Pi) } // gpReadDesign validates a finite real rank-2 design and returns it // flattened row-major. func gpReadDesign(name, label string, x *core.Array) ([]float64, int, int, error) { if x.NDim() != 2 { return nil, 0, 0, base.Errf("%s: %s must be rank 2, got shape %s", name, label, base.ShapeText(x.Shape())) } if x.Dtype() == core.Complex { return nil, 0, 0, base.Errf("%s: complex inputs are not supported", name) } if err := checkFinite(name, label, x); err != nil { return nil, 0, 0, err } n, d := x.Shape()[0], x.Shape()[1] if d < 1 { return nil, 0, 0, base.Errf("%s: %s needs at least one column", name, label) } data := make([]float64, n*d) if fs := rawFloats(x); fs != nil { copy(data, fs) } else { for i := range data { data[i] = x.FloatAt(i) } } return data, n, d, nil } // gpReadResponse validates the response vector against the training // length. func gpReadResponse(name string, y *core.Array, n int) error { if y.NDim() != 1 { return base.Errf("%s: the response must be rank 1", name) } if y.Dtype() == core.Complex { return base.Errf("%s: complex inputs are not supported", name) } if y.Len() != n { return base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len()) } return checkFinite(name, "the response", y) } // gpForwardSolve solves L·v = b for the lower triangular L. func gpForwardSolve(l [][]float64, b []float64) []float64 { v := make([]float64, len(b)) for i := range b { total := b[i] for j := range i { total -= l[i][j] * v[j] } v[i] = total / l[i][i] } return v } // gpBackSolve solves Lᵀ·u = b for the lower triangular L. func gpBackSolve(l [][]float64, b []float64) []float64 { n := len(b) u := make([]float64, n) for i := n - 1; i >= 0; i-- { total := b[i] for j := i + 1; j < n; j++ { total -= l[j][i] * u[j] } u[i] = total / l[i][i] } return u }