Files
tensor/stats/gp.go
T
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

399 lines
14 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package stats
import (
"math"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
// Gaussian-process regression: a prior over functions fixed by a
// covariance kernel, conditioned exactly on the observations through
// the Cholesky factor of the Gram matrix, the same house route the
// multivariate normal takes. The marginal log likelihood is exposed as
// a plain function of the hyperparameters: the dependency graph gives
// stats no edge to optim, so the fitting loop stays with the caller,
// and the house minimiser (optim.Minimise) drives
// MarginalLogLikelihood from outside the package over the two or three
// hyperparameters a kernel carries.
// Kernel is the covariance function of a Gaussian process: it returns
// the prior covariance k(x, y) of the process at one pair of input
// points, the points of equal dimension. Every house kernel carries a
// unit amplitude: k(x, x) = 1, and the constructors below enforce the
// parameter contract, a finite positive scale (and period where one
// belongs), so a Kernel value is always safe to evaluate.
type Kernel interface {
// Covariance returns k(x, y) for two points of equal length.
Covariance(x, y []float64) float64
}
// gpKernelScale validates one finite positive kernel parameter, the
// contract every constructor states.
func gpKernelScale(name, label string, v float64) error {
if !(v > 0) || math.IsInf(v, 0) {
return base.Errf("%s: %s must be finite and positive, got %g", name, label, v)
}
return nil
}
// squaredExponential is the smooth limit kernel
// k(r) = exp(-r²/(2·lengthScale²)), infinitely differentiable, the
// default prior for functions believed smooth.
type squaredExponential struct {
lengthScale float64
}
// SquaredExponentialKernel returns the squared-exponential (RBF)
// kernel of unit amplitude over the Euclidean distance of the input
// points, with the given length scale.
func SquaredExponentialKernel(lengthScale float64) (Kernel, error) {
const name = "SquaredExponentialKernel"
if err := gpKernelScale(name, "the length scale", lengthScale); err != nil {
return nil, err
}
return squaredExponential{lengthScale: lengthScale}, nil
}
// Covariance implements Kernel.
func (k squaredExponential) Covariance(x, y []float64) float64 {
r2 := sqDistance(x, y)
return math.Exp(-r2 / (2 * k.lengthScale * k.lengthScale))
}
// matern32 is the once differentiable Matérn with ν = 3/2,
// k(r) = (1 + √3·r/lengthScale)·exp(-√3·r/lengthScale), the spectral
// density of which is S(ω) = 4a³/(a² + ω²)² for a = √3/lengthScale:
// the closed form above and that spectral form are a Fourier pair,
// k(r) = (1/π)∫₀^∞ S(ω)·cos(ω·r) dω, the defining equation the tests
// hold the closed form against.
type matern32 struct {
lengthScale float64
}
// Matern32Kernel returns the Matérn kernel with ν = 3/2 and the given
// length scale, of unit amplitude over the Euclidean distance.
func Matern32Kernel(lengthScale float64) (Kernel, error) {
const name = "Matern32Kernel"
if err := gpKernelScale(name, "the length scale", lengthScale); err != nil {
return nil, err
}
return matern32{lengthScale: lengthScale}, nil
}
// Covariance implements Kernel.
func (k matern32) Covariance(x, y []float64) float64 {
a := math.Sqrt(3) * math.Sqrt(sqDistance(x, y)) / k.lengthScale
return (1 + a) * math.Exp(-a)
}
// matern52 is the twice differentiable Matérn with ν = 5/2,
// k(r) = (1 + √5·r/lengthScale + 5r²/(3·lengthScale²))·exp(-√5·r/lengthScale),
// the roughness the incidence of real fields usually lands between the
// two lower Matérns and the smooth exponential square.
type matern52 struct {
lengthScale float64
}
// Matern52Kernel returns the Matérn kernel with ν = 5/2 and the given
// length scale, of unit amplitude over the Euclidean distance.
func Matern52Kernel(lengthScale float64) (Kernel, error) {
const name = "Matern52Kernel"
if err := gpKernelScale(name, "the length scale", lengthScale); err != nil {
return nil, err
}
return matern52{lengthScale: lengthScale}, nil
}
// Covariance implements Kernel.
func (k matern52) Covariance(x, y []float64) float64 {
r := math.Sqrt(sqDistance(x, y))
a := math.Sqrt(5) * r / k.lengthScale
return (1 + a + a*a/3) * math.Exp(-a)
}
// periodic is the periodic kernel
// k(r) = exp(-2·sin²(π·r/period)/lengthScale²), the prior over
// functions that repeat exactly with the given period, the length
// scale setting how sharply neighbouring periods decorrelate.
type periodic struct {
lengthScale float64
period float64
}
// PeriodicKernel returns the periodic kernel with the given length
// scale and period, of unit amplitude over the Euclidean distance.
func PeriodicKernel(lengthScale, period float64) (Kernel, error) {
const name = "PeriodicKernel"
if err := gpKernelScale(name, "the length scale", lengthScale); err != nil {
return nil, err
}
if err := gpKernelScale(name, "the period", period); err != nil {
return nil, err
}
return periodic{lengthScale: lengthScale, period: period}, nil
}
// Covariance implements Kernel.
func (k periodic) Covariance(x, y []float64) float64 {
r := math.Sqrt(sqDistance(x, y))
s := math.Sin(math.Pi * r / k.period)
return math.Exp(-2 * s * s / (k.lengthScale * k.lengthScale))
}
// GaussianProcessResult carries the posterior of a Gaussian process at
// the test points.
type GaussianProcessResult struct {
// Mean is the posterior mean function evaluated at the test
// points.
Mean []float64
// Covariance is the posterior covariance between the test
// points, m-by-m row-major in the order the test points were
// given, the full uncertainty the posterior carries.
Covariance []float64
// Variance is the diagonal of Covariance, the marginal posterior
// variance per test point, clamped at zero: rounding can push a
// training-point diagonal a hair below zero where the truth is
// zero, and a negative variance reports nothing.
Variance []float64
// LogLikelihood is the log marginal likelihood of the training
// observations under the prior, the objective a hyperparameter
// fit maximises (see MarginalLogLikelihood).
LogLikelihood float64
}
// GaussianProcessRegression conditions the prior defined by the
// kernel on the n noisy observations y over the n training rows of
// trainX (n rows, d columns), and evaluates the posterior mean and
// covariance at the m rows of testX. The noise variance sits on the
// Gram diagonal, K = k(X, X) + noiseVariance·I: zero is a legitimate
// noiseless fit, and with it the posterior interpolates the training
// data exactly and has zero variance there, while duplicated training
// rows name the singular row in an error. The posterior algebra is the
// Cholesky route: the Gram matrix factors through the multivariate
// normal machinery of mvn.go, the weights alpha solve the triangular
// systems against it, and the predictive covariance subtracts the
// forward-solved cross covariances from the prior.
//
// The inputs must be real and finite, the training and test designs
// of equal width, y of the training length, and the noise variance
// finite and non-negative.
func GaussianProcessRegression(kernel Kernel, trainX, trainY *core.Array, noiseVariance float64, testX *core.Array) (*GaussianProcessResult, error) {
const name = "GaussianProcessRegression"
l, alpha, train, n, d, err := gpFit(name, kernel, trainX, trainY, noiseVariance)
if err != nil {
return nil, err
}
test, m, testD, err := gpReadDesign(name, "the test design", testX)
if err != nil {
return nil, err
}
if testD != d {
return nil, base.Errf("%s: the training design is %d wide but the test design %d", name, d, testD)
}
out := &GaussianProcessResult{
Mean: make([]float64, m),
Covariance: make([]float64, m*m),
Variance: make([]float64, m),
}
// The cross covariances A: A[i][j] = k(test i, train j), and the
// forward-solved V = L⁻¹Aᵀ whose rows pair off against each other
// in the predictive covariance.
cross := make([][]float64, m)
solved := make([][]float64, m)
for i := range m {
cross[i] = make([]float64, n)
for j := range n {
cross[i][j] = kernel.Covariance(test[i*d:i*d+d], train[j*d:j*d+d])
}
solved[i] = gpForwardSolve(l, cross[i])
}
for i := range m {
total := 0.0
for j := range n {
total += cross[i][j] * alpha[j]
}
out.Mean[i] = total
for j := i; j < m; j++ {
// The prior term is the kernel's own diagonal entry,
// k(x*, x*), which the unit-amplitude house kernels fix at
// 1 but a custom kernel may set otherwise.
prior := kernel.Covariance(test[j*d:j*d+d], test[j*d:j*d+d])
if j > i {
prior = kernel.Covariance(test[i*d:i*d+d], test[j*d:j*d+d])
}
cov := prior
for r := range n {
cov -= solved[i][r] * solved[j][r]
}
out.Covariance[i*m+j] = cov
out.Covariance[j*m+i] = cov
}
diag := out.Covariance[i*m+i]
if diag < 0 {
diag = 0
}
out.Variance[i] = diag
}
out.LogLikelihood = gpLogLikelihood(l, alpha, trainY, n)
return out, nil
}
// MarginalLogLikelihood returns the log marginal likelihood of the
// observations y under the prior the kernel defines over the training
// design: the evidence of the hyperparameters, integrating the
// training responses against their multivariate normal prior. It is
// the objective a hyperparameter fit maximises; the dependency graph
// gives stats no edge to optim, so the house minimiser drives this
// function from outside the package, over the kernel's scale (and
// period) and the noise variance.
func MarginalLogLikelihood(kernel Kernel, trainX, trainY *core.Array, noiseVariance float64) (float64, error) {
const name = "MarginalLogLikelihood"
l, alpha, _, n, _, err := gpFit(name, kernel, trainX, trainY, noiseVariance)
if err != nil {
return 0, err
}
return gpLogLikelihood(l, alpha, trainY, n), nil
}
// gpFit validates the training inputs and returns the Cholesky factor
// of the regularised Gram matrix, the solved weights alpha and the
// flattened training design.
func gpFit(name string, kernel Kernel, trainX, trainY *core.Array, noiseVariance float64) ([][]float64, []float64, []float64, int, int, error) {
if kernel == nil {
return nil, nil, nil, 0, 0, base.Errf("%s: the kernel is nil", name)
}
if math.IsNaN(noiseVariance) || math.IsInf(noiseVariance, 0) || noiseVariance < 0 {
return nil, nil, nil, 0, 0, base.Errf("%s: the noise variance must be finite and non-negative, got %g", name, noiseVariance)
}
train, n, d, err := gpReadDesign(name, "the training design", trainX)
if err != nil {
return nil, nil, nil, 0, 0, err
}
if err := gpReadResponse(name, trainY, n); err != nil {
return nil, nil, nil, 0, 0, err
}
// The Gram matrix, computed on the lower triangle and mirrored
// exactly, so the factorisation's symmetry check reads two
// bit-identical halves. It is written straight into one flat
// row-major slice, the layout the factorisation reads, instead of
// through a row-of-rows form another pass would flatten.
gram := make([]float64, n*n)
for i := range n {
for j := range i + 1 {
v := kernel.Covariance(train[i*d:i*d+d], train[j*d:j*d+d])
if i == j {
v += noiseVariance
}
gram[i*n+j] = v
gram[j*n+i] = v
}
}
l, err := mvnCholeskyFlat(name, gram, n)
if err != nil {
return nil, nil, nil, 0, 0, base.Errf("%s: %w", name, err)
}
y := make([]float64, n)
fy := rawFloats(trainY)
for i := range n {
if fy != nil {
y[i] = fy[i]
} else {
y[i] = trainY.FloatAt(i)
}
}
alpha := gpBackSolve(l, gpForwardSolve(l, y))
return l, alpha, train, n, d, nil
}
// gpLogLikelihood assembles the log marginal likelihood from the
// factor and the solved weights:
// -½·yᵀ·alpha - Σ ln Lᵢᵢ - (n/2)·ln 2π.
func gpLogLikelihood(l [][]float64, alpha []float64, trainY *core.Array, n int) float64 {
fy := rawFloats(trainY)
total := 0.0
for i := range n {
var yv float64
if fy != nil {
yv = fy[i]
} else {
yv = trainY.FloatAt(i)
}
total -= 0.5 * yv * alpha[i]
total -= math.Log(l[i][i])
}
return total - 0.5*float64(n)*math.Log(2*math.Pi)
}
// gpReadDesign validates a finite real rank-2 design and returns it
// flattened row-major.
func gpReadDesign(name, label string, x *core.Array) ([]float64, int, int, error) {
if x.NDim() != 2 {
return nil, 0, 0, base.Errf("%s: %s must be rank 2, got shape %s", name, label, base.ShapeText(x.Shape()))
}
if x.Dtype() == core.Complex {
return nil, 0, 0, base.Errf("%s: complex inputs are not supported", name)
}
if err := checkFinite(name, label, x); err != nil {
return nil, 0, 0, err
}
n, d := x.Shape()[0], x.Shape()[1]
if d < 1 {
return nil, 0, 0, base.Errf("%s: %s needs at least one column", name, label)
}
data := make([]float64, n*d)
if fs := rawFloats(x); fs != nil {
copy(data, fs)
} else {
for i := range data {
data[i] = x.FloatAt(i)
}
}
return data, n, d, nil
}
// gpReadResponse validates the response vector against the training
// length.
func gpReadResponse(name string, y *core.Array, n int) error {
if y.NDim() != 1 {
return base.Errf("%s: the response must be rank 1", name)
}
if y.Dtype() == core.Complex {
return base.Errf("%s: complex inputs are not supported", name)
}
if y.Len() != n {
return base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len())
}
return checkFinite(name, "the response", y)
}
// gpForwardSolve solves L·v = b for the lower triangular L.
func gpForwardSolve(l [][]float64, b []float64) []float64 {
v := make([]float64, len(b))
for i := range b {
total := b[i]
for j := range i {
total -= l[i][j] * v[j]
}
v[i] = total / l[i][i]
}
return v
}
// gpBackSolve solves Lᵀ·u = b for the lower triangular L.
func gpBackSolve(l [][]float64, b []float64) []float64 {
n := len(b)
u := make([]float64, n)
for i := n - 1; i >= 0; i-- {
total := b[i]
for j := i + 1; j < n; j++ {
total -= l[j][i] * u[j]
}
u[i] = total / l[i][i]
}
return u
}