Files
tensor/stats/glm.go
T
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

700 lines
22 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package stats
import (
"math"
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
// Generalised linear models. The logistic regression fits a binary
// response through the logit link by Newton-Raphson on the exact
// likelihood, which is the iteratively reweighted least squares the
// literature names, and reports Wald inference from the observed
// Fisher information.
// LogisticRegressionResult carries the fit of a binary response.
type LogisticRegressionResult struct {
// Coefficients are the maximum-likelihood estimates β̂ on the
// logit scale.
Coefficients []float64
// StandardErrors are the Wald standard errors from the inverse
// Fisher information at the optimum.
StandardErrors []float64
// ZStatistics are β̂/SE per coefficient.
ZStatistics []float64
// PValues are the two-sided normal-tail probabilities.
PValues []float64
// Fitted holds the predicted probability for every sample, clamped
// into [1e-12, 1 − 1e-12] exactly as the fitting loop clamps it, so
// a far-out covariate saturates the probability without turning the
// likelihood into a logarithm of zero.
Fitted []float64
// LogLikelihood is the maximised Bernoulli log likelihood.
LogLikelihood float64
// Iterations counts the Newton steps taken; Converged reports
// whether the coefficient update fell under the tolerance.
Iterations int
Converged bool
}
// logisticProbability returns the Bernoulli probability of the linear
// predictor eta, clamped away from the saturating ends: the weights and
// the logarithms the fit evaluates both need a live value there, so the
// clamp is what keeps a far-out covariate from driving the reported
// likelihood to a NaN. The fitting loop and the final pass share it, so
// the reported Fitted values are the ones the loop maximised.
func logisticProbability(eta float64) float64 {
pr := 1 / (1 + math.Exp(-eta))
if pr < 1e-12 {
pr = 1e-12
}
if pr > 1-1e-12 {
pr = 1 - 1e-12
}
return pr
}
// PoissonRegressionResult carries the fit of a count response.
type PoissonRegressionResult struct {
// Coefficients are the maximum-likelihood estimates β̂ on the log
// scale.
Coefficients []float64
// StandardErrors are the Wald standard errors from the inverse
// Fisher information at the optimum.
StandardErrors []float64
// ZStatistics are β̂/SE per coefficient.
ZStatistics []float64
// PValues are the two-sided normal-tail probabilities.
PValues []float64
// Fitted holds the predicted mean count for every sample, clamped
// into [1e-12, 1e300] exactly as the fitting loop clamps it, so a
// far-out covariate saturates the mean without turning the
// likelihood into a logarithm of zero or an infinity.
Fitted []float64
// LogLikelihood is the maximised Poisson log likelihood, evaluated
// on the clamped Fitted values.
LogLikelihood float64
// Iterations counts the Newton steps taken; Converged reports
// whether the coefficient update fell under the tolerance.
Iterations int
Converged bool
}
// mirrorUpper fills the lower triangle of a symmetric matrix from the
// upper one. Each mirrored entry accumulates exactly the product chain
// the upper entry did: the row-wise product commutes bitwise and both
// entries sum the rows in the same order, so today's direct
// accumulation already holds equal bits on both sides of the diagonal
// and the copy reproduces them.
func mirrorUpper(m [][]float64) {
for i := range m {
for j := range i {
m[i][j] = m[j][i]
}
}
}
// poissonMean returns the Poisson mean of the linear predictor eta,
// clamped away from the values where the fit cannot keep going: the
// exponential overflows above η ≈ 709.78 and underflows to zero below
// η ≈ −745, and both the y·log μ term and the Fisher weights need a
// live finite mean there. The fitting loop and the final pass share
// it, so the reported Fitted values are the ones the loop maximised.
func poissonMean(eta float64) float64 {
const ceiling = 1e300
mu := math.Exp(eta)
if math.IsInf(mu, 1) || mu > ceiling {
mu = ceiling
}
if mu < 1e-12 {
mu = 1e-12
}
return mu
}
// PoissonRegression fits y = Poisson(exp(X·β)) by maximum likelihood.
// The design carries n rows and p columns exactly as LinearRegression's
// (intercept included by the caller as a constant column when wanted),
// y holds non-negative integer counts, and the fit runs Newton-Raphson
// until the largest coefficient update drops under 1e-10, halving any
// step that does not raise the likelihood: the unbounded Poisson
// weights let an undamped step overshoot into oscillation. On the
// canonical log link the observed information equals the Fisher
// information, so this is the iteratively reweighted least squares the
// literature names, with the weights equal to the means. A design
// whose Fisher information is singular, a duplicated column among
// them, is reported as an error; data that drives the iteration
// without settling exhausts the iteration budget and is reported
// rather than returned as a diverged fit.
func PoissonRegression(x, y *core.Array) (*PoissonRegressionResult, error) {
const name = "PoissonRegression"
if x.NDim() != 2 {
return nil, base.Errf("%s: the design must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
}
if y.NDim() != 1 {
return nil, base.Errf("%s: the response must be rank 1", name)
}
if x.Dtype() == core.Complex || y.Dtype() == core.Complex {
return nil, base.Errf("%s: complex inputs are not supported", name)
}
n, p := x.Shape()[0], x.Shape()[1]
if y.Len() != n {
return nil, base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len())
}
if n <= p {
return nil, base.Errf("%s: need n > p, got %d observations and %d columns", name, n, p)
}
if p == 0 {
return nil, base.Errf("%s: the design must carry at least one column", name)
}
// A non-finite design would flow through the exponential and the
// Newton step into a "converged" all-NaN fit: as in
// LinearRegression, non-finite input has no answer to report.
if err := checkFinite(name, "the design", x); err != nil {
return nil, err
}
for i := range y.Len() {
v := y.FloatAt(i)
if math.IsNaN(v) || math.IsInf(v, 0) {
return nil, base.Errf("%s: response sample %d is not finite", name, i)
}
if v < 0 {
return nil, base.Errf("%s: response sample %d is %g, want a non-negative count", name, i, v)
}
if v != math.Trunc(v) {
return nil, base.Errf("%s: response sample %d is %g, want an integer count", name, i, v)
}
}
beta := make([]float64, p)
mean := make([]float64, n)
// The design and the response are read through raw payload slices
// when dense: the elements are the ones FloatAt returns, so every
// product and sum below keeps its exact operand bits.
fx := rawFloats(x)
fy := rawFloats(y)
// The log likelihood without the constant log(y!): every candidate
// point pays the same constant, so the comparison the step damping
// makes needs only this part. It reads the mean through the same
// clamped exponential the fit maximises.
logLikeAt := func(b []float64) float64 {
total := 0.0
for r := range n {
eta := 0.0
if fx != nil {
row := fx[r*p : r*p+p]
for j, xj := range row {
eta += xj * b[j]
}
} else {
for j := range p {
eta += x.FloatAt(r*p+j) * b[j]
}
}
mu := poissonMean(eta)
var yv float64
if fy != nil {
yv = fy[r]
} else {
yv = y.FloatAt(r)
}
total += yv*math.Log(mu) - mu
}
return total
}
const maxIter = 100
converged := false
iterations := maxIter
// The normal-equation buffers are allocated once and cleared per
// iteration: the accumulation adds into them, so every pass starts
// from an explicitly zeroed state, the one a fresh allocation had.
fisher := make([][]float64, p)
for i := range p {
fisher[i] = make([]float64, p)
}
gradient := make([]float64, p)
applied := make([]float64, p)
for iter := 1; iter <= maxIter; iter++ {
currentLogLike := 0.0
for r := range n {
eta := 0.0
if fx != nil {
row := fx[r*p : r*p+p]
for j, xj := range row {
eta += xj * beta[j]
}
} else {
for j := range p {
eta += x.FloatAt(r*p+j) * beta[j]
}
}
mean[r] = poissonMean(eta)
var yv float64
if fy != nil {
yv = fy[r]
} else {
yv = y.FloatAt(r)
}
currentLogLike += yv*math.Log(mean[r]) - mean[r]
}
// The Fisher information is symmetric and each lower-triangle
// entry equals its upper twin bit for bit (mirrorUpper), so the
// accumulation runs the upper triangle alone and mirrors it once.
for i := range p {
clear(fisher[i])
}
clear(gradient)
for r := range n {
mu := mean[r]
var yv float64
if fy != nil {
yv = fy[r]
} else {
yv = y.FloatAt(r)
}
if fx != nil {
row := fx[r*p : r*p+p]
for i, xr := range row {
gradient[i] += xr * (yv - mu)
// Upper triangle, both operands pre-sliced from i:
// the same products in the same order, bounds checks
// elided.
fi := fisher[i][i:]
for j, xj := range row[i:] {
fi[j] += xr * xj * mu
}
}
} else {
for i := range p {
xr := x.FloatAt(r*p + i)
gradient[i] += xr * (yv - mu)
for j := i; j < p; j++ {
fisher[i][j] += xr * x.FloatAt(r*p+j) * mu
}
}
}
}
mirrorUpper(fisher)
step, err := base.SolveSystem(name, fisher, [][]float64{gradient})
if err != nil {
return nil, base.Errf("%s: the Fisher information is singular (%w)", name, err)
}
// Backtracking on the log likelihood: unlike the logistic
// weights, the Poisson ones are unbounded, so an undamped step
// can overshoot into the clamped means where the next step
// explodes and the iteration oscillates instead of converging.
// The step is halved while the likelihood does not rise, the
// same damping the root finder applies to its residual norm.
// The acceptance is >=, the standard Armijo condition: a flat
// likelihood must accept the step rather than spend the halving
// budget shrinking it into what only looks like convergence.
worst := 0.0
damping := 1.0
for halving := 0; ; halving++ {
worst = 0.0
for j := range p {
applied[j] = beta[j] + damping*step[0][j]
if d := math.Abs(damping * step[0][j]); d > worst {
worst = d
}
}
if logLikeAt(applied) >= currentLogLike || halving == 30 {
break
}
damping /= 2
}
copy(beta, applied)
if worst < 1e-10 {
converged = true
iterations = iter
// One final pass for the fitted means at the settled
// coefficients, clamped exactly as the loop clamped them:
// the unclamped exponential overflows to an infinity for
// |eta| past the log of the ceiling, and the likelihood
// below is evaluated on these values.
for r := range n {
eta := 0.0
if fx != nil {
row := fx[r*p : r*p+p]
for j, xj := range row {
eta += xj * beta[j]
}
} else {
for j := range p {
eta += x.FloatAt(r*p+j) * beta[j]
}
}
mean[r] = poissonMean(eta)
}
break
}
}
if !converged {
return nil, base.Errf("%s: %d iterations did not converge", name, maxIter)
}
logLike := 0.0
for r := range n {
var yv float64
if fy != nil {
yv = fy[r]
} else {
yv = y.FloatAt(r)
}
logGamma, _ := math.Lgamma(yv + 1)
logLike += yv*math.Log(mean[r]) - mean[r] - logGamma
}
// Wald inference from the inverse Fisher information at the
// optimum. The iteration's last Fisher matrix belongs to the
// previous point, one damped step behind, so it is rebuilt from
// the settled means before the solves, into the reused buffer and
// on the same mirrored upper triangle.
for i := range p {
clear(fisher[i])
}
for r := range n {
mu := mean[r]
if fx != nil {
row := fx[r*p : r*p+p]
for i, xr := range row {
fi := fisher[i][i:]
for j, xj := range row[i:] {
fi[j] += xr * xj * mu
}
}
} else {
for i := range p {
xr := x.FloatAt(r*p + i)
for j := i; j < p; j++ {
fisher[i][j] += xr * x.FloatAt(r*p+j) * mu
}
}
}
}
mirrorUpper(fisher)
out := &PoissonRegressionResult{
Coefficients: beta,
Fitted: mean,
LogLikelihood: logLike,
Iterations: iterations,
Converged: true,
}
out.StandardErrors = make([]float64, p)
out.ZStatistics = make([]float64, p)
out.PValues = make([]float64, p)
// One unit vector per coefficient, all solved through a single
// factorisation of the Fisher information: the per-coefficient
// solves refactored the same matrix p times, while the shared solve
// substitutes each column through the identical factor.
unit := make([][]float64, p)
for j := range p {
unit[j] = make([]float64, p)
unit[j][j] = 1
}
inv, err := base.SolveSystem(name, fisher, unit)
if err != nil {
return nil, base.Errf("%s: %w", name, err)
}
for j := range p {
// The Wald variance is the diagonal of the inverse Fisher
// information. A near-collinear design can drive the solve to a
// tiny negative diagonal entry through rounding alone: the bare
// square root would then be a NaN reported beside a nil error.
// An exact zero stays a zero standard error; a negative entry
// means the design is near-collinear and is refused.
v := inv[j][j]
switch {
case v > 0:
out.StandardErrors[j] = math.Sqrt(v)
case v == 0:
out.StandardErrors[j] = 0
default:
return nil, base.Errf("%s: the design is near-collinear: the Wald variance of coefficient %d came out negative (%g)", name, j, v)
}
if out.StandardErrors[j] == 0 {
// A zero Wald variance: an exact fit reports total evidence
// for a live coefficient and nothing to test for a zero
// one, never the 0/0 NaN pair.
if beta[j] != 0 {
out.ZStatistics[j] = math.Copysign(math.Inf(1), beta[j])
out.PValues[j] = 0
} else {
out.ZStatistics[j] = 0
out.PValues[j] = 1
}
continue
}
out.ZStatistics[j] = beta[j] / out.StandardErrors[j]
z := out.ZStatistics[j]
// The two-sided normal tail in one Erfc call on the magnitude.
// The algebraic form 2·(1−Φ(z)) cancels to exactly zero once z
// passes about 8.3, where the true tail nears 1e-17 and keeps
// another three hundred orders below before the smallest
// float64.
out.PValues[j] = math.Erfc(math.Abs(z) / math.Sqrt2)
}
return out, nil
}
// LogisticRegression fits y = Bernoulli(sigmoid(X·β)) by maximum
// likelihood. The design carries n rows and p columns exactly as
// LinearRegression's (intercept included by the caller as a constant
// column when wanted), y holds zeros and ones, and the fit runs
// Newton-Raphson until the largest coefficient update drops under
// 1e-10. Perfectly separable data has no finite optimum: the run
// reports an error rather than diverging coefficients.
func LogisticRegression(x, y *core.Array) (*LogisticRegressionResult, error) {
const name = "LogisticRegression"
if x.NDim() != 2 {
return nil, base.Errf("%s: the design must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
}
if y.NDim() != 1 {
return nil, base.Errf("%s: the response must be rank 1", name)
}
if x.Dtype() == core.Complex || y.Dtype() == core.Complex {
return nil, base.Errf("%s: complex inputs are not supported", name)
}
n, p := x.Shape()[0], x.Shape()[1]
if y.Len() != n {
return nil, base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len())
}
if n <= p {
return nil, base.Errf("%s: need n > p, got %d observations and %d columns", name, n, p)
}
if p == 0 {
return nil, base.Errf("%s: the design must carry at least one column", name)
}
// A non-finite design would flow through the sigmoid and the
// Newton step into a "converged" all-NaN fit: as in
// LinearRegression, non-finite input has no answer to report.
if err := checkFinite(name, "the design", x); err != nil {
return nil, err
}
for i := range y.Len() {
v := y.FloatAt(i)
if v != 0 && v != 1 {
return nil, base.Errf("%s: response sample %d is %g, want 0 or 1", name, i, v)
}
}
beta := make([]float64, p)
prob := make([]float64, n)
// The design and the response are read through raw payload slices
// when dense: the elements are the ones FloatAt returns, so every
// product and sum below keeps its exact operand bits.
fx := rawFloats(x)
fy := rawFloats(y)
const maxIter = 100
converged := false
iterations := maxIter
// The normal-equation buffers are allocated once and cleared per
// iteration; every accumulation pass starts from the zero state a
// fresh allocation carried.
hessian := make([][]float64, p)
for i := range p {
hessian[i] = make([]float64, p)
}
gradient := make([]float64, p)
for iter := 1; iter <= maxIter; iter++ {
for r := range n {
eta := 0.0
if fx != nil {
row := fx[r*p : r*p+p]
for j, xj := range row {
eta += xj * beta[j]
}
} else {
for j := range p {
eta += x.FloatAt(r*p+j) * beta[j]
}
}
// The sigmoid clamped away from its saturating ends: the
// weights and the log both need a live second derivative.
pr := logisticProbability(eta)
prob[r] = pr
}
// The observed information is symmetric and each lower-triangle
// entry equals its upper twin bit for bit (mirrorUpper), so the
// accumulation runs the upper triangle alone and mirrors it once.
for i := range p {
clear(hessian[i])
}
clear(gradient)
for r := range n {
pr := prob[r]
w := pr * (1 - pr)
var yv float64
if fy != nil {
yv = fy[r]
} else {
yv = y.FloatAt(r)
}
if fx != nil {
row := fx[r*p : r*p+p]
for i, xr := range row {
gradient[i] += xr * (yv - pr)
// Upper triangle, both operands pre-sliced from i:
// the same products in the same order, bounds checks
// elided.
hi := hessian[i][i:]
for j, xj := range row[i:] {
hi[j] += xr * xj * w
}
}
} else {
for i := range p {
xr := x.FloatAt(r*p + i)
gradient[i] += xr * (yv - pr)
for j := i; j < p; j++ {
hessian[i][j] += xr * x.FloatAt(r*p+j) * w
}
}
}
}
mirrorUpper(hessian)
step, err := base.SolveSystem(name, hessian, [][]float64{gradient})
if err != nil {
return nil, base.Errf("%s: the Fisher information is singular (%w)", name, err)
}
worst := 0.0
for j := range p {
beta[j] += step[0][j]
if math.Abs(step[0][j]) > worst {
worst = math.Abs(step[0][j])
}
}
if worst < 1e-10 {
converged = true
iterations = iter
// One final pass for the fitted probabilities at the
// settled coefficients, clamped exactly as the loop
// clamped them: the unclamped form reaches exactly 0 and 1
// for |eta| > ~37, and the likelihood below is evaluated on
// these values.
for r := range n {
eta := 0.0
if fx != nil {
row := fx[r*p : r*p+p]
for j, xj := range row {
eta += xj * beta[j]
}
} else {
for j := range p {
eta += x.FloatAt(r*p+j) * beta[j]
}
}
prob[r] = logisticProbability(eta)
}
break
}
}
if !converged {
return nil, base.Errf("%s: %d iterations did not converge; the response may be perfectly separable", name, maxIter)
}
// Wald inference from the inverse Fisher information at the
// optimum.
//
// The matrix is built row by row into the reused buffer, the way
// the fitting loop builds its own: a row streams the design once
// instead of once per coefficient, and the row's weight is formed
// once. Each entry sums its products over the rows in ascending
// order on the mirrored upper triangle, so the entries are the ones
// the direct walk accumulated.
fisher := hessian
for i := range p {
clear(fisher[i])
}
for r := range n {
w := prob[r] * (1 - prob[r])
if fx != nil {
row := fx[r*p : r*p+p]
for i, xi := range row {
fi := fisher[i][i:]
for j, xj := range row[i:] {
fi[j] += xi * xj * w
}
}
} else {
for i := range p {
xi := x.FloatAt(r*p + i)
fi := fisher[i]
for j := i; j < p; j++ {
fi[j] += xi * x.FloatAt(r*p+j) * w
}
}
}
}
mirrorUpper(fisher)
out := &LogisticRegressionResult{
Coefficients: beta,
Fitted: prob,
Iterations: iterations,
Converged: true,
}
logLike := 0.0
for r := range n {
var yv float64
if fy != nil {
yv = fy[r]
} else {
yv = y.FloatAt(r)
}
logLike += yv*math.Log(prob[r]) + (1-yv)*math.Log(1-prob[r])
}
out.LogLikelihood = logLike
out.StandardErrors = make([]float64, p)
out.ZStatistics = make([]float64, p)
out.PValues = make([]float64, p)
// One unit vector per coefficient, all solved through a single
// factorisation of the Fisher information: the per-coefficient
// solves refactored the same matrix p times, while the shared solve
// substitutes each column through the identical factor.
unit := make([][]float64, p)
for j := range p {
unit[j] = make([]float64, p)
unit[j][j] = 1
}
inv, err := base.SolveSystem(name, fisher, unit)
if err != nil {
return nil, base.Errf("%s: %w", name, err)
}
for j := range p {
// The Wald variance is the diagonal of the inverse Fisher
// information. A near-collinear design can drive the solve to a
// tiny negative diagonal entry through rounding alone: the bare
// square root would then be a NaN reported beside a nil error.
// An exact zero stays a zero standard error; a negative entry
// means the design is near-collinear and is refused.
v := inv[j][j]
switch {
case v > 0:
out.StandardErrors[j] = math.Sqrt(v)
case v == 0:
out.StandardErrors[j] = 0
default:
return nil, base.Errf("%s: the design is near-collinear: the Wald variance of coefficient %d came out negative (%g)", name, j, v)
}
if out.StandardErrors[j] == 0 {
// A zero Wald variance: an exact fit reports total evidence
// for a live coefficient and nothing to test for a zero
// one, never the 0/0 NaN pair.
if beta[j] != 0 {
out.ZStatistics[j] = math.Copysign(math.Inf(1), beta[j])
out.PValues[j] = 0
} else {
out.ZStatistics[j] = 0
out.PValues[j] = 1
}
continue
}
out.ZStatistics[j] = beta[j] / out.StandardErrors[j]
z := out.ZStatistics[j]
// The two-sided normal tail in one Erfc call on the magnitude,
// as in PoissonRegression: 2·(1−Φ(z)) cancels to exactly zero
// once z passes about 8.3.
out.PValues[j] = math.Erfc(math.Abs(z) / math.Sqrt2)
}
return out, nil
}