700 lines
22 KiB
Go
700 lines
22 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
|
|
// SPDX-License-Identifier: MIT
|
|||
|
|
|
|||
|
|
package stats
|
|||
|
|
|
|||
|
|
import (
|
|||
|
|
"math"
|
|||
|
|
|
|||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
|||
|
|
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
// Generalised linear models. The logistic regression fits a binary
|
|||
|
|
// response through the logit link by Newton-Raphson on the exact
|
|||
|
|
// likelihood, which is the iteratively reweighted least squares the
|
|||
|
|
// literature names, and reports Wald inference from the observed
|
|||
|
|
// Fisher information.
|
|||
|
|
|
|||
|
|
// LogisticRegressionResult carries the fit of a binary response.
|
|||
|
|
type LogisticRegressionResult struct {
|
|||
|
|
// Coefficients are the maximum-likelihood estimates β̂ on the
|
|||
|
|
// logit scale.
|
|||
|
|
Coefficients []float64
|
|||
|
|
// StandardErrors are the Wald standard errors from the inverse
|
|||
|
|
// Fisher information at the optimum.
|
|||
|
|
StandardErrors []float64
|
|||
|
|
// ZStatistics are β̂/SE per coefficient.
|
|||
|
|
ZStatistics []float64
|
|||
|
|
// PValues are the two-sided normal-tail probabilities.
|
|||
|
|
PValues []float64
|
|||
|
|
// Fitted holds the predicted probability for every sample, clamped
|
|||
|
|
// into [1e-12, 1 − 1e-12] exactly as the fitting loop clamps it, so
|
|||
|
|
// a far-out covariate saturates the probability without turning the
|
|||
|
|
// likelihood into a logarithm of zero.
|
|||
|
|
Fitted []float64
|
|||
|
|
// LogLikelihood is the maximised Bernoulli log likelihood.
|
|||
|
|
LogLikelihood float64
|
|||
|
|
// Iterations counts the Newton steps taken; Converged reports
|
|||
|
|
// whether the coefficient update fell under the tolerance.
|
|||
|
|
Iterations int
|
|||
|
|
Converged bool
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// logisticProbability returns the Bernoulli probability of the linear
|
|||
|
|
// predictor eta, clamped away from the saturating ends: the weights and
|
|||
|
|
// the logarithms the fit evaluates both need a live value there, so the
|
|||
|
|
// clamp is what keeps a far-out covariate from driving the reported
|
|||
|
|
// likelihood to a NaN. The fitting loop and the final pass share it, so
|
|||
|
|
// the reported Fitted values are the ones the loop maximised.
|
|||
|
|
func logisticProbability(eta float64) float64 {
|
|||
|
|
pr := 1 / (1 + math.Exp(-eta))
|
|||
|
|
if pr < 1e-12 {
|
|||
|
|
pr = 1e-12
|
|||
|
|
}
|
|||
|
|
if pr > 1-1e-12 {
|
|||
|
|
pr = 1 - 1e-12
|
|||
|
|
}
|
|||
|
|
return pr
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// PoissonRegressionResult carries the fit of a count response.
|
|||
|
|
type PoissonRegressionResult struct {
|
|||
|
|
// Coefficients are the maximum-likelihood estimates β̂ on the log
|
|||
|
|
// scale.
|
|||
|
|
Coefficients []float64
|
|||
|
|
// StandardErrors are the Wald standard errors from the inverse
|
|||
|
|
// Fisher information at the optimum.
|
|||
|
|
StandardErrors []float64
|
|||
|
|
// ZStatistics are β̂/SE per coefficient.
|
|||
|
|
ZStatistics []float64
|
|||
|
|
// PValues are the two-sided normal-tail probabilities.
|
|||
|
|
PValues []float64
|
|||
|
|
// Fitted holds the predicted mean count for every sample, clamped
|
|||
|
|
// into [1e-12, 1e300] exactly as the fitting loop clamps it, so a
|
|||
|
|
// far-out covariate saturates the mean without turning the
|
|||
|
|
// likelihood into a logarithm of zero or an infinity.
|
|||
|
|
Fitted []float64
|
|||
|
|
// LogLikelihood is the maximised Poisson log likelihood, evaluated
|
|||
|
|
// on the clamped Fitted values.
|
|||
|
|
LogLikelihood float64
|
|||
|
|
// Iterations counts the Newton steps taken; Converged reports
|
|||
|
|
// whether the coefficient update fell under the tolerance.
|
|||
|
|
Iterations int
|
|||
|
|
Converged bool
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// mirrorUpper fills the lower triangle of a symmetric matrix from the
|
|||
|
|
// upper one. Each mirrored entry accumulates exactly the product chain
|
|||
|
|
// the upper entry did: the row-wise product commutes bitwise and both
|
|||
|
|
// entries sum the rows in the same order, so today's direct
|
|||
|
|
// accumulation already holds equal bits on both sides of the diagonal
|
|||
|
|
// and the copy reproduces them.
|
|||
|
|
func mirrorUpper(m [][]float64) {
|
|||
|
|
for i := range m {
|
|||
|
|
for j := range i {
|
|||
|
|
m[i][j] = m[j][i]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// poissonMean returns the Poisson mean of the linear predictor eta,
|
|||
|
|
// clamped away from the values where the fit cannot keep going: the
|
|||
|
|
// exponential overflows above η ≈ 709.78 and underflows to zero below
|
|||
|
|
// η ≈ −745, and both the y·log μ term and the Fisher weights need a
|
|||
|
|
// live finite mean there. The fitting loop and the final pass share
|
|||
|
|
// it, so the reported Fitted values are the ones the loop maximised.
|
|||
|
|
func poissonMean(eta float64) float64 {
|
|||
|
|
const ceiling = 1e300
|
|||
|
|
mu := math.Exp(eta)
|
|||
|
|
if math.IsInf(mu, 1) || mu > ceiling {
|
|||
|
|
mu = ceiling
|
|||
|
|
}
|
|||
|
|
if mu < 1e-12 {
|
|||
|
|
mu = 1e-12
|
|||
|
|
}
|
|||
|
|
return mu
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// PoissonRegression fits y = Poisson(exp(X·β)) by maximum likelihood.
|
|||
|
|
// The design carries n rows and p columns exactly as LinearRegression's
|
|||
|
|
// (intercept included by the caller as a constant column when wanted),
|
|||
|
|
// y holds non-negative integer counts, and the fit runs Newton-Raphson
|
|||
|
|
// until the largest coefficient update drops under 1e-10, halving any
|
|||
|
|
// step that does not raise the likelihood: the unbounded Poisson
|
|||
|
|
// weights let an undamped step overshoot into oscillation. On the
|
|||
|
|
// canonical log link the observed information equals the Fisher
|
|||
|
|
// information, so this is the iteratively reweighted least squares the
|
|||
|
|
// literature names, with the weights equal to the means. A design
|
|||
|
|
// whose Fisher information is singular, a duplicated column among
|
|||
|
|
// them, is reported as an error; data that drives the iteration
|
|||
|
|
// without settling exhausts the iteration budget and is reported
|
|||
|
|
// rather than returned as a diverged fit.
|
|||
|
|
func PoissonRegression(x, y *core.Array) (*PoissonRegressionResult, error) {
|
|||
|
|
const name = "PoissonRegression"
|
|||
|
|
if x.NDim() != 2 {
|
|||
|
|
return nil, base.Errf("%s: the design must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
|
|||
|
|
}
|
|||
|
|
if y.NDim() != 1 {
|
|||
|
|
return nil, base.Errf("%s: the response must be rank 1", name)
|
|||
|
|
}
|
|||
|
|
if x.Dtype() == core.Complex || y.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("%s: complex inputs are not supported", name)
|
|||
|
|
}
|
|||
|
|
n, p := x.Shape()[0], x.Shape()[1]
|
|||
|
|
if y.Len() != n {
|
|||
|
|
return nil, base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len())
|
|||
|
|
}
|
|||
|
|
if n <= p {
|
|||
|
|
return nil, base.Errf("%s: need n > p, got %d observations and %d columns", name, n, p)
|
|||
|
|
}
|
|||
|
|
if p == 0 {
|
|||
|
|
return nil, base.Errf("%s: the design must carry at least one column", name)
|
|||
|
|
}
|
|||
|
|
// A non-finite design would flow through the exponential and the
|
|||
|
|
// Newton step into a "converged" all-NaN fit: as in
|
|||
|
|
// LinearRegression, non-finite input has no answer to report.
|
|||
|
|
if err := checkFinite(name, "the design", x); err != nil {
|
|||
|
|
return nil, err
|
|||
|
|
}
|
|||
|
|
for i := range y.Len() {
|
|||
|
|
v := y.FloatAt(i)
|
|||
|
|
if math.IsNaN(v) || math.IsInf(v, 0) {
|
|||
|
|
return nil, base.Errf("%s: response sample %d is not finite", name, i)
|
|||
|
|
}
|
|||
|
|
if v < 0 {
|
|||
|
|
return nil, base.Errf("%s: response sample %d is %g, want a non-negative count", name, i, v)
|
|||
|
|
}
|
|||
|
|
if v != math.Trunc(v) {
|
|||
|
|
return nil, base.Errf("%s: response sample %d is %g, want an integer count", name, i, v)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
beta := make([]float64, p)
|
|||
|
|
mean := make([]float64, n)
|
|||
|
|
// The design and the response are read through raw payload slices
|
|||
|
|
// when dense: the elements are the ones FloatAt returns, so every
|
|||
|
|
// product and sum below keeps its exact operand bits.
|
|||
|
|
fx := rawFloats(x)
|
|||
|
|
fy := rawFloats(y)
|
|||
|
|
// The log likelihood without the constant log(y!): every candidate
|
|||
|
|
// point pays the same constant, so the comparison the step damping
|
|||
|
|
// makes needs only this part. It reads the mean through the same
|
|||
|
|
// clamped exponential the fit maximises.
|
|||
|
|
logLikeAt := func(b []float64) float64 {
|
|||
|
|
total := 0.0
|
|||
|
|
for r := range n {
|
|||
|
|
eta := 0.0
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for j, xj := range row {
|
|||
|
|
eta += xj * b[j]
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for j := range p {
|
|||
|
|
eta += x.FloatAt(r*p+j) * b[j]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mu := poissonMean(eta)
|
|||
|
|
var yv float64
|
|||
|
|
if fy != nil {
|
|||
|
|
yv = fy[r]
|
|||
|
|
} else {
|
|||
|
|
yv = y.FloatAt(r)
|
|||
|
|
}
|
|||
|
|
total += yv*math.Log(mu) - mu
|
|||
|
|
}
|
|||
|
|
return total
|
|||
|
|
}
|
|||
|
|
const maxIter = 100
|
|||
|
|
converged := false
|
|||
|
|
iterations := maxIter
|
|||
|
|
// The normal-equation buffers are allocated once and cleared per
|
|||
|
|
// iteration: the accumulation adds into them, so every pass starts
|
|||
|
|
// from an explicitly zeroed state, the one a fresh allocation had.
|
|||
|
|
fisher := make([][]float64, p)
|
|||
|
|
for i := range p {
|
|||
|
|
fisher[i] = make([]float64, p)
|
|||
|
|
}
|
|||
|
|
gradient := make([]float64, p)
|
|||
|
|
applied := make([]float64, p)
|
|||
|
|
for iter := 1; iter <= maxIter; iter++ {
|
|||
|
|
currentLogLike := 0.0
|
|||
|
|
for r := range n {
|
|||
|
|
eta := 0.0
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for j, xj := range row {
|
|||
|
|
eta += xj * beta[j]
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for j := range p {
|
|||
|
|
eta += x.FloatAt(r*p+j) * beta[j]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mean[r] = poissonMean(eta)
|
|||
|
|
var yv float64
|
|||
|
|
if fy != nil {
|
|||
|
|
yv = fy[r]
|
|||
|
|
} else {
|
|||
|
|
yv = y.FloatAt(r)
|
|||
|
|
}
|
|||
|
|
currentLogLike += yv*math.Log(mean[r]) - mean[r]
|
|||
|
|
}
|
|||
|
|
// The Fisher information is symmetric and each lower-triangle
|
|||
|
|
// entry equals its upper twin bit for bit (mirrorUpper), so the
|
|||
|
|
// accumulation runs the upper triangle alone and mirrors it once.
|
|||
|
|
for i := range p {
|
|||
|
|
clear(fisher[i])
|
|||
|
|
}
|
|||
|
|
clear(gradient)
|
|||
|
|
for r := range n {
|
|||
|
|
mu := mean[r]
|
|||
|
|
var yv float64
|
|||
|
|
if fy != nil {
|
|||
|
|
yv = fy[r]
|
|||
|
|
} else {
|
|||
|
|
yv = y.FloatAt(r)
|
|||
|
|
}
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for i, xr := range row {
|
|||
|
|
gradient[i] += xr * (yv - mu)
|
|||
|
|
// Upper triangle, both operands pre-sliced from i:
|
|||
|
|
// the same products in the same order, bounds checks
|
|||
|
|
// elided.
|
|||
|
|
fi := fisher[i][i:]
|
|||
|
|
for j, xj := range row[i:] {
|
|||
|
|
fi[j] += xr * xj * mu
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for i := range p {
|
|||
|
|
xr := x.FloatAt(r*p + i)
|
|||
|
|
gradient[i] += xr * (yv - mu)
|
|||
|
|
for j := i; j < p; j++ {
|
|||
|
|
fisher[i][j] += xr * x.FloatAt(r*p+j) * mu
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mirrorUpper(fisher)
|
|||
|
|
step, err := base.SolveSystem(name, fisher, [][]float64{gradient})
|
|||
|
|
if err != nil {
|
|||
|
|
return nil, base.Errf("%s: the Fisher information is singular (%w)", name, err)
|
|||
|
|
}
|
|||
|
|
// Backtracking on the log likelihood: unlike the logistic
|
|||
|
|
// weights, the Poisson ones are unbounded, so an undamped step
|
|||
|
|
// can overshoot into the clamped means where the next step
|
|||
|
|
// explodes and the iteration oscillates instead of converging.
|
|||
|
|
// The step is halved while the likelihood does not rise, the
|
|||
|
|
// same damping the root finder applies to its residual norm.
|
|||
|
|
// The acceptance is >=, the standard Armijo condition: a flat
|
|||
|
|
// likelihood must accept the step rather than spend the halving
|
|||
|
|
// budget shrinking it into what only looks like convergence.
|
|||
|
|
worst := 0.0
|
|||
|
|
damping := 1.0
|
|||
|
|
for halving := 0; ; halving++ {
|
|||
|
|
worst = 0.0
|
|||
|
|
for j := range p {
|
|||
|
|
applied[j] = beta[j] + damping*step[0][j]
|
|||
|
|
if d := math.Abs(damping * step[0][j]); d > worst {
|
|||
|
|
worst = d
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if logLikeAt(applied) >= currentLogLike || halving == 30 {
|
|||
|
|
break
|
|||
|
|
}
|
|||
|
|
damping /= 2
|
|||
|
|
}
|
|||
|
|
copy(beta, applied)
|
|||
|
|
if worst < 1e-10 {
|
|||
|
|
converged = true
|
|||
|
|
iterations = iter
|
|||
|
|
// One final pass for the fitted means at the settled
|
|||
|
|
// coefficients, clamped exactly as the loop clamped them:
|
|||
|
|
// the unclamped exponential overflows to an infinity for
|
|||
|
|
// |eta| past the log of the ceiling, and the likelihood
|
|||
|
|
// below is evaluated on these values.
|
|||
|
|
for r := range n {
|
|||
|
|
eta := 0.0
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for j, xj := range row {
|
|||
|
|
eta += xj * beta[j]
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for j := range p {
|
|||
|
|
eta += x.FloatAt(r*p+j) * beta[j]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mean[r] = poissonMean(eta)
|
|||
|
|
}
|
|||
|
|
break
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if !converged {
|
|||
|
|
return nil, base.Errf("%s: %d iterations did not converge", name, maxIter)
|
|||
|
|
}
|
|||
|
|
logLike := 0.0
|
|||
|
|
for r := range n {
|
|||
|
|
var yv float64
|
|||
|
|
if fy != nil {
|
|||
|
|
yv = fy[r]
|
|||
|
|
} else {
|
|||
|
|
yv = y.FloatAt(r)
|
|||
|
|
}
|
|||
|
|
logGamma, _ := math.Lgamma(yv + 1)
|
|||
|
|
logLike += yv*math.Log(mean[r]) - mean[r] - logGamma
|
|||
|
|
}
|
|||
|
|
// Wald inference from the inverse Fisher information at the
|
|||
|
|
// optimum. The iteration's last Fisher matrix belongs to the
|
|||
|
|
// previous point, one damped step behind, so it is rebuilt from
|
|||
|
|
// the settled means before the solves, into the reused buffer and
|
|||
|
|
// on the same mirrored upper triangle.
|
|||
|
|
for i := range p {
|
|||
|
|
clear(fisher[i])
|
|||
|
|
}
|
|||
|
|
for r := range n {
|
|||
|
|
mu := mean[r]
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for i, xr := range row {
|
|||
|
|
fi := fisher[i][i:]
|
|||
|
|
for j, xj := range row[i:] {
|
|||
|
|
fi[j] += xr * xj * mu
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for i := range p {
|
|||
|
|
xr := x.FloatAt(r*p + i)
|
|||
|
|
for j := i; j < p; j++ {
|
|||
|
|
fisher[i][j] += xr * x.FloatAt(r*p+j) * mu
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mirrorUpper(fisher)
|
|||
|
|
out := &PoissonRegressionResult{
|
|||
|
|
Coefficients: beta,
|
|||
|
|
Fitted: mean,
|
|||
|
|
LogLikelihood: logLike,
|
|||
|
|
Iterations: iterations,
|
|||
|
|
Converged: true,
|
|||
|
|
}
|
|||
|
|
out.StandardErrors = make([]float64, p)
|
|||
|
|
out.ZStatistics = make([]float64, p)
|
|||
|
|
out.PValues = make([]float64, p)
|
|||
|
|
// One unit vector per coefficient, all solved through a single
|
|||
|
|
// factorisation of the Fisher information: the per-coefficient
|
|||
|
|
// solves refactored the same matrix p times, while the shared solve
|
|||
|
|
// substitutes each column through the identical factor.
|
|||
|
|
unit := make([][]float64, p)
|
|||
|
|
for j := range p {
|
|||
|
|
unit[j] = make([]float64, p)
|
|||
|
|
unit[j][j] = 1
|
|||
|
|
}
|
|||
|
|
inv, err := base.SolveSystem(name, fisher, unit)
|
|||
|
|
if err != nil {
|
|||
|
|
return nil, base.Errf("%s: %w", name, err)
|
|||
|
|
}
|
|||
|
|
for j := range p {
|
|||
|
|
// The Wald variance is the diagonal of the inverse Fisher
|
|||
|
|
// information. A near-collinear design can drive the solve to a
|
|||
|
|
// tiny negative diagonal entry through rounding alone: the bare
|
|||
|
|
// square root would then be a NaN reported beside a nil error.
|
|||
|
|
// An exact zero stays a zero standard error; a negative entry
|
|||
|
|
// means the design is near-collinear and is refused.
|
|||
|
|
v := inv[j][j]
|
|||
|
|
switch {
|
|||
|
|
case v > 0:
|
|||
|
|
out.StandardErrors[j] = math.Sqrt(v)
|
|||
|
|
case v == 0:
|
|||
|
|
out.StandardErrors[j] = 0
|
|||
|
|
default:
|
|||
|
|
return nil, base.Errf("%s: the design is near-collinear: the Wald variance of coefficient %d came out negative (%g)", name, j, v)
|
|||
|
|
}
|
|||
|
|
if out.StandardErrors[j] == 0 {
|
|||
|
|
// A zero Wald variance: an exact fit reports total evidence
|
|||
|
|
// for a live coefficient and nothing to test for a zero
|
|||
|
|
// one, never the 0/0 NaN pair.
|
|||
|
|
if beta[j] != 0 {
|
|||
|
|
out.ZStatistics[j] = math.Copysign(math.Inf(1), beta[j])
|
|||
|
|
out.PValues[j] = 0
|
|||
|
|
} else {
|
|||
|
|
out.ZStatistics[j] = 0
|
|||
|
|
out.PValues[j] = 1
|
|||
|
|
}
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
out.ZStatistics[j] = beta[j] / out.StandardErrors[j]
|
|||
|
|
z := out.ZStatistics[j]
|
|||
|
|
// The two-sided normal tail in one Erfc call on the magnitude.
|
|||
|
|
// The algebraic form 2·(1−Φ(z)) cancels to exactly zero once z
|
|||
|
|
// passes about 8.3, where the true tail nears 1e-17 and keeps
|
|||
|
|
// another three hundred orders below before the smallest
|
|||
|
|
// float64.
|
|||
|
|
out.PValues[j] = math.Erfc(math.Abs(z) / math.Sqrt2)
|
|||
|
|
}
|
|||
|
|
return out, nil
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
// LogisticRegression fits y = Bernoulli(sigmoid(X·β)) by maximum
|
|||
|
|
// likelihood. The design carries n rows and p columns exactly as
|
|||
|
|
// LinearRegression's (intercept included by the caller as a constant
|
|||
|
|
// column when wanted), y holds zeros and ones, and the fit runs
|
|||
|
|
// Newton-Raphson until the largest coefficient update drops under
|
|||
|
|
// 1e-10. Perfectly separable data has no finite optimum: the run
|
|||
|
|
// reports an error rather than diverging coefficients.
|
|||
|
|
func LogisticRegression(x, y *core.Array) (*LogisticRegressionResult, error) {
|
|||
|
|
const name = "LogisticRegression"
|
|||
|
|
if x.NDim() != 2 {
|
|||
|
|
return nil, base.Errf("%s: the design must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
|
|||
|
|
}
|
|||
|
|
if y.NDim() != 1 {
|
|||
|
|
return nil, base.Errf("%s: the response must be rank 1", name)
|
|||
|
|
}
|
|||
|
|
if x.Dtype() == core.Complex || y.Dtype() == core.Complex {
|
|||
|
|
return nil, base.Errf("%s: complex inputs are not supported", name)
|
|||
|
|
}
|
|||
|
|
n, p := x.Shape()[0], x.Shape()[1]
|
|||
|
|
if y.Len() != n {
|
|||
|
|
return nil, base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len())
|
|||
|
|
}
|
|||
|
|
if n <= p {
|
|||
|
|
return nil, base.Errf("%s: need n > p, got %d observations and %d columns", name, n, p)
|
|||
|
|
}
|
|||
|
|
if p == 0 {
|
|||
|
|
return nil, base.Errf("%s: the design must carry at least one column", name)
|
|||
|
|
}
|
|||
|
|
// A non-finite design would flow through the sigmoid and the
|
|||
|
|
// Newton step into a "converged" all-NaN fit: as in
|
|||
|
|
// LinearRegression, non-finite input has no answer to report.
|
|||
|
|
if err := checkFinite(name, "the design", x); err != nil {
|
|||
|
|
return nil, err
|
|||
|
|
}
|
|||
|
|
for i := range y.Len() {
|
|||
|
|
v := y.FloatAt(i)
|
|||
|
|
if v != 0 && v != 1 {
|
|||
|
|
return nil, base.Errf("%s: response sample %d is %g, want 0 or 1", name, i, v)
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
beta := make([]float64, p)
|
|||
|
|
prob := make([]float64, n)
|
|||
|
|
// The design and the response are read through raw payload slices
|
|||
|
|
// when dense: the elements are the ones FloatAt returns, so every
|
|||
|
|
// product and sum below keeps its exact operand bits.
|
|||
|
|
fx := rawFloats(x)
|
|||
|
|
fy := rawFloats(y)
|
|||
|
|
const maxIter = 100
|
|||
|
|
converged := false
|
|||
|
|
iterations := maxIter
|
|||
|
|
// The normal-equation buffers are allocated once and cleared per
|
|||
|
|
// iteration; every accumulation pass starts from the zero state a
|
|||
|
|
// fresh allocation carried.
|
|||
|
|
hessian := make([][]float64, p)
|
|||
|
|
for i := range p {
|
|||
|
|
hessian[i] = make([]float64, p)
|
|||
|
|
}
|
|||
|
|
gradient := make([]float64, p)
|
|||
|
|
for iter := 1; iter <= maxIter; iter++ {
|
|||
|
|
for r := range n {
|
|||
|
|
eta := 0.0
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for j, xj := range row {
|
|||
|
|
eta += xj * beta[j]
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for j := range p {
|
|||
|
|
eta += x.FloatAt(r*p+j) * beta[j]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
// The sigmoid clamped away from its saturating ends: the
|
|||
|
|
// weights and the log both need a live second derivative.
|
|||
|
|
pr := logisticProbability(eta)
|
|||
|
|
prob[r] = pr
|
|||
|
|
}
|
|||
|
|
// The observed information is symmetric and each lower-triangle
|
|||
|
|
// entry equals its upper twin bit for bit (mirrorUpper), so the
|
|||
|
|
// accumulation runs the upper triangle alone and mirrors it once.
|
|||
|
|
for i := range p {
|
|||
|
|
clear(hessian[i])
|
|||
|
|
}
|
|||
|
|
clear(gradient)
|
|||
|
|
for r := range n {
|
|||
|
|
pr := prob[r]
|
|||
|
|
w := pr * (1 - pr)
|
|||
|
|
var yv float64
|
|||
|
|
if fy != nil {
|
|||
|
|
yv = fy[r]
|
|||
|
|
} else {
|
|||
|
|
yv = y.FloatAt(r)
|
|||
|
|
}
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for i, xr := range row {
|
|||
|
|
gradient[i] += xr * (yv - pr)
|
|||
|
|
// Upper triangle, both operands pre-sliced from i:
|
|||
|
|
// the same products in the same order, bounds checks
|
|||
|
|
// elided.
|
|||
|
|
hi := hessian[i][i:]
|
|||
|
|
for j, xj := range row[i:] {
|
|||
|
|
hi[j] += xr * xj * w
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for i := range p {
|
|||
|
|
xr := x.FloatAt(r*p + i)
|
|||
|
|
gradient[i] += xr * (yv - pr)
|
|||
|
|
for j := i; j < p; j++ {
|
|||
|
|
hessian[i][j] += xr * x.FloatAt(r*p+j) * w
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mirrorUpper(hessian)
|
|||
|
|
step, err := base.SolveSystem(name, hessian, [][]float64{gradient})
|
|||
|
|
if err != nil {
|
|||
|
|
return nil, base.Errf("%s: the Fisher information is singular (%w)", name, err)
|
|||
|
|
}
|
|||
|
|
worst := 0.0
|
|||
|
|
for j := range p {
|
|||
|
|
beta[j] += step[0][j]
|
|||
|
|
if math.Abs(step[0][j]) > worst {
|
|||
|
|
worst = math.Abs(step[0][j])
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if worst < 1e-10 {
|
|||
|
|
converged = true
|
|||
|
|
iterations = iter
|
|||
|
|
// One final pass for the fitted probabilities at the
|
|||
|
|
// settled coefficients, clamped exactly as the loop
|
|||
|
|
// clamped them: the unclamped form reaches exactly 0 and 1
|
|||
|
|
// for |eta| > ~37, and the likelihood below is evaluated on
|
|||
|
|
// these values.
|
|||
|
|
for r := range n {
|
|||
|
|
eta := 0.0
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for j, xj := range row {
|
|||
|
|
eta += xj * beta[j]
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for j := range p {
|
|||
|
|
eta += x.FloatAt(r*p+j) * beta[j]
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
prob[r] = logisticProbability(eta)
|
|||
|
|
}
|
|||
|
|
break
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if !converged {
|
|||
|
|
return nil, base.Errf("%s: %d iterations did not converge; the response may be perfectly separable", name, maxIter)
|
|||
|
|
}
|
|||
|
|
// Wald inference from the inverse Fisher information at the
|
|||
|
|
// optimum.
|
|||
|
|
//
|
|||
|
|
// The matrix is built row by row into the reused buffer, the way
|
|||
|
|
// the fitting loop builds its own: a row streams the design once
|
|||
|
|
// instead of once per coefficient, and the row's weight is formed
|
|||
|
|
// once. Each entry sums its products over the rows in ascending
|
|||
|
|
// order on the mirrored upper triangle, so the entries are the ones
|
|||
|
|
// the direct walk accumulated.
|
|||
|
|
fisher := hessian
|
|||
|
|
for i := range p {
|
|||
|
|
clear(fisher[i])
|
|||
|
|
}
|
|||
|
|
for r := range n {
|
|||
|
|
w := prob[r] * (1 - prob[r])
|
|||
|
|
if fx != nil {
|
|||
|
|
row := fx[r*p : r*p+p]
|
|||
|
|
for i, xi := range row {
|
|||
|
|
fi := fisher[i][i:]
|
|||
|
|
for j, xj := range row[i:] {
|
|||
|
|
fi[j] += xi * xj * w
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
} else {
|
|||
|
|
for i := range p {
|
|||
|
|
xi := x.FloatAt(r*p + i)
|
|||
|
|
fi := fisher[i]
|
|||
|
|
for j := i; j < p; j++ {
|
|||
|
|
fi[j] += xi * x.FloatAt(r*p+j) * w
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
mirrorUpper(fisher)
|
|||
|
|
out := &LogisticRegressionResult{
|
|||
|
|
Coefficients: beta,
|
|||
|
|
Fitted: prob,
|
|||
|
|
Iterations: iterations,
|
|||
|
|
Converged: true,
|
|||
|
|
}
|
|||
|
|
logLike := 0.0
|
|||
|
|
for r := range n {
|
|||
|
|
var yv float64
|
|||
|
|
if fy != nil {
|
|||
|
|
yv = fy[r]
|
|||
|
|
} else {
|
|||
|
|
yv = y.FloatAt(r)
|
|||
|
|
}
|
|||
|
|
logLike += yv*math.Log(prob[r]) + (1-yv)*math.Log(1-prob[r])
|
|||
|
|
}
|
|||
|
|
out.LogLikelihood = logLike
|
|||
|
|
out.StandardErrors = make([]float64, p)
|
|||
|
|
out.ZStatistics = make([]float64, p)
|
|||
|
|
out.PValues = make([]float64, p)
|
|||
|
|
// One unit vector per coefficient, all solved through a single
|
|||
|
|
// factorisation of the Fisher information: the per-coefficient
|
|||
|
|
// solves refactored the same matrix p times, while the shared solve
|
|||
|
|
// substitutes each column through the identical factor.
|
|||
|
|
unit := make([][]float64, p)
|
|||
|
|
for j := range p {
|
|||
|
|
unit[j] = make([]float64, p)
|
|||
|
|
unit[j][j] = 1
|
|||
|
|
}
|
|||
|
|
inv, err := base.SolveSystem(name, fisher, unit)
|
|||
|
|
if err != nil {
|
|||
|
|
return nil, base.Errf("%s: %w", name, err)
|
|||
|
|
}
|
|||
|
|
for j := range p {
|
|||
|
|
// The Wald variance is the diagonal of the inverse Fisher
|
|||
|
|
// information. A near-collinear design can drive the solve to a
|
|||
|
|
// tiny negative diagonal entry through rounding alone: the bare
|
|||
|
|
// square root would then be a NaN reported beside a nil error.
|
|||
|
|
// An exact zero stays a zero standard error; a negative entry
|
|||
|
|
// means the design is near-collinear and is refused.
|
|||
|
|
v := inv[j][j]
|
|||
|
|
switch {
|
|||
|
|
case v > 0:
|
|||
|
|
out.StandardErrors[j] = math.Sqrt(v)
|
|||
|
|
case v == 0:
|
|||
|
|
out.StandardErrors[j] = 0
|
|||
|
|
default:
|
|||
|
|
return nil, base.Errf("%s: the design is near-collinear: the Wald variance of coefficient %d came out negative (%g)", name, j, v)
|
|||
|
|
}
|
|||
|
|
if out.StandardErrors[j] == 0 {
|
|||
|
|
// A zero Wald variance: an exact fit reports total evidence
|
|||
|
|
// for a live coefficient and nothing to test for a zero
|
|||
|
|
// one, never the 0/0 NaN pair.
|
|||
|
|
if beta[j] != 0 {
|
|||
|
|
out.ZStatistics[j] = math.Copysign(math.Inf(1), beta[j])
|
|||
|
|
out.PValues[j] = 0
|
|||
|
|
} else {
|
|||
|
|
out.ZStatistics[j] = 0
|
|||
|
|
out.PValues[j] = 1
|
|||
|
|
}
|
|||
|
|
continue
|
|||
|
|
}
|
|||
|
|
out.ZStatistics[j] = beta[j] / out.StandardErrors[j]
|
|||
|
|
z := out.ZStatistics[j]
|
|||
|
|
// The two-sided normal tail in one Erfc call on the magnitude,
|
|||
|
|
// as in PoissonRegression: 2·(1−Φ(z)) cancels to exactly zero
|
|||
|
|
// once z passes about 8.3.
|
|||
|
|
out.PValues[j] = math.Erfc(math.Abs(z) / math.Sqrt2)
|
|||
|
|
}
|
|||
|
|
return out, nil
|
|||
|
|
}
|