759 lines
24 KiB
Go
759 lines
24 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: MIT
|
||
|
||
package stats
|
||
|
||
import (
|
||
"math"
|
||
|
||
"sourcedock.dev/petrbalvin/tensor/internal/base"
|
||
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
||
)
|
||
|
||
// Linear regression with classical inference: the ordinary
|
||
// least-squares fit together with the uncertainty statement every
|
||
// empirical paper needs: standard errors, t-tests on each
|
||
// coefficient, R², adjusted R² and the F-test of the model as a
|
||
// whole. The linear algebra is the normal equations solved by the
|
||
// shared LU: regression designs are small and well conditioned in
|
||
// practice, and a caller with a genuinely ill-conditioned design
|
||
// should regularise (or reach for linalg.SolveTruncated) rather than
|
||
// trust any black-box fit.
|
||
|
||
// LinearRegressionResult carries the fit and its inference. Each
|
||
// slice is indexed by column of the design matrix, in order.
|
||
type LinearRegressionResult struct {
|
||
// Coefficients are the least-squares estimates β̂.
|
||
Coefficients []float64
|
||
// StandardErrors are the estimated standard deviations of the
|
||
// coefficient estimators.
|
||
StandardErrors []float64
|
||
// TStatistics are β̂/SE per coefficient.
|
||
TStatistics []float64
|
||
// PValues are the two-sided p-values of the t-tests.
|
||
PValues []float64
|
||
// ResidualVariance is σ̂² = RSS/(n − p).
|
||
ResidualVariance float64
|
||
// RSquared and AdjustedRSquared measure the fit.
|
||
RSquared float64
|
||
AdjustedRSquared float64
|
||
// FStatistic with DModel and DResidual as its degrees of freedom
|
||
// and FPValue its tail probability. DModel is p − 1 for a design
|
||
// with a constant column and p without one; DResidual is n − p.
|
||
FStatistic float64
|
||
DModel int
|
||
DResidual int
|
||
FPValue float64
|
||
// Fitted and Residuals align with the rows of the design.
|
||
Fitted []float64
|
||
Residuals []float64
|
||
}
|
||
|
||
// LinearRegression fits y = X·β by ordinary least squares over the
|
||
// design matrix X (n rows, p columns, the intercept included by the
|
||
// caller as a constant column when wanted) and reports the full
|
||
// classical inference. Inputs must be rank-2 / rank-1 of matching
|
||
// length, real-valued, with n > p and X of full column rank; a
|
||
// rank-deficient design is an error naming the condition.
|
||
func LinearRegression(x, y *core.Array) (*LinearRegressionResult, error) {
|
||
const name = "LinearRegression"
|
||
if x.NDim() != 2 {
|
||
return nil, base.Errf("%s: the design must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
|
||
}
|
||
if y.NDim() != 1 {
|
||
return nil, base.Errf("%s: the response must be rank 1, got shape %s", name, base.ShapeText(y.Shape()))
|
||
}
|
||
if x.Dtype() == core.Complex || y.Dtype() == core.Complex {
|
||
return nil, base.Errf("%s: complex inputs are not supported", name)
|
||
}
|
||
n, p := x.Shape()[0], x.Shape()[1]
|
||
if y.Len() != n {
|
||
return nil, base.Errf("%s: the design has %d rows but the response %d", name, n, y.Len())
|
||
}
|
||
if n <= p {
|
||
return nil, base.Errf("%s: need n > p, got %d observations and %d columns", name, n, p)
|
||
}
|
||
// Non-finite input has no answer to report: a single NaN would
|
||
// propagate into every coefficient and every statistic, and the
|
||
// other tests in the package refuse it for the same reason. Both
|
||
// scans are bounded by the visible element counts: a rebased view's
|
||
// payload may run past them, and a non-finite slot there is nobody's
|
||
// observation.
|
||
nVis := x.Len()
|
||
fx := rawFloats(x)
|
||
fy := rawFloats(y)
|
||
if fx == nil {
|
||
for i := range x.Len() {
|
||
if v := x.FloatAt(i); math.IsNaN(v) || math.IsInf(v, 0) {
|
||
return nil, base.Errf("%s: the design holds the non-finite value %g", name, v)
|
||
}
|
||
}
|
||
} else {
|
||
for _, v := range fx[:nVis] {
|
||
if math.IsNaN(v) || math.IsInf(v, 0) {
|
||
return nil, base.Errf("%s: the design holds the non-finite value %g", name, v)
|
||
}
|
||
}
|
||
}
|
||
if fy == nil {
|
||
for i := range n {
|
||
if v := y.FloatAt(i); math.IsNaN(v) || math.IsInf(v, 0) {
|
||
return nil, base.Errf("%s: the response holds the non-finite value %g", name, v)
|
||
}
|
||
}
|
||
} else {
|
||
for _, v := range fy[:n] {
|
||
if math.IsNaN(v) || math.IsInf(v, 0) {
|
||
return nil, base.Errf("%s: the response holds the non-finite value %g", name, v)
|
||
}
|
||
}
|
||
}
|
||
// Whether the caller supplied an intercept, as a constant column.
|
||
// It decides what the null model is: with a constant column it is
|
||
// the mean of y (centred total sum of squares), without one it is
|
||
// zero and Σy² plays that role. The distinction changes R², the
|
||
// model degrees of freedom and the F statistic.
|
||
hasConstant := hasConstantColumn(x, n, p)
|
||
|
||
// Normal equations: (XᵀX)β = Xᵀy. The row-wise walk below visits
|
||
// the rows in the same order the column-wise walk did, so every
|
||
// entry sums identical products in identical order; the hoisted
|
||
// row value re-reads the same bits the inner loop re-read. XᵀX is
|
||
// symmetric and each lower-triangle entry equals its upper twin bit
|
||
// for bit (mirrorUpper: the row-wise product commutes bitwise and
|
||
// both entries sum the rows in the same order), so the accumulation
|
||
// runs the upper triangle alone and mirrors it once.
|
||
xtx := make([][]float64, p)
|
||
for i := range p {
|
||
xtx[i] = make([]float64, p)
|
||
}
|
||
xty := make([]float64, p)
|
||
if fx != nil && fy != nil {
|
||
for r := range n {
|
||
row := fx[r*p : r*p+p]
|
||
yv := fy[r]
|
||
for i, xi := range row {
|
||
xty[i] += xi * yv
|
||
// Upper triangle, both operands pre-sliced from i: the
|
||
// same products in the same order, bounds checks elided.
|
||
ai := xtx[i][i:]
|
||
for j, xj := range row[i:] {
|
||
ai[j] += xi * xj
|
||
}
|
||
}
|
||
}
|
||
} else {
|
||
for r := range n {
|
||
yv := y.FloatAt(r)
|
||
for i := range p {
|
||
xi := x.FloatAt(r*p + i)
|
||
xty[i] += xi * yv
|
||
for j := i; j < p; j++ {
|
||
xtx[i][j] += xi * x.FloatAt(r*p+j)
|
||
}
|
||
}
|
||
}
|
||
}
|
||
mirrorUpper(xtx)
|
||
// SolveSystem factors its matrix in place, so the covariance pass
|
||
// below needs a pristine copy of the normal equations.
|
||
xtxPristine := make([][]float64, p)
|
||
for i := range p {
|
||
xtxPristine[i] = append([]float64(nil), xtx[i]...)
|
||
}
|
||
// SolveSystem consumes columns: one column holding Xᵀy.
|
||
solved, err := base.SolveSystem(name, xtx, [][]float64{xty})
|
||
if err != nil {
|
||
return nil, base.Errf("%s: the design is rank deficient (%w)", name, err)
|
||
}
|
||
beta := make([]float64, p)
|
||
for i := range p {
|
||
beta[i] = solved[0][i]
|
||
}
|
||
|
||
out := &LinearRegressionResult{Coefficients: beta}
|
||
out.Fitted = make([]float64, n)
|
||
out.Residuals = make([]float64, n)
|
||
rss := 0.0
|
||
tss := 0.0
|
||
uncentred := 0.0
|
||
maxRes, maxDev, maxY := 0.0, 0.0, 0.0
|
||
mean := 0.0
|
||
if fy != nil {
|
||
for _, v := range fy[:n] {
|
||
mean += v
|
||
}
|
||
} else {
|
||
for i := range n {
|
||
mean += y.FloatAt(i)
|
||
}
|
||
}
|
||
mean /= float64(n)
|
||
for r := range n {
|
||
f := 0.0
|
||
if fx != nil {
|
||
row := fx[r*p : r*p+p]
|
||
for i, xi := range row {
|
||
f += beta[i] * xi
|
||
}
|
||
} else {
|
||
for i := range p {
|
||
f += beta[i] * x.FloatAt(r*p+i)
|
||
}
|
||
}
|
||
out.Fitted[r] = f
|
||
var res, yv float64
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
res = yv - f
|
||
out.Residuals[r] = res
|
||
rss += res * res
|
||
tss += (yv - mean) * (yv - mean)
|
||
uncentred += yv * yv
|
||
if a := math.Abs(res); a > maxRes {
|
||
maxRes = a
|
||
}
|
||
if a := math.Abs(yv - mean); a > maxDev {
|
||
maxDev = a
|
||
}
|
||
if a := math.Abs(yv); a > maxY {
|
||
maxY = a
|
||
}
|
||
}
|
||
if !hasConstant {
|
||
// The null model is y = 0, so the uncentred total is what the
|
||
// model has to beat, and it carries n degrees of freedom.
|
||
tss = uncentred
|
||
}
|
||
// The factored sums of squares: a response on a scale whose squared
|
||
// deviations fall below the subnormal floor reads as a zero sum while
|
||
// its deviations are live, and the statistics below would report the
|
||
// evidence backwards (an exact fit the t statistics cannot support,
|
||
// an F of zero beside them). Each pair keeps the largest deviation as
|
||
// the scale and the scaled sum as the unit, so scale²·unit is the
|
||
// true sum wherever the plain product underflows; the unit stays 1
|
||
// whenever the plain sum already holds.
|
||
rssScale, rssUnit := 1.0, rss
|
||
if rss == 0 && maxRes > 0 {
|
||
rssScale, rssUnit = maxRes, 0.0
|
||
for _, res := range out.Residuals {
|
||
d := res / maxRes
|
||
rssUnit += d * d
|
||
}
|
||
}
|
||
tssScale, tssUnit := 1.0, tss
|
||
if tss == 0 {
|
||
devScale := maxDev
|
||
if !hasConstant {
|
||
devScale = maxY
|
||
}
|
||
if devScale > 0 {
|
||
tssScale = devScale
|
||
tssUnit = 0.0
|
||
for r := range n {
|
||
var yv, dev float64
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
if !hasConstant {
|
||
dev = yv
|
||
} else {
|
||
dev = yv - mean
|
||
}
|
||
d := dev / devScale
|
||
tssUnit += d * d
|
||
}
|
||
}
|
||
}
|
||
dof := n - p
|
||
out.ResidualVariance = rss / float64(dof)
|
||
tssDOF := n - 1
|
||
if !hasConstant {
|
||
tssDOF = n
|
||
}
|
||
if tss == 0 && tssScale == 1 {
|
||
// A constant response reproduced exactly: R² is 1 by the
|
||
// perfect-fit convention, not the 1 − 0/0 NaN every consumer
|
||
// would propagate. The same guard the F statistic below has.
|
||
out.RSquared = 1
|
||
out.AdjustedRSquared = 1
|
||
} else if tss == 0 {
|
||
// The total underflowed while the response varies: the ratio of
|
||
// the factored forms, the scale factors divided out one at a
|
||
// time. Both R² measures round back to 1 here, but the F
|
||
// statistic below reads the same factored pieces and does not.
|
||
ratio := rssUnit / tssUnit * (rssScale / tssScale) * (rssScale / tssScale)
|
||
out.RSquared = 1 - ratio
|
||
out.AdjustedRSquared = 1 - ratio*float64(tssDOF)/float64(dof)
|
||
} else {
|
||
out.RSquared = 1 - rss/tss
|
||
out.AdjustedRSquared = 1 - (rss/float64(dof))/(tss/float64(tssDOF))
|
||
}
|
||
out.DModel = p - 1
|
||
if !hasConstant {
|
||
out.DModel = p
|
||
}
|
||
out.DResidual = dof
|
||
|
||
// Covariance of β̂: σ̂²(XᵀX)⁻¹, its diagonal read from one
|
||
// factorisation of the pristine normal equations against all p unit
|
||
// columns at once. Solving one unit vector per coefficient
|
||
// refactors the same matrix p times; the shared solve factors once
|
||
// and substitutes each column through the identical factor, so the
|
||
// diagonal is the one the p separate solves produced, bit for bit.
|
||
out.StandardErrors = make([]float64, p)
|
||
out.TStatistics = make([]float64, p)
|
||
out.PValues = make([]float64, p)
|
||
unit := make([][]float64, p)
|
||
for j := range p {
|
||
unit[j] = make([]float64, p)
|
||
unit[j][j] = 1
|
||
}
|
||
inv, err := base.SolveSystem(name, xtxPristine, unit)
|
||
if err != nil {
|
||
return nil, base.Errf("%s: %w", name, err)
|
||
}
|
||
for j := range p {
|
||
v := out.ResidualVariance * inv[j][j] // σ̂²·(XᵀX)⁻¹_jj
|
||
switch {
|
||
case v > 0:
|
||
se := math.Sqrt(v)
|
||
out.StandardErrors[j] = se
|
||
out.TStatistics[j] = beta[j] / se
|
||
pv, err := twoSidedT(out.TStatistics[j], dof)
|
||
if err != nil {
|
||
return nil, base.Errf("%s: %w", name, err)
|
||
}
|
||
out.PValues[j] = pv
|
||
case v == 0:
|
||
// The residual sum of squares may have underflowed while the
|
||
// residuals live: the factored standard error is representable
|
||
// where the squared one is not, and the t test then reports
|
||
// the evidence it actually holds instead of an unearned
|
||
// infinity.
|
||
if rssScale != 1 && inv[j][j] > 0 {
|
||
se := rssScale * math.Sqrt(rssUnit*inv[j][j]/float64(dof))
|
||
if se > 0 {
|
||
out.StandardErrors[j] = se
|
||
out.TStatistics[j] = beta[j] / se
|
||
pv, err := twoSidedT(out.TStatistics[j], dof)
|
||
if err != nil {
|
||
return nil, base.Errf("%s: %w", name, err)
|
||
}
|
||
out.PValues[j] = pv
|
||
continue
|
||
}
|
||
}
|
||
// An exact fit: the coefficient is infinitely many standard
|
||
// errors from zero, and the evidence is total. Reporting
|
||
// t = 0 next to p = 0 would contradict itself. A zero
|
||
// coefficient beside the zero standard error has nothing
|
||
// to test and reports p = 1.
|
||
out.StandardErrors[j] = 0
|
||
if beta[j] != 0 {
|
||
out.TStatistics[j] = math.Copysign(math.Inf(1), beta[j])
|
||
out.PValues[j] = 0
|
||
} else {
|
||
out.PValues[j] = 1
|
||
}
|
||
default:
|
||
// A near-collinear design drives the solve's diagonal
|
||
// negative through rounding alone: the Wald variance would
|
||
// be a NaN beside a nil error, the same refusal GLM makes.
|
||
return nil, base.Errf("%s: the design is near-collinear: the variance of coefficient %d came out negative (%g)", name, j, v)
|
||
}
|
||
}
|
||
|
||
// F-test of the model: H₀: every coefficient is zero. With a
|
||
// constant column this is the usual regression F against the mean;
|
||
// without one it is the test against the zero model (DModel = p).
|
||
if out.DModel > 0 {
|
||
explained := tss - rss
|
||
if explained < 0 {
|
||
explained = 0 // rounding only, and a negative F is meaningless
|
||
}
|
||
out.FStatistic = explained / float64(out.DModel) / out.ResidualVariance
|
||
if (math.IsInf(out.FStatistic, 0) || math.IsNaN(out.FStatistic)) && (rssScale != 1 || tssScale != 1) {
|
||
// An underflowed sum of squares drove the quotient to Inf or
|
||
// 0/0 while the factored pieces live: F from the factored
|
||
// forms, every scale factor applied one division at a time so
|
||
// no intermediate leaves the representable range before the
|
||
// answer does. rssUnit 0 is the exact fit, whose F is
|
||
// genuinely infinite.
|
||
out.FStatistic = (tssUnit*tssScale/rssScale/rssScale*tssScale - rssUnit) *
|
||
float64(dof) / (float64(out.DModel) * rssUnit)
|
||
if out.FStatistic < 0 {
|
||
out.FStatistic = 0
|
||
}
|
||
}
|
||
if math.IsNaN(out.FStatistic) {
|
||
// 0/0: a response with no variation at all, reproduced
|
||
// exactly by the fit. There is no evidence of a model, so
|
||
// the statistic is the zero the test reads as p = 1, not a
|
||
// NaN that every consumer would propagate.
|
||
out.FStatistic = 0
|
||
}
|
||
// Tail of F(d1, d2) at f: the regularised incomplete beta
|
||
// I_{d2/(d2+d1·f)}(d2/2, d1/2). The argument is clamped to
|
||
// [0,1]: the identity is only defined there, and an F of 0
|
||
// (explained 0) or an infinite one would step outside through
|
||
// rounding alone.
|
||
d1, d2 := float64(out.DModel), float64(out.DResidual)
|
||
xi := d2 / (d2 + d1*out.FStatistic)
|
||
xi = min(max(xi, 0), 1)
|
||
tail, err := BetaIncomplete(xi, d2/2, d1/2)
|
||
if err != nil {
|
||
return nil, base.Errf("%s: %w", name, err)
|
||
}
|
||
out.FPValue = tail
|
||
} else {
|
||
// An intercept-only design has no model term to test: the F
|
||
// stays at its zero value and the p value is 1, the same
|
||
// convention the F = 0 guard below uses. Leaving the zero
|
||
// value in FPValue would report the null model as maximally
|
||
// significant.
|
||
out.FPValue = 1
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
// twoSidedT returns P(|T| > |t|) for Student-t with df degrees of
|
||
// freedom, by the closed-form tail I_z(df/2, 1/2) with z = df/(df+t²).
|
||
// The identity is used rather than 2·(1 − T_cdf(t)): near t = 0 the
|
||
// subtraction cancels catastrophically, while the incomplete beta
|
||
// stays accurate into the far tail where p values matter most. The
|
||
// upper-tail helper carries it, so the asymptotic forms that hold the
|
||
// df ≤ 2 tails past t²'s overflow serve here too: the bare closed form
|
||
// answers a silent 0 there while the true tail is still representable.
|
||
func twoSidedT(t float64, df int) (float64, error) {
|
||
upper, err := studentTUpperTail(math.Abs(t), df)
|
||
if err != nil {
|
||
return 0, err
|
||
}
|
||
return 2 * upper, nil
|
||
}
|
||
|
||
// hasConstantColumn reports whether an (n, p) design holds a column of
|
||
// one repeated value, the caller-supplied intercept. The detection
|
||
// must read the design as handed in: a column that is constant there
|
||
// can stop being constant under a further transformation (Weighted-
|
||
// LinearRegression's sqrt-weighted design is the case in point), and
|
||
// the caller's null model follows the design it actually supplied.
|
||
func hasConstantColumn(x *core.Array, n, p int) bool {
|
||
fx := rawFloats(x)
|
||
for j := range p {
|
||
first := x.FloatAt(j)
|
||
constant := true
|
||
for r := 1; r < n; r++ {
|
||
var v float64
|
||
if fx != nil {
|
||
v = fx[r*p+j]
|
||
} else {
|
||
v = x.FloatAt(r*p + j)
|
||
}
|
||
if v != first {
|
||
constant = false
|
||
break
|
||
}
|
||
}
|
||
if constant {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
// WeightedLinearRegression fits y = X·β by weighted least squares,
|
||
// observation i carrying the positive weight w[i]: the normal
|
||
// equations run on the sqrt-weighted system, so every statistic is
|
||
// the classical weighted-theory one (σ̂² on Σw·r² with n − p degrees
|
||
// of freedom, SEs from σ̂²(XᵀWX)⁻¹), while Fitted and Residuals are
|
||
// reported in the original, unweighted units. The weights must be
|
||
// finite and positive; everything else validates as LinearRegression
|
||
// does.
|
||
//
|
||
// The model statistics are the weighted-theory ones as well: R², the
|
||
// adjusted R² and the model F test are the centred quantities against
|
||
// the weighted mean Σw·y/Σw whenever the design as supplied carries a
|
||
// constant column. That column is detected on the unweighted design,
|
||
// where it is still constant: sqrt(w) makes even the intercept column
|
||
// non-constant in the system that is actually solved.
|
||
func WeightedLinearRegression(x, y, w *core.Array) (*LinearRegressionResult, error) {
|
||
const name = "WeightedLinearRegression"
|
||
if x.NDim() != 2 {
|
||
return nil, base.Errf("%s: the design must be rank 2, got shape %s", name, base.ShapeText(x.Shape()))
|
||
}
|
||
if y.NDim() != 1 || w.NDim() != 1 {
|
||
return nil, base.Errf("%s: the response and the weights must be rank 1", name)
|
||
}
|
||
if x.Dtype() == core.Complex || y.Dtype() == core.Complex || w.Dtype() == core.Complex {
|
||
return nil, base.Errf("%s: complex inputs are not supported", name)
|
||
}
|
||
n, p := x.Shape()[0], x.Shape()[1]
|
||
if y.Len() != n || w.Len() != n {
|
||
return nil, base.Errf("%s: the design has %d rows, the response %d and the weights %d",
|
||
name, n, y.Len(), w.Len())
|
||
}
|
||
xw := core.New(core.Float, n, p)
|
||
yw := core.New(core.Float, n)
|
||
xwVals := xw.RawFloats()
|
||
ywVals := yw.RawFloats()
|
||
// The payload walks below read the dense slices directly where they
|
||
// exist: the elements are the ones FloatAt returns, so every product
|
||
// and every sum keeps its exact operand bits.
|
||
fw := rawFloats(w)
|
||
fx := rawFloats(x)
|
||
fy := rawFloats(y)
|
||
for r := range n {
|
||
var weight float64
|
||
if fw != nil {
|
||
weight = fw[r]
|
||
} else {
|
||
weight = w.FloatAt(r)
|
||
}
|
||
if math.IsNaN(weight) || math.IsInf(weight, 0) || weight <= 0 {
|
||
return nil, base.Errf("%s: weight %d is %g, want a finite positive value", name, r, weight)
|
||
}
|
||
sqrtW := math.Sqrt(weight)
|
||
for j := range p {
|
||
var xj float64
|
||
if fx != nil {
|
||
xj = fx[r*p+j]
|
||
} else {
|
||
xj = x.FloatAt(r*p + j)
|
||
}
|
||
xwVals[r*p+j] = xj * sqrtW
|
||
}
|
||
var yv float64
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
ywVals[r] = yv * sqrtW
|
||
}
|
||
out, err := LinearRegression(xw, yw)
|
||
if err != nil {
|
||
return nil, base.Errf("%s: %w", name, err)
|
||
}
|
||
// hasConstant is read off the unweighted design, because that is the
|
||
// model the caller described; the sqrt-weighted system cannot answer
|
||
// the question, its intercept column is sqrt(w).
|
||
hasConstant := hasConstantColumn(x, n, p)
|
||
// Fitted and Residuals back in the original units, against the
|
||
// same coefficients.
|
||
for r := range n {
|
||
f := 0.0
|
||
if fx != nil {
|
||
row := fx[r*p : r*p+p]
|
||
for i, xi := range row {
|
||
f += out.Coefficients[i] * xi
|
||
}
|
||
} else {
|
||
for i := range p {
|
||
f += out.Coefficients[i] * x.FloatAt(r*p+i)
|
||
}
|
||
}
|
||
var yv float64
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
out.Fitted[r] = f
|
||
out.Residuals[r] = yv - f
|
||
}
|
||
// The weighted model statistics, from the residuals just computed:
|
||
// RSS_w = Σw·r², the weighted mean ȳ_w = Σw·y/Σw, and the centred
|
||
// total Σw·(y − ȳ_w)². The delegated fit had to answer the same
|
||
// questions for the sqrt-weighted system, which is a different
|
||
// regression and reports the uncentred conventions whenever the
|
||
// weights vary, so the four model-level fields are overwritten here.
|
||
sumW, sumWY, rssW := 0.0, 0.0, 0.0
|
||
maxResW := 0.0
|
||
for r := range n {
|
||
var wr float64
|
||
if fw != nil {
|
||
wr = fw[r]
|
||
} else {
|
||
wr = w.FloatAt(r)
|
||
}
|
||
var yv float64
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
sumW += wr
|
||
sumWY += wr * yv
|
||
rssW += wr * out.Residuals[r] * out.Residuals[r]
|
||
if a := math.Abs(out.Residuals[r]); a > maxResW {
|
||
maxResW = a
|
||
}
|
||
}
|
||
tssW := 0.0
|
||
maxDevW := 0.0
|
||
if hasConstant {
|
||
meanW := sumWY / sumW
|
||
for r := range n {
|
||
var wr, yv float64
|
||
if fw != nil {
|
||
wr = fw[r]
|
||
} else {
|
||
wr = w.FloatAt(r)
|
||
}
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
d := yv - meanW
|
||
tssW += wr * d * d
|
||
if a := math.Abs(d); a > maxDevW {
|
||
maxDevW = a
|
||
}
|
||
}
|
||
} else {
|
||
// Without an intercept the null model is zero, so Σw·y² is the
|
||
// total the model has to beat and it carries n degrees of freedom.
|
||
for r := range n {
|
||
var wr, yv float64
|
||
if fw != nil {
|
||
wr = fw[r]
|
||
} else {
|
||
wr = w.FloatAt(r)
|
||
}
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
tssW += wr * yv * yv
|
||
if a := math.Abs(yv); a > maxDevW {
|
||
maxDevW = a
|
||
}
|
||
}
|
||
}
|
||
// The weighted sums of squares carry the same factored form the
|
||
// unweighted fit keeps: a response scale whose weighted squared
|
||
// deviations fall below the subnormal floor reads as a zero sum
|
||
// while the deviations live, and the F below would report zero
|
||
// evidence beside the t statistics' infinity.
|
||
rssScaleW, rssUnitW := 1.0, rssW
|
||
if rssW == 0 && maxResW > 0 {
|
||
rssScaleW = maxResW
|
||
rssUnitW = 0.0
|
||
for r := range n {
|
||
var wr float64
|
||
if fw != nil {
|
||
wr = fw[r]
|
||
} else {
|
||
wr = w.FloatAt(r)
|
||
}
|
||
d := out.Residuals[r] / maxResW
|
||
rssUnitW += wr * d * d
|
||
}
|
||
}
|
||
tssScaleW, tssUnitW := 1.0, tssW
|
||
if tssW == 0 && maxDevW > 0 {
|
||
tssScaleW = maxDevW
|
||
tssUnitW = 0.0
|
||
if hasConstant {
|
||
meanW := sumWY / sumW
|
||
for r := range n {
|
||
var wr, yv float64
|
||
if fw != nil {
|
||
wr = fw[r]
|
||
} else {
|
||
wr = w.FloatAt(r)
|
||
}
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
d := (yv - meanW) / maxDevW
|
||
tssUnitW += wr * d * d
|
||
}
|
||
} else {
|
||
for r := range n {
|
||
var wr, yv float64
|
||
if fw != nil {
|
||
wr = fw[r]
|
||
} else {
|
||
wr = w.FloatAt(r)
|
||
}
|
||
if fy != nil {
|
||
yv = fy[r]
|
||
} else {
|
||
yv = y.FloatAt(r)
|
||
}
|
||
d := yv / maxDevW
|
||
tssUnitW += wr * d * d
|
||
}
|
||
}
|
||
}
|
||
tssDOF := n - 1
|
||
if !hasConstant {
|
||
tssDOF = n
|
||
}
|
||
if tssW == 0 && tssScaleW == 1 {
|
||
// Constant weighted response, exact fit: 1, as above.
|
||
out.RSquared = 1
|
||
out.AdjustedRSquared = 1
|
||
} else if tssW == 0 {
|
||
// The weighted total underflowed while the weighted response
|
||
// varies: the factored ratio, both R² measures rounding back
|
||
// to 1 while the F below reads the same pieces and does not.
|
||
ratio := rssUnitW / tssUnitW * (rssScaleW / tssScaleW) * (rssScaleW / tssScaleW)
|
||
out.RSquared = 1 - ratio
|
||
out.AdjustedRSquared = 1 - ratio*float64(tssDOF)/float64(out.DResidual)
|
||
} else {
|
||
out.RSquared = 1 - rssW/tssW
|
||
out.AdjustedRSquared = 1 - (rssW/float64(out.DResidual))/(tssW/float64(tssDOF))
|
||
}
|
||
out.DModel = p - 1
|
||
if !hasConstant {
|
||
out.DModel = p
|
||
}
|
||
if out.DModel > 0 {
|
||
explained := tssW - rssW
|
||
if explained < 0 {
|
||
explained = 0 // rounding only, and a negative F is meaningless
|
||
}
|
||
out.FStatistic = explained / float64(out.DModel) / out.ResidualVariance
|
||
if (math.IsInf(out.FStatistic, 0) || math.IsNaN(out.FStatistic)) && (rssScaleW != 1 || tssScaleW != 1) {
|
||
// The factored F, as in the unweighted path: every scale
|
||
// factor divided out one step at a time.
|
||
out.FStatistic = (tssUnitW*tssScaleW/rssScaleW/rssScaleW*tssScaleW - rssUnitW) *
|
||
float64(out.DResidual) / (float64(out.DModel) * rssUnitW)
|
||
if out.FStatistic < 0 {
|
||
out.FStatistic = 0
|
||
}
|
||
}
|
||
if math.IsNaN(out.FStatistic) {
|
||
// 0/0, as in the unweighted path: nothing to test, p = 1.
|
||
out.FStatistic = 0
|
||
}
|
||
d1, d2 := float64(out.DModel), float64(out.DResidual)
|
||
xi := d2 / (d2 + d1*out.FStatistic)
|
||
xi = min(max(xi, 0), 1)
|
||
tail, err := BetaIncomplete(xi, d2/2, d1/2)
|
||
if err != nil {
|
||
return nil, base.Errf("%s: %w", name, err)
|
||
}
|
||
out.FPValue = tail
|
||
} else {
|
||
// Intercept-only, as in the unweighted path: no model term to
|
||
// test, F 0 and p 1.
|
||
out.FStatistic = 0
|
||
out.FPValue = 1
|
||
}
|
||
return out, nil
|
||
}
|