119 lines
3.5 KiB
Go
119 lines
3.5 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
// Command regression fits a linear trend to a noisy time series,
|
|
// reports the full inference table (coefficients, standard errors,
|
|
// t-statistics, p-values, R²) and checks that the residuals are
|
|
// actually uncorrelated, which is the assumption the t-tests rest on.
|
|
//
|
|
// Usage: go run ./examples/regression
|
|
package main
|
|
|
|
import (
|
|
"fmt"
|
|
"log"
|
|
|
|
"sourcedock.dev/petrbalvin/tensor"
|
|
"sourcedock.dev/petrbalvin/tensor/signal"
|
|
"sourcedock.dev/petrbalvin/tensor/stats"
|
|
)
|
|
|
|
func main() {
|
|
const n = 400
|
|
// A trend of 0.05 per sample on a level of 2, with AR(1) noise
|
|
// (rho = 0.3), drawn from the reproducible generator.
|
|
g := tensor.NewGenerator(7)
|
|
white, err := tensor.Normal(g, n, 0, 1)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
y := make([]float64, n)
|
|
ar := 0.0
|
|
for i := range n {
|
|
w, _ := tensor.FloatAt(white, i)
|
|
ar = 0.3*ar + w
|
|
y[i] = 2 + 0.05*float64(i) + 0.4*ar
|
|
}
|
|
yArr, err := tensor.FromFloats(y, n)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
|
|
// The design carries its own intercept column, the convention of
|
|
// the classic linear model.
|
|
design := make([]float64, 2*n)
|
|
for i := range n {
|
|
design[2*i] = 1
|
|
design[2*i+1] = float64(i)
|
|
}
|
|
xArr, err := tensor.FromFloats(design, n, 2)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
fit, err := stats.LinearRegression(xArr, yArr)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
|
|
fmt.Println("ordinary least squares fit, y = intercept + slope * t")
|
|
fmt.Println("term estimate std error t-stat p-value")
|
|
fmt.Printf("intercept %9.4f %9.4f %7.3f %.3g\n",
|
|
fit.Coefficients[0], fit.StandardErrors[0], fit.TStatistics[0], fit.PValues[0])
|
|
fmt.Printf("slope %9.4f %9.4f %7.3f %.3g\n",
|
|
fit.Coefficients[1], fit.StandardErrors[1], fit.TStatistics[1], fit.PValues[1])
|
|
fmt.Printf("\nR² = %.4f, adjusted R² = %.4f, residual variance = %.4f\n",
|
|
fit.RSquared, fit.AdjustedRSquared, fit.ResidualVariance)
|
|
fmt.Println("(the generating values were intercept 2, slope 0.05)")
|
|
|
|
// The t-tests assume uncorrelated residuals. Pull them out and
|
|
// check the autocorrelation at the first few lags; with rho = 0.3
|
|
// in the noise, lag 1 must show clear correlation, which is the
|
|
// honest caveat for the standard errors above.
|
|
resid := make([]float64, n)
|
|
for i := range n {
|
|
pred := fit.Coefficients[0] + fit.Coefficients[1]*float64(i)
|
|
resid[i] = y[i] - pred
|
|
}
|
|
rArr, err := tensor.FromFloats(resid, n)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
ac, err := signal.Autocorrelate(rArr, 5)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
// The transform returns lags 0..5; lag 0 is 1 by definition, the
|
|
// AR(1) memory shows from lag 1 on.
|
|
fmt.Print("\nresidual autocorrelation:")
|
|
for lag := 1; lag <= 5; lag++ {
|
|
v, _ := tensor.FloatAt(ac, lag)
|
|
fmt.Printf(" lag %d: %+.3f", lag, v)
|
|
}
|
|
fmt.Println()
|
|
|
|
// A two-sample test on the first and last halves: with a trend of
|
|
// 0.05 over 200 samples the means must differ decisively.
|
|
first, err := tensor.Slice(yArr, 0, 0, n/2)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
last, err := tensor.Slice(yArr, 0, n/2, n)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
t, df, p, err := stats.WelchTTest(first, last)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
meanOf := func(a *tensor.Array) float64 {
|
|
m, err := tensor.Mean(a)
|
|
if err != nil {
|
|
log.Fatal(err)
|
|
}
|
|
return m
|
|
}
|
|
fmt.Printf("\nWelch t-test, first half vs second half:\n")
|
|
fmt.Printf(" means %.3f vs %.3f, t = %.2f, df = %.1f, p = %.3g\n",
|
|
meanOf(first), meanOf(last), t, df, p)
|
|
}
|