297 lines
11 KiB
Go
297 lines
11 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: MIT
|
||
|
||
package stats
|
||
|
||
import (
|
||
"math"
|
||
"strings"
|
||
"testing"
|
||
|
||
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
||
)
|
||
|
||
// quantileSubgradient evaluates the check loss's subgradient at a fit
|
||
// and the slack the zero-band residuals buy it. At the optimum of
|
||
// Σρ_τ(y − Xβ) there must be a choice of s ∈ [τ−1, τ] per observation
|
||
// with Xᵀs = 0; residuals above the band fix their s (τ above the
|
||
// fit, τ−1 below it), and residuals within 1e-6 of zero, where no
|
||
// floating point fit can place them exactly, are free to slide in
|
||
// [τ−1, τ]. The certificate passes when every column's forced
|
||
// gradient is within the slack its zero-band residuals can absorb.
|
||
func quantileSubgradient(t *testing.T, x, y *core.Array, beta []float64, tau float64) (grad, free []float64) {
|
||
t.Helper()
|
||
const band = 1e-6
|
||
n, p := x.Shape()[0], x.Shape()[1]
|
||
grad = make([]float64, p)
|
||
free = make([]float64, p)
|
||
for i := range n {
|
||
r := y.FloatAt(i)
|
||
for j := range p {
|
||
r -= beta[j] * x.FloatAt(i*p+j)
|
||
}
|
||
var s float64
|
||
switch {
|
||
case r > band:
|
||
s = tau
|
||
case r < -band:
|
||
s = tau - 1
|
||
default:
|
||
for j := range p {
|
||
free[j] += math.Abs(x.FloatAt(i*p + j))
|
||
}
|
||
continue
|
||
}
|
||
for j := range p {
|
||
grad[j] += x.FloatAt(i*p+j) * s
|
||
}
|
||
}
|
||
for j := range p {
|
||
free[j] *= math.Max(tau, 1-tau)
|
||
}
|
||
return grad, free
|
||
}
|
||
|
||
// quantileHeteroFixture builds the seeded heteroscedastic model the
|
||
// coverage and monotonicity pins share: x uniform on [−2, 2], the
|
||
// response moved by 1 + 2x and widened by a scale that grows with
|
||
// |x|, so the tau-th conditional quantile is a genuinely different
|
||
// line from the mean's and no single slope fits every tau.
|
||
func quantileHeteroFixture(t *testing.T, n int, seed int64) (*core.Array, *core.Array) {
|
||
t.Helper()
|
||
g := core.NewGenerator(seed)
|
||
vals := make([]float64, 0, 2*n)
|
||
yv := make([]float64, 0, n)
|
||
for range n {
|
||
x := 4*g.Unit() - 2
|
||
y := 1 + 2*x + (0.3+0.3*math.Abs(x))*g.NormalUnit()
|
||
vals = append(vals, 1, x)
|
||
yv = append(yv, y)
|
||
}
|
||
return mustFromFloats(t, vals, n, 2), mustFromFloats(t, yv, n)
|
||
}
|
||
|
||
// TestQuantileMedianCertificate pins tau = 0.5 against the L1 median
|
||
// regression answer, twice over. An intercept-only design has the
|
||
// sample median for its answer, and the general design is verified
|
||
// against the subgradient optimality certificate, which is the exact
|
||
// first-order condition of the check loss rather than another
|
||
// algorithm's output.
|
||
func TestQuantileMedianCertificate(t *testing.T) {
|
||
// Intercept only, odd sample: the fit is the median, the single
|
||
// point every absolute deviation bends toward.
|
||
const n = 25
|
||
g := core.NewGenerator(19)
|
||
ys := make([]float64, n)
|
||
for i := range n {
|
||
ys[i] = math.Round(40*g.Unit() - 20)
|
||
}
|
||
y := mustFromFloats(t, ys, n)
|
||
ones := make([]float64, n)
|
||
for i := range n {
|
||
ones[i] = 1
|
||
}
|
||
design := mustFromFloats(t, ones, n, 1)
|
||
res, err := QuantileRegression(design, y, 0.5)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression: %v", err)
|
||
}
|
||
if !res.Converged {
|
||
t.Fatalf("the intercept-only fit did not converge")
|
||
}
|
||
median, err := Median(y)
|
||
if err != nil {
|
||
t.Fatalf("Median: %v", err)
|
||
}
|
||
if math.Abs(res.Coefficients[0]-median) > 1e-7 {
|
||
t.Fatalf("the tau = 0.5 intercept-only fit is %.12g, want the median %.12g", res.Coefficients[0], median)
|
||
}
|
||
// A general design against the subgradient certificate.
|
||
xs := make([]float64, 0, 2*n)
|
||
yv := make([]float64, 0, n)
|
||
for range n {
|
||
x := 4*g.Unit() - 2
|
||
yv = append(yv, 1+2*x+0.4*g.NormalUnit())
|
||
xs = append(xs, 1, x)
|
||
}
|
||
design2 := mustFromFloats(t, xs, n, 2)
|
||
y2 := mustFromFloats(t, yv, n)
|
||
fit, err := QuantileRegression(design2, y2, 0.5)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression: %v", err)
|
||
}
|
||
grad, free := quantileSubgradient(t, design2, y2, fit.Coefficients, 0.5)
|
||
for j := range 2 {
|
||
t.Logf("tau = 0.5 column %d: forced gradient %.3g against slack %.3g", j, grad[j], free[j])
|
||
if math.Abs(grad[j]) > free[j]+1e-6 {
|
||
t.Fatalf("column %d fails the subgradient certificate: %.3g against slack %.3g", j, grad[j], free[j])
|
||
}
|
||
}
|
||
}
|
||
|
||
// TestQuantileObjectiveMonotone instruments the interior-point run on
|
||
// the heteroscedastic fixture: the recorded objective is the running
|
||
// best check loss, monotone non-increasing from the ordinary least
|
||
// squares start, and the record has to show real descent: the
|
||
// tau = 0.9 quantile line is not the mean line, and the check loss at
|
||
// the optimum must sit measurably below the start.
|
||
func TestQuantileObjectiveMonotone(t *testing.T) {
|
||
design, y := quantileHeteroFixture(t, 400, 23)
|
||
fit, err := QuantileRegression(design, y, 0.9)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression: %v", err)
|
||
}
|
||
if !fit.Converged {
|
||
t.Fatalf("the interior-point run did not converge")
|
||
}
|
||
if len(fit.Objective) < 3 {
|
||
t.Fatalf("the run recorded only %d objectives, too few to show a descent", len(fit.Objective))
|
||
}
|
||
for k := 1; k < len(fit.Objective); k++ {
|
||
if fit.Objective[k] > fit.Objective[k-1] {
|
||
t.Fatalf("the objective rose at record %d: %.12g after %.12g",
|
||
k, fit.Objective[k], fit.Objective[k-1])
|
||
}
|
||
}
|
||
if math.Abs(fit.Objective[len(fit.Objective)-1]-fit.CheckLoss) > 1e-9 {
|
||
t.Fatalf("the record ends at %.12g but the fit reports %.12g",
|
||
fit.Objective[len(fit.Objective)-1], fit.CheckLoss)
|
||
}
|
||
t.Logf("check loss descended from %.6f to %.6f over %d iterations",
|
||
fit.Objective[0], fit.CheckLoss, fit.Iterations)
|
||
if fit.Objective[0]-fit.CheckLoss < 1 {
|
||
t.Fatalf("the tau = 0.9 fit barely left the mean fit: descent %.6g", fit.Objective[0]-fit.CheckLoss)
|
||
}
|
||
}
|
||
|
||
// TestQuantileTracksConditionalQuantile measures the fit where it
|
||
// matters: on six hundred held-out draws of the heteroscedastic model,
|
||
// the tau = 0.9 line must cover ninety percent of the responses and
|
||
// the tau = 0.1 line ten percent, the defining property of a
|
||
// conditional quantile estimate.
|
||
func TestQuantileTracksConditionalQuantile(t *testing.T) {
|
||
design, y := quantileHeteroFixture(t, 400, 23)
|
||
fit90, err := QuantileRegression(design, y, 0.9)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression tau 0.9: %v", err)
|
||
}
|
||
fit10, err := QuantileRegression(design, y, 0.1)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression tau 0.1: %v", err)
|
||
}
|
||
if !fit90.Converged || !fit10.Converged {
|
||
t.Fatalf("a coverage fit did not converge")
|
||
}
|
||
// The fitted slopes track the true conditional quantile slope 2.
|
||
if math.Abs(fit90.Coefficients[1]-2) > 0.5 {
|
||
t.Fatalf("the tau = 0.9 slope is %.4f, far from the truth 2", fit90.Coefficients[1])
|
||
}
|
||
g := core.NewGenerator(97)
|
||
const held = 600
|
||
below90, below10 := 0, 0
|
||
for range held {
|
||
x := 4*g.Unit() - 2
|
||
yi := 1 + 2*x + (0.3+0.3*math.Abs(x))*g.NormalUnit()
|
||
q90 := fit90.Coefficients[0] + fit90.Coefficients[1]*x
|
||
q10 := fit10.Coefficients[0] + fit10.Coefficients[1]*x
|
||
if yi <= q90 {
|
||
below90++
|
||
}
|
||
if yi <= q10 {
|
||
below10++
|
||
}
|
||
}
|
||
coverage90 := float64(below90) / held
|
||
coverage10 := float64(below10) / held
|
||
t.Logf("held-out coverage: tau = 0.9 covers %.3f, tau = 0.1 covers %.3f", coverage90, coverage10)
|
||
if coverage90 < 0.84 || coverage90 > 0.96 {
|
||
t.Fatalf("the tau = 0.9 fit covers %.3f of held-out draws, want near 0.9", coverage90)
|
||
}
|
||
if coverage10 < 0.04 || coverage10 > 0.16 {
|
||
t.Fatalf("the tau = 0.1 fit covers %.3f of held-out draws, want near 0.1", coverage10)
|
||
}
|
||
}
|
||
|
||
// TestQuantileExactFit walks the zero-loss corner: a constant response
|
||
// is reproduced exactly by the start, the check loss is zero, and the
|
||
// run is settled before the first interior-point iteration.
|
||
func TestQuantileExactFit(t *testing.T) {
|
||
design := mustFromFloats(t, []float64{1, 0, 1, 1, 1, 2, 1, 3, 1, 4}, 5, 2)
|
||
y := mustFromFloats(t, []float64{4.2, 4.2, 4.2, 4.2, 4.2}, 5)
|
||
res, err := QuantileRegression(design, y, 0.3)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression: %v", err)
|
||
}
|
||
if !res.Converged {
|
||
t.Fatalf("the exact fit did not report convergence")
|
||
}
|
||
if res.CheckLoss != 0 {
|
||
t.Fatalf("the exact fit reports check loss %g, want 0", res.CheckLoss)
|
||
}
|
||
if res.Iterations != 0 {
|
||
t.Fatalf("the exact fit spent %d iterations, want 0", res.Iterations)
|
||
}
|
||
for i := range 5 {
|
||
if math.Abs(res.Fitted[i]-4.2) > 1e-9 || math.Abs(res.Residuals[i]) > 1e-9 {
|
||
t.Fatalf("the exact fit moved row %d: fitted %.12g", i, res.Fitted[i])
|
||
}
|
||
}
|
||
}
|
||
|
||
// TestQuantileOnIntegerArrays exercises the widening accessor's
|
||
// fallback paths in the loss and the recovery: an integer design and
|
||
// response reach the fit through FloatAt rather than a raw float
|
||
// payload.
|
||
func TestQuantileOnIntegerArrays(t *testing.T) {
|
||
design := mustFromInts(t, []int64{1, 0, 1, 1, 1, 2, 1, 3, 1, 4, 1, 5, 1, 6}, 7, 2)
|
||
y := mustFromInts(t, []int64{2, 3, 4, 5, 6, 7, 8}, 7)
|
||
res, err := QuantileRegression(design, y, 0.75)
|
||
if err != nil {
|
||
t.Fatalf("QuantileRegression on integer input: %v", err)
|
||
}
|
||
if !res.Converged {
|
||
t.Fatalf("the integer-input fit did not converge")
|
||
}
|
||
grad, free := quantileSubgradient(t, design, y, res.Coefficients, 0.75)
|
||
for j := range 2 {
|
||
if math.Abs(grad[j]) > free[j]+1e-6 {
|
||
t.Fatalf("column %d fails the subgradient certificate: %.3g against slack %.3g", j, grad[j], free[j])
|
||
}
|
||
}
|
||
}
|
||
|
||
// TestQuantileValidation refuses the malformed inputs: tau outside the
|
||
// open interval, wrong shapes, non-finite samples and a rank-deficient
|
||
// design.
|
||
func TestQuantileValidation(t *testing.T) {
|
||
design := mustFromFloats(t, []float64{1, 0, 1, 1, 1, 2, 1, 3, 1, 4, 1, 5}, 6, 2)
|
||
y := mustFromFloats(t, []float64{1, 3, 2, 5, 4, 7}, 6)
|
||
for _, tau := range []float64{0, 1, -0.5, 1.5, math.NaN()} {
|
||
if _, err := QuantileRegression(design, y, tau); err == nil || !strings.Contains(err.Error(), "strictly inside") {
|
||
t.Fatalf("tau = %g: got %v, want the tau refusal", tau, err)
|
||
}
|
||
}
|
||
if _, err := QuantileRegression(mustFromFloats(t, []float64{1, 2, 3, 4, 5, 6}, 6), y, 0.5); err == nil || !strings.Contains(err.Error(), "must be rank 2") {
|
||
t.Fatalf("a rank 1 design: got %v, want the rank refusal", err)
|
||
}
|
||
if _, err := QuantileRegression(design, mustFromFloats(t, []float64{1, 2, 3, 4, 5, 6}, 3, 2), 0.5); err == nil || !strings.Contains(err.Error(), "must be rank 1") {
|
||
t.Fatalf("a rank 2 response: got %v, want the rank refusal", err)
|
||
}
|
||
if _, err := QuantileRegression(design, mustFromFloats(t, []float64{1, 2, 3}, 3), 0.5); err == nil || !strings.Contains(err.Error(), "rows but the response") {
|
||
t.Fatalf("a length mismatch: got %v, want the length refusal", err)
|
||
}
|
||
if _, err := QuantileRegression(mustFromFloats(t, []float64{1, 0, 1, 1}, 2, 2), mustFromFloats(t, []float64{1, 2}, 2), 0.5); err == nil || !strings.Contains(err.Error(), "n > p") {
|
||
t.Fatalf("n = p: got %v, want the n > p refusal", err)
|
||
}
|
||
if _, err := QuantileRegression(design, mustFromFloats(t, []float64{1, 2, 3, math.NaN(), 5, 7}, 6), 0.5); err == nil || !strings.Contains(err.Error(), "non-finite") {
|
||
t.Fatalf("a non-finite response: got %v, want the non-finite refusal", err)
|
||
}
|
||
if _, err := QuantileRegression(core.New(core.Complex, 6, 2), y, 0.5); err == nil || !strings.Contains(err.Error(), "complex") {
|
||
t.Fatalf("complex input: got %v, want the complex refusal", err)
|
||
}
|
||
singular := mustFromFloats(t, []float64{1, 2, 3, 1, 2, 3, 1, 2, 3, 1, 2, 3}, 4, 3)
|
||
if _, err := QuantileRegression(singular, mustFromFloats(t, []float64{1, 2, 3, 4}, 4), 0.5); err == nil || !strings.Contains(err.Error(), "singular") {
|
||
t.Fatalf("a rank-deficient design: got %v, want the singular-system refusal", err)
|
||
}
|
||
}
|