Files
tensor/stats/distrib2.go
T
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

434 lines
15 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package stats
import (
"sourcedock.dev/petrbalvin/tensor/internal/base"
"sourcedock.dev/petrbalvin/tensor/internal/core"
)
import (
"math"
)
// Distribution additions to the families of distrib.go and cdf.go: the
// Weibull, lognormal, Pareto and negative binomial laws, and the
// Dirichlet. The continuous laws carry closed density, CDF and
// quantile forms; the negative binomial follows the discrete house
// shape of Poisson and the binomial, counting on the same integer axis
// the Poisson counts on. The support convention of the existing
// distributions holds throughout: a CDF is 0 below the support and a
// density 0 outside it, while a faulty parameter is an error and never
// a silent value.
// WeibullDensity returns the Weibull density with shape k > 0 and
// scale λ > 0 at x ≥ 0: (k/λ)(x/λ)^{k−1}e^{−(x/λ)^k}. Below the
// support, and at +∞, the density is 0; at x = 0 the formula speaks
// for itself, 0 for k > 1, the finite 1/λ for k = 1 and the +Inf the
// integrable singularity carries for k < 1.
func WeibullDensity(x, k, lambda float64) (float64, error) {
if !(k > 0) || math.IsInf(k, 0) {
return 0, base.Errf("WeibullDensity: shape k must be finite and positive, got %g", k)
}
if !(lambda > 0) || math.IsInf(lambda, 0) {
return 0, base.Errf("WeibullDensity: scale λ must be finite and positive, got %g", lambda)
}
if math.IsNaN(x) {
return 0, base.Errf("WeibullDensity: x must be a number, got %g", x)
}
if x < 0 || math.IsInf(x, 1) {
return 0, nil
}
z := x / lambda
return k / lambda * math.Pow(z, k-1) * math.Exp(-math.Pow(z, k)), nil
}
// WeibullCDF returns P(X ≤ x) for X ~ Weibull(k, λ), the closed form
// 1 − e^{−(x/λ)^k}.
func WeibullCDF(x, k, lambda float64) (float64, error) {
if !(k > 0) || math.IsInf(k, 0) {
return 0, base.Errf("WeibullCDF: shape k must be finite and positive, got %g", k)
}
if !(lambda > 0) || math.IsInf(lambda, 0) {
return 0, base.Errf("WeibullCDF: scale λ must be finite and positive, got %g", lambda)
}
if math.IsNaN(x) {
return 0, base.Errf("WeibullCDF: x must be a number, got %g", x)
}
if x <= 0 {
return 0, nil
}
// Expm1 keeps the left tail, as in ExponentialCDF.
return -math.Expm1(-math.Pow(x/lambda, k)), nil
}
// WeibullQuantile returns the q-quantile of Weibull(k, λ), the closed
// form λ(−ln(1−q))^{1/k}.
func WeibullQuantile(q, k, lambda float64) (float64, error) {
if !(k > 0) || math.IsInf(k, 0) {
return 0, base.Errf("WeibullQuantile: shape k must be finite and positive, got %g", k)
}
if !(lambda > 0) || math.IsInf(lambda, 0) {
return 0, base.Errf("WeibullQuantile: scale λ must be finite and positive, got %g", lambda)
}
// NaN-rejecting on purpose, as in NormalQuantile.
if !(q >= 0 && q <= 1) {
return 0, base.Errf("WeibullQuantile: q must lie in [0, 1], got %g", q)
}
if q == 0 || q == 1 {
return 0, base.Errf("WeibullQuantile: q = %g has no finite quantile", q)
}
return lambda * math.Pow(-math.Log1p(-q), 1/k), nil
}
// LognormalDensity returns the lognormal density with location μ and
// log-scale σ > 0 at x > 0: 1/(xσ√(2π))e^{−(ln x−μ)²/(2σ²)}. The
// support convention gives 0 at x ≤ 0 and at +∞.
func LognormalDensity(x, mu, sigma float64) (float64, error) {
if math.IsNaN(mu) || math.IsInf(mu, 0) {
return 0, base.Errf("LognormalDensity: location μ must be finite, got %g", mu)
}
if !(sigma > 0) || math.IsInf(sigma, 0) {
return 0, base.Errf("LognormalDensity: log-scale σ must be finite and positive, got %g", sigma)
}
if math.IsNaN(x) {
return 0, base.Errf("LognormalDensity: x must be a number, got %g", x)
}
if x <= 0 || math.IsInf(x, 1) {
return 0, nil
}
z := (math.Log(x) - mu) / sigma
// The tail is assembled in log space: at a subnormal x both the
// numerator and the denominator of the closed form underflow to
// exact zero, and their division reports NaN for a point inside
// the support whose density is an honest 0.
return math.Exp(-z*z/2 - math.Log(x) - math.Log(sigma) - 0.5*math.Log(2*math.Pi)), nil
}
// LognormalCDF returns P(X ≤ x) for X ~ lognormal(μ, σ), the normal
// CDF at (ln x − μ)/σ, the reduction the law is named for.
func LognormalCDF(x, mu, sigma float64) (float64, error) {
if math.IsNaN(mu) || math.IsInf(mu, 0) {
return 0, base.Errf("LognormalCDF: location μ must be finite, got %g", mu)
}
if !(sigma > 0) || math.IsInf(sigma, 0) {
return 0, base.Errf("LognormalCDF: log-scale σ must be finite and positive, got %g", sigma)
}
if math.IsNaN(x) {
return 0, base.Errf("LognormalCDF: x must be a number, got %g", x)
}
if x <= 0 {
return 0, nil
}
return NormalCDF((math.Log(x) - mu) / sigma), nil
}
// LognormalQuantile returns the q-quantile of lognormal(μ, σ) through
// the existing normal quantile: e^{μ + σ·Φ^{−1}(q)}.
func LognormalQuantile(q, mu, sigma float64) (float64, error) {
if math.IsNaN(mu) || math.IsInf(mu, 0) {
return 0, base.Errf("LognormalQuantile: location μ must be finite, got %g", mu)
}
if !(sigma > 0) || math.IsInf(sigma, 0) {
return 0, base.Errf("LognormalQuantile: log-scale σ must be finite and positive, got %g", sigma)
}
z, err := NormalQuantile(q)
if err != nil {
return 0, base.Errf("LognormalQuantile: %w", err)
}
return math.Exp(mu + sigma*z), nil
}
// ParetoDensity returns the Pareto density with scale x_m > 0 and tail
// index α > 0 at x ≥ x_m: α·x_m^α/x^{α+1}. Below the support, and at
// +∞, the density is 0.
func ParetoDensity(x, xm, alpha float64) (float64, error) {
if !(xm > 0) || math.IsInf(xm, 0) {
return 0, base.Errf("ParetoDensity: scale x_m must be finite and positive, got %g", xm)
}
if !(alpha > 0) || math.IsInf(alpha, 0) {
return 0, base.Errf("ParetoDensity: tail index α must be finite and positive, got %g", alpha)
}
if math.IsNaN(x) {
return 0, base.Errf("ParetoDensity: x must be a number, got %g", x)
}
if x < xm || math.IsInf(x, 1) {
return 0, nil
}
return alpha / x * math.Pow(xm/x, alpha), nil
}
// ParetoCDF returns P(X ≤ x) for X ~ Pareto(x_m, α), the closed form
// 1 − (x_m/x)^α, evaluated through Expm1 so the answers just above the
// support keep their digits.
func ParetoCDF(x, xm, alpha float64) (float64, error) {
if !(xm > 0) || math.IsInf(xm, 0) {
return 0, base.Errf("ParetoCDF: scale x_m must be finite and positive, got %g", xm)
}
if !(alpha > 0) || math.IsInf(alpha, 0) {
return 0, base.Errf("ParetoCDF: tail index α must be finite and positive, got %g", alpha)
}
if math.IsNaN(x) {
return 0, base.Errf("ParetoCDF: x must be a number, got %g", x)
}
if x < xm {
return 0, nil
}
return -math.Expm1(alpha * math.Log(xm/x)), nil
}
// ParetoQuantile returns the q-quantile of Pareto(x_m, α), the closed
// form x_m(1−q)^{−1/α}.
func ParetoQuantile(q, xm, alpha float64) (float64, error) {
if !(xm > 0) || math.IsInf(xm, 0) {
return 0, base.Errf("ParetoQuantile: scale x_m must be finite and positive, got %g", xm)
}
if !(alpha > 0) || math.IsInf(alpha, 0) {
return 0, base.Errf("ParetoQuantile: tail index α must be finite and positive, got %g", alpha)
}
if !(q >= 0 && q <= 1) {
return 0, base.Errf("ParetoQuantile: q must lie in [0, 1], got %g", q)
}
if q == 0 || q == 1 {
return 0, base.Errf("ParetoQuantile: q = %g has no finite quantile", q)
}
return xm * math.Pow(1-q, -1/alpha), nil
}
// NegativeBinomialPMF returns P(X = k), the probability of k failures
// before the r-th success in independent trials of probability p: the
// law the Poisson draws and BinomialDraws count on the same integer
// axis. The mass is assembled in log space with lgamma, the form that
// keeps every term representable for large r and k.
func NegativeBinomialPMF(k, r int, p float64) (float64, error) {
if r < 1 {
return 0, base.Errf("NegativeBinomialPMF: r must be ≥ 1, got %d", r)
}
if !(p > 0 && p < 1) {
return 0, base.Errf("NegativeBinomialPMF: p must lie in (0, 1), got %g", p)
}
if k < 0 {
return 0, nil
}
lf := logGamma(float64(r+k)) - logGamma(float64(r)) - logGamma(float64(k+1)) +
float64(r)*math.Log(p) + float64(k)*math.Log1p(-p)
return math.Exp(lf), nil
}
// NegativeBinomialCDF returns P(X ≤ k) for the number of failures X
// before the r-th success, by direct summation of the PMF terms under
// the multiplicative recurrence term_{j+1} = term_j·(r+j)/(j+1)·(1−p).
// Every term is positive, so the sum carries no cancellation; the same
// probability equals the regularised beta I_p(r, k+1), which the tests
// hold the summation against. The summation seeds from p^r, and a seed
// the format cannot hold rounds to an exact zero the recurrence never
// recovers from: every later term would stay zero while the true mass
// sits further out. That far regime answers through the beta identity
// instead, which keeps the whole support live.
func NegativeBinomialCDF(k, r int, p float64) (float64, error) {
if r < 1 {
return 0, base.Errf("NegativeBinomialCDF: r must be ≥ 1, got %d", r)
}
if !(p > 0 && p < 1) {
return 0, base.Errf("NegativeBinomialCDF: p must lie in (0, 1), got %g", p)
}
if k < 0 {
return 0, nil
}
term := math.Exp(float64(r) * math.Log(p))
if term == 0 {
// p^r underflowed: the recurrence multiplies zeros, so the sum
// would answer 0 at every k. The identity I_p(r, k+1) = P(X ≤ k)
// evaluates the same probability through the incomplete beta,
// accurate across this regime.
return BetaIncomplete(p, float64(r), float64(k+1))
}
sum := 0.0
for j := 0; j <= k; j++ {
sum += term
term *= (float64(r) + float64(j)) / float64(j+1) * (1 - p)
}
return sum, nil
}
// NegativeBinomialQuantile returns the smallest k with P(X ≤ k) ≥ q,
// through the same discrete bracketed search the Poisson and binomial
// quantiles use.
func NegativeBinomialQuantile(q, p float64, r int) (float64, error) {
if r < 1 {
return 0, base.Errf("NegativeBinomialQuantile: r must be ≥ 1, got %d", r)
}
if !(p > 0 && p < 1) {
return 0, base.Errf("NegativeBinomialQuantile: p must lie in (0, 1), got %g", p)
}
return discreteQuantile("NegativeBinomialQuantile", q, func(k int) (float64, error) {
return NegativeBinomialCDF(k, r, p)
})
}
// logGamma wraps math.Lgamma, keeping the log-space call sites free of
// the ignored-error idiom.
func logGamma(x float64) float64 {
l, _ := math.Lgamma(x)
return l
}
// DirichletDensity returns the Dirichlet density with concentration
// vector α at the simplex point x: Πx_i^{α_i−1}/B(α), the normalising
// constant assembled with lgamma. Both vectors must be finite, every
// α_i positive and every x_i non-negative, and the x must sum to 1 (to
// within 1e-9); a boundary x_i = 0 gives +Inf below α_i = 1, the value
// 1 continues to contribute nothing at α_i = 1 exactly, and 0 above.
func DirichletDensity(alpha, x []float64) (float64, error) {
const name = "DirichletDensity"
if len(alpha) < 2 {
return 0, base.Errf("%s: needs at least two components, got %d", name, len(alpha))
}
if len(x) != len(alpha) {
return 0, base.Errf("%s: the concentration has %d components, the point %d", name, len(alpha), len(x))
}
total := 0.0
for i, a := range alpha {
if !(a > 0) || math.IsInf(a, 0) {
return 0, base.Errf("%s: alpha[%d] must be finite and positive, got %g", name, i, a)
}
total += a
}
sum := 0.0
for i, v := range x {
if math.IsNaN(v) || math.IsInf(v, 0) {
return 0, base.Errf("%s: x[%d] must be finite, got %g", name, i, v)
}
if v < 0 {
return 0, base.Errf("%s: x[%d] = %g lies outside the simplex", name, i, v)
}
sum += v
}
if math.Abs(sum-1) > 1e-9 {
return 0, base.Errf("%s: the point must sum to 1, got %g", name, sum)
}
// ln B(α) = Σ lgamma(α_i) − lgamma(α₀).
lb := -logGamma(total)
for _, a := range alpha {
lb += logGamma(a)
}
s := -lb
for i, a := range alpha {
switch {
case x[i] == 0:
if a < 1 {
return math.Inf(1), nil
}
if a > 1 {
return 0, nil
}
default:
s += (a - 1) * math.Log(x[i])
}
}
return math.Exp(s), nil
}
// DirichletMean returns the mean of the Dirichlet with concentration
// α: the normalised concentration α_i/α₀.
func DirichletMean(alpha []float64) ([]float64, error) {
if len(alpha) < 2 {
return nil, base.Errf("DirichletMean: needs at least two components, got %d", len(alpha))
}
total := 0.0
for i, a := range alpha {
if !(a > 0) || math.IsInf(a, 0) {
return nil, base.Errf("DirichletMean: alpha[%d] must be finite and positive, got %g", i, a)
}
total += a
}
mean := make([]float64, len(alpha))
for i, a := range alpha {
mean[i] = a / total
}
return mean, nil
}
// DirichletMode returns the interior mode (α_i−1)/(α₀−k), which exists
// only when every concentration exceeds 1; any α_i ≤ 1 pushes the mode
// onto the boundary and is refused rather than answered with a vector
// that is not a mode.
func DirichletMode(alpha []float64) ([]float64, error) {
if len(alpha) < 2 {
return nil, base.Errf("DirichletMode: needs at least two components, got %d", len(alpha))
}
total := 0.0
for i, a := range alpha {
if !(a > 0) || math.IsInf(a, 0) {
return nil, base.Errf("DirichletMode: alpha[%d] must be finite and positive, got %g", i, a)
}
if a <= 1 {
return nil, base.Errf("DirichletMode: alpha[%d] = %g leaves no interior mode; every concentration must exceed 1", i, a)
}
total += a
}
den := total - float64(len(alpha))
mode := make([]float64, len(alpha))
for i, a := range alpha {
mode[i] = (a - 1) / den
}
return mode, nil
}
// DirichletDraws returns n draws from the Dirichlet with concentration
// α, as an (n, k) array whose rows are the draws. Each row scales the
// k independent gamma(α_i, 1) draws of GammaDraws' generator to sum to
// one; the all-underflow row (every α_i far below the float64 floor)
// would otherwise divide by zero and falls back to the uniform row.
func DirichletDraws(g *core.Generator, n int, alpha []float64) (*core.Array, error) {
const name = "DirichletDraws"
if n < 1 {
return nil, base.Errf("%s: n must be ≥ 1", name)
}
if len(alpha) < 2 {
return nil, base.Errf("%s: needs at least two components, got %d", name, len(alpha))
}
for i, a := range alpha {
if !(a > 0) || math.IsInf(a, 0) {
return nil, base.Errf("%s: alpha[%d] must be finite and positive, got %g", name, i, a)
}
}
k := len(alpha)
flat := make([]float64, n*k)
for r := range n {
sum := 0.0
row := flat[r*k : r*k+k]
for c, a := range alpha {
row[c] = gammaOne(g, a)
sum += row[c]
}
if sum == 0 {
// Every gamma draw underflowed: the uniform row is the
// honest stand-in, an Inf row would poison the draw.
for c := range row {
row[c] = 1 / float64(k)
}
continue
}
for c := range row {
row[c] /= sum
}
}
return floatsToArray(flat, []int{n, k}), nil
}
// gammaOne draws one gamma(α, 1) variate, the single-draw form of the
// GammaDraws loop: Marsaglia-Tsang for α ≥ 1 and the boost with the
// exponential for α < 1, reusing the package squeeze.
func gammaOne(g *core.Generator, alpha float64) float64 {
if alpha >= 1 {
d := alpha - 1.0/3
return gammaMarsagliaTsang(g, alpha, d, 1/math.Sqrt(9*d))
}
boost := alpha + 1
d := boost - 1.0/3
v := gammaMarsagliaTsang(g, boost, d, 1/math.Sqrt(9*d))
return v * math.Pow(g.Unit(), 1/alpha)
}