feat: initial release
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-03 10:00:00 +02:00
commit af4ee19703
617 changed files with 191195 additions and 0 deletions
+171
View File
@@ -0,0 +1,171 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package engine
import (
"sync"
"testing"
)
// TestParallelCoversEveryIndexExactlyOnce is the core invariant: the
// chunks partition [0, n), whatever the worker count does to their
// boundaries.
func TestParallelCoversEveryIndexExactlyOnce(t *testing.T) {
for _, n := range []int{0, 1, 2, 7, 33, 100, 1024} {
touched := make([]int, n)
// The chunk check records violations instead of calling
// Fatalf from inside the worker goroutines: FailNow is defined
// for the test's own goroutine only. The append sits under a
// mutex because the chunks run concurrently.
var mu sync.Mutex
illegal := make([][3]int, 0, 4)
Parallel(n, func(start, end int) {
if start < 0 || end > n || start > end {
mu.Lock()
illegal = append(illegal, [3]int{n, start, end})
mu.Unlock()
}
for i := start; i < end; i++ {
touched[i]++
}
})
if len(illegal) > 0 {
t.Fatalf("illegal chunks: %v", illegal)
}
for i, c := range touched {
if c != 1 && n > 0 {
t.Fatalf("n=%d: index %d visited %d times", n, i, c)
}
}
}
}
// TestParallelSmallWorkloadRunsInline pins the small-workload rule:
// with a single effective worker the callback runs on the caller's
// goroutine before Parallel returns: no goroutine churn for tiny
// kernels. A panicked chunk therefore crashes this test instead of
// hiding behind the WaitGroup.
func TestParallelSmallWorkloadRunsInline(t *testing.T) {
prev := SetNumWorkers(1)
defer SetNumWorkers(prev)
called := false
Parallel(4, func(start, end int) {
called = true
if start != 0 || end != 4 {
t.Fatalf("single-worker chunk [%d, %d), want [0, 4)", start, end)
}
})
if !called {
t.Fatal("callback never ran")
}
}
// TestWorkersForBounds checks both ceilings and the floor.
func TestWorkersForBounds(t *testing.T) {
prev := SetNumWorkers(8)
defer SetNumWorkers(prev)
for _, tc := range []struct{ n, want int }{
{0, 1}, {1, 1}, {3, 3}, {8, 8}, {500, 8},
} {
if got := WorkersFor(tc.n); got != tc.want {
t.Errorf("WorkersFor(%d) = %d, want %d", tc.n, got, tc.want)
}
}
if got := SetNumWorkers(0); got != 8 {
t.Errorf("SetNumWorkers(0) reported previous %d, want 8", got)
}
if NumWorkers() < 1 {
t.Error("reset landed on an unusable worker count")
}
}
// TestFloat64PoolRoundTripKeepsLengthAndCapacity pins the borrow
// contract: every GetFloat64Buf returns exactly the requested length
// with capacity for at least that many, buffers survive a Put/Get round
// trip as usable memory, and a buffer handed back dirty arrives cleared.
// The pool enforces the zero-on-borrow guarantee itself, so no kernel
// can leak an earlier borrower's sums into its result (the regression
// TestMatMul2DFloat32ScratchCleaned pins end-to-end in the tensor
// package).
func TestFloat64PoolRoundTripKeepsLengthAndCapacity(t *testing.T) {
buf := GetFloat64Buf(64)
if len(buf) != 64 {
t.Fatalf("borrowed len %d, want 64", len(buf))
}
if cap(buf) < 64 {
t.Fatalf("borrowed cap %d, want at least 64", cap(buf))
}
for i := range buf {
buf[i] = float64(i) // fill the whole window: prove it is writable
}
PutFloat64Buf(buf)
again := GetFloat64Buf(32)
if len(again) != 32 {
t.Fatalf("re-borrowed len %d, want 32", len(again))
}
if cap(again) < 32 {
t.Fatalf("re-borrowed capacity %d, want at least 32", cap(again))
}
// The dirty residue the test just put back must never surface: the
// pool clears on borrow, so the window arrives all zeros. (sync.Pool
// may also drop the buffer at any GC, in which case a fresh (and
// therefore zeroed) allocation takes its place; the guarantee holds
// on both paths.)
for i := range again {
if again[i] != 0 {
t.Fatalf("slot %d = %v on arrival, want 0", i, again[i])
}
again[i] = float64(i)
if again[i] != float64(i) {
t.Fatalf("slot %d = %v after write, want %v", i, again[i], float64(i))
}
}
PutFloat64Buf(again)
}
// TestFloat64PoolGrowsForLargerBorrow checks the grow path returns a
// slice of exactly the requested length even when the pooled buffer
// must be reallocated.
func TestFloat64PoolGrowsForLargerBorrow(t *testing.T) {
small := GetFloat64Buf(4)
PutFloat64Buf(small)
big := GetFloat64Buf(4096)
if len(big) != 4096 {
t.Fatalf("grown len %d, want 4096", len(big))
}
big[4095] = 1 // writable end to end
PutFloat64Buf(big)
}
// TestFloat64PoolRetentionCap pins the size rule: the cap itself
// round-trips, anything above it is dropped rather than retained per
// processor, and the drop path leaves the pool usable.
func TestFloat64PoolRetentionCap(t *testing.T) {
if !keepPooled(maxPooledFloat64) {
t.Fatalf("a buffer of the cap (%d) must be retained", maxPooledFloat64)
}
if keepPooled(maxPooledFloat64 + 1) {
t.Fatalf("a buffer above the cap (%d) must be dropped", maxPooledFloat64+1)
}
oversized := GetFloat64Buf(4 * maxPooledFloat64)
oversized[len(oversized)-1] = 1
PutFloat64Buf(oversized) // dropped: must not corrupt the pool
next := GetFloat64Buf(16)
if len(next) != 16 {
t.Fatalf("borrowed after a dropped buffer: len %d, want 16", len(next))
}
for i := range next {
if next[i] != 0 {
t.Fatalf("slot %d = %v after a dropped buffer, want 0", i, next[i])
}
}
PutFloat64Buf(next)
}