// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package engine import ( "sync" "testing" ) // TestParallelCoversEveryIndexExactlyOnce is the core invariant: the // chunks partition [0, n), whatever the worker count does to their // boundaries. func TestParallelCoversEveryIndexExactlyOnce(t *testing.T) { for _, n := range []int{0, 1, 2, 7, 33, 100, 1024} { touched := make([]int, n) // The chunk check records violations instead of calling // Fatalf from inside the worker goroutines: FailNow is defined // for the test's own goroutine only. The append sits under a // mutex because the chunks run concurrently. var mu sync.Mutex illegal := make([][3]int, 0, 4) Parallel(n, func(start, end int) { if start < 0 || end > n || start > end { mu.Lock() illegal = append(illegal, [3]int{n, start, end}) mu.Unlock() } for i := start; i < end; i++ { touched[i]++ } }) if len(illegal) > 0 { t.Fatalf("illegal chunks: %v", illegal) } for i, c := range touched { if c != 1 && n > 0 { t.Fatalf("n=%d: index %d visited %d times", n, i, c) } } } } // TestParallelSmallWorkloadRunsInline pins the small-workload rule: // with a single effective worker the callback runs on the caller's // goroutine before Parallel returns: no goroutine churn for tiny // kernels. A panicked chunk therefore crashes this test instead of // hiding behind the WaitGroup. func TestParallelSmallWorkloadRunsInline(t *testing.T) { prev := SetNumWorkers(1) defer SetNumWorkers(prev) called := false Parallel(4, func(start, end int) { called = true if start != 0 || end != 4 { t.Fatalf("single-worker chunk [%d, %d), want [0, 4)", start, end) } }) if !called { t.Fatal("callback never ran") } } // TestWorkersForBounds checks both ceilings and the floor. func TestWorkersForBounds(t *testing.T) { prev := SetNumWorkers(8) defer SetNumWorkers(prev) for _, tc := range []struct{ n, want int }{ {0, 1}, {1, 1}, {3, 3}, {8, 8}, {500, 8}, } { if got := WorkersFor(tc.n); got != tc.want { t.Errorf("WorkersFor(%d) = %d, want %d", tc.n, got, tc.want) } } if got := SetNumWorkers(0); got != 8 { t.Errorf("SetNumWorkers(0) reported previous %d, want 8", got) } if NumWorkers() < 1 { t.Error("reset landed on an unusable worker count") } } // TestFloat64PoolRoundTripKeepsLengthAndCapacity pins the borrow // contract: every GetFloat64Buf returns exactly the requested length // with capacity for at least that many, buffers survive a Put/Get round // trip as usable memory, and a buffer handed back dirty arrives cleared. // The pool enforces the zero-on-borrow guarantee itself, so no kernel // can leak an earlier borrower's sums into its result (the regression // TestMatMul2DFloat32ScratchCleaned pins end-to-end in the tensor // package). func TestFloat64PoolRoundTripKeepsLengthAndCapacity(t *testing.T) { buf := GetFloat64Buf(64) if len(buf) != 64 { t.Fatalf("borrowed len %d, want 64", len(buf)) } if cap(buf) < 64 { t.Fatalf("borrowed cap %d, want at least 64", cap(buf)) } for i := range buf { buf[i] = float64(i) // fill the whole window: prove it is writable } PutFloat64Buf(buf) again := GetFloat64Buf(32) if len(again) != 32 { t.Fatalf("re-borrowed len %d, want 32", len(again)) } if cap(again) < 32 { t.Fatalf("re-borrowed capacity %d, want at least 32", cap(again)) } // The dirty residue the test just put back must never surface: the // pool clears on borrow, so the window arrives all zeros. (sync.Pool // may also drop the buffer at any GC, in which case a fresh (and // therefore zeroed) allocation takes its place; the guarantee holds // on both paths.) for i := range again { if again[i] != 0 { t.Fatalf("slot %d = %v on arrival, want 0", i, again[i]) } again[i] = float64(i) if again[i] != float64(i) { t.Fatalf("slot %d = %v after write, want %v", i, again[i], float64(i)) } } PutFloat64Buf(again) } // TestFloat64PoolGrowsForLargerBorrow checks the grow path returns a // slice of exactly the requested length even when the pooled buffer // must be reallocated. func TestFloat64PoolGrowsForLargerBorrow(t *testing.T) { small := GetFloat64Buf(4) PutFloat64Buf(small) big := GetFloat64Buf(4096) if len(big) != 4096 { t.Fatalf("grown len %d, want 4096", len(big)) } big[4095] = 1 // writable end to end PutFloat64Buf(big) } // TestFloat64PoolRetentionCap pins the size rule: the cap itself // round-trips, anything above it is dropped rather than retained per // processor, and the drop path leaves the pool usable. func TestFloat64PoolRetentionCap(t *testing.T) { if !keepPooled(maxPooledFloat64) { t.Fatalf("a buffer of the cap (%d) must be retained", maxPooledFloat64) } if keepPooled(maxPooledFloat64 + 1) { t.Fatalf("a buffer above the cap (%d) must be dropped", maxPooledFloat64+1) } oversized := GetFloat64Buf(4 * maxPooledFloat64) oversized[len(oversized)-1] = 1 PutFloat64Buf(oversized) // dropped: must not corrupt the pool next := GetFloat64Buf(16) if len(next) != 16 { t.Fatalf("borrowed after a dropped buffer: len %d, want 16", len(next)) } for i := range next { if next[i] != 0 { t.Fatalf("slot %d = %v after a dropped buffer, want 0", i, next[i]) } } PutFloat64Buf(next) }