// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: MIT package signal import ( "testing" "sourcedock.dev/petrbalvin/tensor/internal/core" ) func BenchmarkConv2DParallel(b *testing.B) { benchConv2DN(b) } func BenchmarkConv2DSerial(b *testing.B) { core.SetNumCPU(1) defer core.SetNumCPU(0) benchConv2DN(b) } func benchConv2DN(b *testing.B) { in, _ := core.FromFloats(make([]float64, 8*64*28*28), 8, 64, 28, 28) k, _ := core.FromFloats(make([]float64, 64*64*3*3), 64, 64, 3, 3) bi, _ := core.FromFloats(make([]float64, 64), 64) b.ResetTimer() for b.Loop() { if _, err := Conv2D(in, k, bi, 1, 1); err != nil { b.Fatal(err) } } } // FFT benchmarks: the radix-2 kernel at a power-of-two length, and the // 2-D separable path whose per-line scratch and payload accessors // dominate at practical sizes. Run with -bench before and after any // change to fft.go. func BenchmarkFFT4096(b *testing.B) { vals := make([]complex128, 4096) for i := range vals { vals[i] = complex(float64(i%17)-8, float64(i%5)-2) } a := complexFromArrayMust(vals, []int{len(vals)}) b.ReportAllocs() for b.Loop() { if _, err := FFT(a); err != nil { b.Fatal(err) } } } func BenchmarkFFT2_128x128(b *testing.B) { vals := make([]complex128, 128*128) for i := range vals { vals[i] = complex(float64(i%17)-8, float64(i%5)-2) } a := complexFromArrayMust(vals, []int{128, 128}) b.ReportAllocs() for b.Loop() { if _, err := FFT2(a); err != nil { b.Fatal(err) } } } // Larger sizes: the 1-D entry where per-stage parallelism and the // cached bit-reversal table pay off, and the 2-D entry where the // unit-stride row pass dominates. func BenchmarkFFT65536(b *testing.B) { vals := make([]complex128, 65536) for i := range vals { vals[i] = complex(float64(i%17)-8, float64(i%5)-2) } a := complexFromArrayMust(vals, []int{len(vals)}) b.ReportAllocs() for b.Loop() { if _, err := FFT(a); err != nil { b.Fatal(err) } } } func BenchmarkFFT2_512x512(b *testing.B) { vals := make([]complex128, 512*512) for i := range vals { vals[i] = complex(float64(i%17)-8, float64(i%5)-2) } a := complexFromArrayMust(vals, []int{512, 512}) b.ReportAllocs() for b.Loop() { if _, err := FFT2(a); err != nil { b.Fatal(err) } } } // Serial twins: one worker, so the parallel contribution of each size // reads directly off the pair. func BenchmarkFFT65536Serial(b *testing.B) { core.SetNumCPU(1) defer core.SetNumCPU(0) benchFFTLine(b, 65536) } func BenchmarkFFT2_512x512Serial(b *testing.B) { core.SetNumCPU(1) defer core.SetNumCPU(0) benchFFT2Square(b, 512) } func benchFFTLine(b *testing.B, n int) { vals := make([]complex128, n) for i := range vals { vals[i] = complex(float64(i%17)-8, float64(i%5)-2) } a := complexFromArrayMust(vals, []int{n}) b.ReportAllocs() for b.Loop() { if _, err := FFT(a); err != nil { b.Fatal(err) } } } func benchFFT2Square(b *testing.B, side int) { vals := make([]complex128, side*side) for i := range vals { vals[i] = complex(float64(i%17)-8, float64(i%5)-2) } a := complexFromArrayMust(vals, []int{side, side}) b.ReportAllocs() for b.Loop() { if _, err := FFT2(a); err != nil { b.Fatal(err) } } }