139 lines
3.2 KiB
Go
139 lines
3.2 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
package signal
|
|
|
|
import (
|
|
"testing"
|
|
|
|
"sourcedock.dev/petrbalvin/tensor/internal/core"
|
|
)
|
|
|
|
func BenchmarkConv2DParallel(b *testing.B) { benchConv2DN(b) }
|
|
|
|
func BenchmarkConv2DSerial(b *testing.B) {
|
|
core.SetNumCPU(1)
|
|
defer core.SetNumCPU(0)
|
|
benchConv2DN(b)
|
|
}
|
|
|
|
func benchConv2DN(b *testing.B) {
|
|
in, _ := core.FromFloats(make([]float64, 8*64*28*28), 8, 64, 28, 28)
|
|
k, _ := core.FromFloats(make([]float64, 64*64*3*3), 64, 64, 3, 3)
|
|
bi, _ := core.FromFloats(make([]float64, 64), 64)
|
|
b.ResetTimer()
|
|
for b.Loop() {
|
|
if _, err := Conv2D(in, k, bi, 1, 1); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// FFT benchmarks: the radix-2 kernel at a power-of-two length, and the
|
|
// 2-D separable path whose per-line scratch and payload accessors
|
|
// dominate at practical sizes. Run with -bench before and after any
|
|
// change to fft.go.
|
|
|
|
func BenchmarkFFT4096(b *testing.B) {
|
|
vals := make([]complex128, 4096)
|
|
for i := range vals {
|
|
vals[i] = complex(float64(i%17)-8, float64(i%5)-2)
|
|
}
|
|
a := complexFromArrayMust(vals, []int{len(vals)})
|
|
b.ReportAllocs()
|
|
for b.Loop() {
|
|
if _, err := FFT(a); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|
|
|
|
func BenchmarkFFT2_128x128(b *testing.B) {
|
|
vals := make([]complex128, 128*128)
|
|
for i := range vals {
|
|
vals[i] = complex(float64(i%17)-8, float64(i%5)-2)
|
|
}
|
|
a := complexFromArrayMust(vals, []int{128, 128})
|
|
b.ReportAllocs()
|
|
for b.Loop() {
|
|
if _, err := FFT2(a); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// Larger sizes: the 1-D entry where per-stage parallelism and the
|
|
// cached bit-reversal table pay off, and the 2-D entry where the
|
|
// unit-stride row pass dominates.
|
|
|
|
func BenchmarkFFT65536(b *testing.B) {
|
|
vals := make([]complex128, 65536)
|
|
for i := range vals {
|
|
vals[i] = complex(float64(i%17)-8, float64(i%5)-2)
|
|
}
|
|
a := complexFromArrayMust(vals, []int{len(vals)})
|
|
b.ReportAllocs()
|
|
for b.Loop() {
|
|
if _, err := FFT(a); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|
|
|
|
func BenchmarkFFT2_512x512(b *testing.B) {
|
|
vals := make([]complex128, 512*512)
|
|
for i := range vals {
|
|
vals[i] = complex(float64(i%17)-8, float64(i%5)-2)
|
|
}
|
|
a := complexFromArrayMust(vals, []int{512, 512})
|
|
b.ReportAllocs()
|
|
for b.Loop() {
|
|
if _, err := FFT2(a); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// Serial twins: one worker, so the parallel contribution of each size
|
|
// reads directly off the pair.
|
|
|
|
func BenchmarkFFT65536Serial(b *testing.B) {
|
|
core.SetNumCPU(1)
|
|
defer core.SetNumCPU(0)
|
|
benchFFTLine(b, 65536)
|
|
}
|
|
|
|
func BenchmarkFFT2_512x512Serial(b *testing.B) {
|
|
core.SetNumCPU(1)
|
|
defer core.SetNumCPU(0)
|
|
benchFFT2Square(b, 512)
|
|
}
|
|
|
|
func benchFFTLine(b *testing.B, n int) {
|
|
vals := make([]complex128, n)
|
|
for i := range vals {
|
|
vals[i] = complex(float64(i%17)-8, float64(i%5)-2)
|
|
}
|
|
a := complexFromArrayMust(vals, []int{n})
|
|
b.ReportAllocs()
|
|
for b.Loop() {
|
|
if _, err := FFT(a); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|
|
|
|
func benchFFT2Square(b *testing.B, side int) {
|
|
vals := make([]complex128, side*side)
|
|
for i := range vals {
|
|
vals[i] = complex(float64(i%17)-8, float64(i%5)-2)
|
|
}
|
|
a := complexFromArrayMust(vals, []int{side, side})
|
|
b.ReportAllocs()
|
|
for b.Loop() {
|
|
if _, err := FFT2(a); err != nil {
|
|
b.Fatal(err)
|
|
}
|
|
}
|
|
}
|