Files
petrbalvin af4ee19703
Release / gates (push) Successful in 4m38s
Test / test (push) Successful in 5m16s
Release / release (push) Successful in 35s
feat: initial release
Assisted-by: GLM 5.3 Flash
2026-09-03 10:00:00 +02:00

319 lines
7.7 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: MIT
package core
import "testing"
// The matmul hot loop of inference: benchmarks keep the loop
// order honest. Compare with -bench=BenchmarkMatMul before and after any
// change to the kernels.
func BenchmarkMatMul(b *testing.B) {
a, _ := FromFloats(make([]float64, 128*128), 128, 128)
for i := range a.Len() {
a.RawFloats()[i] = float64(i%7) - 3
}
c, _ := FromFloats(make([]float64, 128*128), 128, 128)
for i := range c.Len() {
c.RawFloats()[i] = float64(i%5) - 2
}
b.ReportAllocs()
for b.Loop() {
_, err := MatMul2D(a, c)
if err != nil {
b.Fatal(err)
}
}
}
func BenchmarkAdd(b *testing.B) {
x, _ := FromFloats(make([]float64, 100_000), 100_000)
y, _ := FromFloats(make([]float64, 100_000), 100_000)
b.ReportAllocs()
for b.Loop() {
if _, err := Add(x, y); err != nil {
b.Fatal(err)
}
}
}
// Parallel-vs-serial comparisons: run with -bench=. to see the
// speedup the parallel kernels deliver on the host's core count. The
// serial variants pin the worker count to one.
func BenchmarkMatMulParallel(b *testing.B) {
benchMatMulN(b, 512)
}
func BenchmarkMatMulSerial(b *testing.B) {
SetNumCPU(1)
defer SetNumCPU(0)
benchMatMulN(b, 512)
}
func benchMatMulN(b *testing.B, n int) {
a, _ := FromFloats(make([]float64, n*n), n, n)
c, _ := FromFloats(make([]float64, n*n), n, n)
b.ResetTimer()
for b.Loop() {
if _, err := MatMul2D(a, c); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkAddParallel(b *testing.B) { benchAddN(b, 1<<22) }
func BenchmarkAddSerial(b *testing.B) {
SetNumCPU(1)
defer SetNumCPU(0)
benchAddN(b, 1<<22)
}
func benchAddN(b *testing.B, n int) {
a, _ := FromFloats(make([]float64, n), n)
c, _ := FromFloats(make([]float64, n), n)
b.ResetTimer()
for b.Loop() {
if _, err := Add(a, c); err != nil {
b.Fatal(err)
}
}
}
// BenchmarkMatMulFloat32 exercises the production inference dtype at
// sizes the embedding/table workloads actually see.
func BenchmarkMatMulFloat32_128(b *testing.B) { benchMatMulF32N(b, 128) }
func BenchmarkMatMulFloat32_512(b *testing.B) { benchMatMulF32N(b, 512) }
func benchMatMulF32N(b *testing.B, n int) {
af, _ := FromFloats(make([]float64, n*n), n, n)
bf, _ := FromFloats(make([]float64, n*n), n, n)
a, _ := Astype(af, Float32)
c, _ := Astype(bf, Float32)
b.ResetTimer()
for b.Loop() {
if _, err := MatMul2D(a, c); err != nil {
b.Fatal(err)
}
}
}
// BroadcastTo sits inside every bias add and normalisation, so its walk
// is on the training hot path. The bias-shaped case (1, C, 1, 1) to
// (N, C, H, W) is the one the run-fill rewrite targets.
func BenchmarkBroadcastToBiasNCHW(b *testing.B) {
src, _ := FromFloats(make([]float64, 64), 1, 64, 1, 1)
b.ReportAllocs()
for b.Loop() {
if _, err := BroadcastTo(src, 8, 64, 28, 28); err != nil {
b.Fatal(err)
}
}
}
// The trailing-replication case (1, C) to (N, C): the run is the whole
// channel block, the other extreme from a fully strided walk.
func BenchmarkBroadcastToBiasNC(b *testing.B) {
src, _ := FromFloats(make([]float64, 256), 1, 256)
b.ReportAllocs()
for b.Loop() {
if _, err := BroadcastTo(src, 128, 256); err != nil {
b.Fatal(err)
}
}
}
// The non-broadcast expansion (prepending leading dimensions only):
// every element is distinct, so this is the worst case for the rewrite.
func BenchmarkBroadcastToPrepend(b *testing.B) {
src, _ := FromFloats(make([]float64, 128*256), 128, 256)
b.ReportAllocs()
for b.Loop() {
if _, err := BroadcastTo(src, 4, 128, 256); err != nil {
b.Fatal(err)
}
}
}
// Reduction and vector-shaped hot paths: Dot and Min walk a single
// payload, MatVec is the 2-D×1-D product. Run with -bench before and
// after any change to reduce.go or the vector kernels in mat.go.
func benchVecF64(b *testing.B, n int) (*Array, *Array) {
a, _ := FromFloats(make([]float64, n), n)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%11) - 5
}
c, _ := FromFloats(make([]float64, n), n)
for i := range c.RawFloats() {
c.RawFloats()[i] = float64(i%7) - 3
}
return a, c
}
func BenchmarkDot(b *testing.B) {
a, c := benchVecF64(b, 1<<20)
b.ReportAllocs()
for b.Loop() {
if _, err := Dot(a, c); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkMinMax(b *testing.B) {
a, _ := benchVecF64(b, 1<<20)
b.ReportAllocs()
for b.Loop() {
if _, err := Max(a); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkMatVec(b *testing.B) {
m, _ := FromFloats(make([]float64, 512*512), 512, 512)
for i := range m.RawFloats() {
m.RawFloats()[i] = float64(i%13) - 6
}
v, _ := FromFloats(make([]float64, 512), 512)
for i := range v.RawFloats() {
v.RawFloats()[i] = float64(i%7) - 3
}
b.ReportAllocs()
for b.Loop() {
if _, err := MatMul2D(m, v); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkTranspose(b *testing.B) {
a, _ := FromFloats(make([]float64, 256*256), 256, 256)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%9) - 4
}
b.ReportAllocs()
for b.Loop() {
Transpose(a)
}
}
func BenchmarkRowCol(b *testing.B) {
a, _ := FromFloats(make([]float64, 512*512), 512, 512)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%9) - 4
}
b.ReportAllocs()
for b.Loop() {
if _, err := Row(a, 7); err != nil {
b.Fatal(err)
}
if _, err := Col(a, 7); err != nil {
b.Fatal(err)
}
}
}
// ColTall isolates the column gather from output allocation: the tall
// shape makes the per-row strided read the dominant cost.
func BenchmarkColTall(b *testing.B) {
a, _ := FromFloats(make([]float64, 65536*8), 65536, 8)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%9) - 4
}
b.ReportAllocs()
for b.Loop() {
if _, err := Col(a, 3); err != nil {
b.Fatal(err)
}
}
}
// Along-dim reductions: Norm folds with math.Pow per element, Prod is
// the multiply twin of the axis sum. Run with -bench before and after
// any change to reduction2.go.
func BenchmarkNorm2D(b *testing.B) {
a, _ := FromFloats(make([]float64, 512*512), 512, 512)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%9) - 4
}
b.ReportAllocs()
for b.Loop() {
if _, err := Norm(a, 2, 1, false); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkProd2D(b *testing.B) {
a, _ := FromFloats(make([]float64, 512*512), 512, 512)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%9) + 1
}
b.ReportAllocs()
for b.Loop() {
if _, err := Prod(a, 1, false); err != nil {
b.Fatal(err)
}
}
}
// CumSum walks the prefix scan along the trailing dimension.
func BenchmarkCumSum2D(b *testing.B) {
a, _ := FromFloats(make([]float64, 512*512), 512, 512)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%9) - 4
}
b.ReportAllocs()
for b.Loop() {
if _, err := CumSum(a, 1); err != nil {
b.Fatal(err)
}
}
}
// Element-wise math walks the payload through a real function; Exp and
// Sqrt are the training hot ones. Run with -bench before and after any
// change to mathfunc.go.
func benchmarkRealFunc(b *testing.B, f func(*Array) (*Array, error)) {
a, _ := FromFloats(make([]float64, 1<<20), 1<<20)
for i := range a.RawFloats() {
a.RawFloats()[i] = float64(i%97) + 1
}
b.ReportAllocs()
for b.Loop() {
if _, err := f(a); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkExp(b *testing.B) { benchmarkRealFunc(b, Exp) }
func BenchmarkSqrt(b *testing.B) { benchmarkRealFunc(b, Sqrt) }
func BenchmarkProd1D(b *testing.B) {
a, _ := FromFloats(prodFixture(1<<20), 1<<20)
b.ResetTimer()
for i := 0; i < b.N; i++ {
if _, err := Prod(a, 0, false); err != nil {
b.Fatal(err)
}
}
}
func BenchmarkNorm1D(b *testing.B) {
a, _ := FromFloats(prodFixture(1<<20), 1<<20)
b.ResetTimer()
for i := 0; i < b.N; i++ {
if _, err := Norm(a, 2, 0, false); err != nil {
b.Fatal(err)
}
}
}