296 lines
13 KiB
Makefile
296 lines
13 KiB
Makefile
# tensor: a scientific computing library in pure Go.
|
|||
|
|
#
|
||
|
|
# A library: nothing to install, nothing to run. Below the variable
|
||
|
|
# block is the standard recipe contract; docs-check and fuzz-all are
|
||
|
|
# the project extensions. The portable build is the only build.
|
||
|
|
|
||
|
|
# What the test, race, unit and bench recipes sweep: every logic
|
||
|
|
# package, the examples aside. The examples are main programs with no
|
||
|
|
# tests; the build compiles them, and the coverage floor is a property
|
||
|
|
# of the library packages alone.
|
||
|
|
packages := ". ./internal/... ./grad/... ./integrate/... ./io/... ./linalg/... ./optim/... ./plot/... ./signal/... ./spmd/... ./stats/..."
|
||
|
|
|
||
|
|
# Where the fuzz targets live; the fuzz-all sweep discovers them by name.
|
||
|
|
fuzz_packages := "./io ./grad"
|
||
|
|
|
||
|
|
# The memory fence for the test recipes: a cgroup ceiling with swap off, so a
|
||
|
|
# runaway run dies as a failed run and never eats the machine. 4G is the
|
||
|
|
# default; raise it only with a reason recorded here.
|
||
|
|
memlimit := "4G"
|
||
|
|
|
||
|
|
# The representative set the bench-report measures, one benchmark per
|
||
|
|
# kernel family across the packages. The recipe measures the names this
|
||
|
|
# release tag and the head tree share, so an old tag never blocks the
|
||
|
|
# report.
|
||
|
|
bench_set := "Add1M Exp100k Dot SumAxis2D SumAxis3DDim1 MeanAxis2D Norm2D Prod2D CumSum2D ArgSort1M MatMul1024 MatMulSmall64 MatMulOdd130 MatMulTall512x64x2048 MatVec KernelTransposeTiled256 EinsumBatchedMatMul Cholesky256 Solve256 SVD128 FFT4096 FFT2_512x512 Conv1D CWTMorlet SavitzkyGolay WelchPSD IntegrateRK4 IntegrateBackwardEuler IntegrateHeat2D MinimiseLBFGSBounded MinimiseDifferentialEvolution BackwardTwoLayer LinearRegression LogisticRegression KernelDensity Median CovarianceMatrix"
|
||
|
|
|
||
|
|
default:
|
||
|
|
@just --list
|
||
|
|
|
||
|
|
# Compile. Zero errors, zero warnings.
|
||
|
|
build:
|
||
|
|
go build ./...
|
||
|
|
|
||
|
|
# The test gate: the suite, no cache, the coverage floor, under the memory fence.
|
||
|
|
test:
|
||
|
|
#!/usr/bin/env perl
|
||
|
|
my @fence = (q{systemd-run}, q{--user}, q{--scope},
|
||
|
|
q{-p}, q{MemoryMax={{memlimit}}}, q{-p}, q{MemorySwapMax=0});
|
||
|
|
system(@fence, q{go}, q{test}, q{-count=1}, q{-timeout}, q{30m},
|
||
|
|
q{-coverprofile}, q{coverage.out}, qw({{packages}})) == 0
|
||
|
|
or die qq{the test suite failed\n};
|
||
|
|
open(my $c, q{-|}, q{go}, q{tool}, q{cover}, q{-func=coverage.out}) or die qq{cover: $!};
|
||
|
|
my $total;
|
||
|
|
while (my $l = <$c>) { $total = $1 if $l =~ m{^total:\s+\S+\s+([0-9.]+)%} }
|
||
|
|
close($c);
|
||
|
|
die qq{no total line in coverage.out\n} unless defined $total;
|
||
|
|
printf qq{Total coverage: %s%%\n}, $total;
|
||
|
|
exit($total < 80 ? 1 : 0);
|
||
|
|
|
||
|
|
# The same suite under the race detector. The expensive one, still fenced.
|
||
|
|
race:
|
||
|
|
systemd-run --user --scope -p MemoryMax={{memlimit}} -p MemorySwapMax=0 go test -race -count=1 -timeout 30m {{packages}}
|
||
|
|
|
||
|
|
# Fast scoped run for iterating. This is the one that runs after every edit.
|
||
|
|
unit pkgs=packages run=".*":
|
||
|
|
systemd-run --user --scope -p MemoryMax={{memlimit}} -p MemorySwapMax=0 go test {{pkgs}} -run '{{run}}'
|
||
|
|
|
||
|
|
# Time-boxed fuzz of one target in one package. The package is required; never a gate.
|
||
|
|
fuzz target pkg fuzztime="60s":
|
||
|
|
systemd-run --user --scope -p MemoryMax={{memlimit}} -p MemorySwapMax=0 go test -run '^$' -fuzz '{{target}}' -fuzztime={{fuzztime}} {{pkg}}
|
||
|
|
|
||
|
|
# Benchmarks. On an idle machine only, deliberately unfenced: a ceiling would distort the measurement.
|
||
|
|
bench pkgs=packages:
|
||
|
|
go test -run '^$' -bench=. -benchmem -count=5 {{pkgs}}
|
||
|
|
|
||
|
|
# Format in place.
|
||
|
|
fmt:
|
||
|
|
gofmt -w .
|
||
|
|
|
||
|
|
# Zero diff. Prints nothing when everything is formatted.
|
||
|
|
fmt-check:
|
||
|
|
#!/usr/bin/env perl
|
||
|
|
open(my $g, q{-|}, q{gofmt}, q{-l}, q{.}) or die qq{gofmt: $!};
|
||
|
|
my @bad = <$g>;
|
||
|
|
close($g);
|
||
|
|
print @bad;
|
||
|
|
exit(@bad ? 1 : 0);
|
||
|
|
|
||
|
|
# Both static gates: go vet and go fix -diff.
|
||
|
|
vet:
|
||
|
|
go vet ./...
|
||
|
|
go fix -diff ./...
|
||
|
|
|
||
|
|
# The definition of done, in one command. Once per task, never per edit.
|
||
|
|
gates: build fmt-check vet test race
|
||
|
|
|
||
|
|
# Remove the build artefacts.
|
||
|
|
clean:
|
||
|
|
rm -rf bin/ coverage.out
|
||
|
|
|
||
|
|
# Run every Go program in README.md. An extension: no standard gate covers prose.
|
||
|
|
docs-check:
|
||
|
|
#!/usr/bin/env perl
|
||
|
|
my $root = `pwd`; chomp $root;
|
||
|
|
my $tmp = $ENV{TMPDIR} || q{/tmp};
|
||
|
|
my $base = qq{$tmp/tensor-docs-check};
|
||
|
|
system(q{rm}, q{-rf}, $base) == 0 or die qq{rm $base failed\n};
|
||
|
|
mkdir $base or die qq{mkdir $base: $!};
|
||
|
|
open(my $md, q{<}, q{README.md}) or die qq{README.md: $!};
|
||
|
|
my (@blocks, $cur);
|
||
|
|
while (my $l = <$md>) {
|
||
|
|
if ($l =~ m{^```go\s*$}) { $cur = q{}; next }
|
||
|
|
if (defined $cur && $l =~ m{^```\s*$}) { push @blocks, $cur; undef $cur; next }
|
||
|
|
$cur .= $l if defined $cur;
|
||
|
|
}
|
||
|
|
close($md) or die qq{README.md close failed\n};
|
||
|
|
my ($i, $failed) = (0, 0);
|
||
|
|
for my $b (@blocks) {
|
||
|
|
$i++;
|
||
|
|
my $dir = sprintf qq{%s/%02d}, $base, $i;
|
||
|
|
mkdir $dir or die qq{mkdir $dir: $!};
|
||
|
|
open(my $g, q{>}, qq{$dir/go.mod}) or die qq{$dir/go.mod: $!};
|
||
|
|
print $g qq{module readme$i\n\ngo 1.27.1\n\n}
|
||
|
|
. qq{require sourcedock.dev/petrbalvin/tensor v0.0.0\n\n}
|
||
|
|
. qq{replace sourcedock.dev/petrbalvin/tensor => $root\n};
|
||
|
|
close($g) or die qq{go.mod close failed\n};
|
||
|
|
open(my $m, q{>}, qq{$dir/main.go}) or die qq{$dir/main.go: $!};
|
||
|
|
print $m $b;
|
||
|
|
close($m) or die qq{main.go close failed\n};
|
||
|
|
chdir $dir or die qq{chdir $dir: $!};
|
||
|
|
local $ENV{GOFLAGS} = q{-mod=mod};
|
||
|
|
open(my $r, q{-|}, q{go}, q{run}, q{.}) or die qq{go run $dir: $!};
|
||
|
|
my @out = <$r>;
|
||
|
|
# close reports true on a clean exit; $? carries the status of one that failed.
|
||
|
|
my $ok = close($r);
|
||
|
|
my $status = $? >> 8;
|
||
|
|
unless ($ok) {
|
||
|
|
$failed++;
|
||
|
|
print qq{FAILED: README program $i in $dir (exit $status)\n}, @out;
|
||
|
|
}
|
||
|
|
chdir $root or die qq{chdir $root: $!};
|
||
|
|
}
|
||
|
|
print qq{README programs: $i, failures: $failed\n};
|
||
|
|
exit($failed ? 1 : 0);
|
||
|
|
|
||
|
|
# Fuzz every target for FUZZTIME each, under the same fence. Exploration, never a gate.
|
||
|
|
fuzz-all fuzztime="5s":
|
||
|
|
#!/usr/bin/env perl
|
||
|
|
my @fence = (q{systemd-run}, q{--user}, q{--scope},
|
||
|
|
q{-p}, q{MemoryMax={{memlimit}}}, q{-p}, q{MemorySwapMax=0});
|
||
|
|
my $bad = 0;
|
||
|
|
for my $pkg (split q{ }, q({{fuzz_packages}})) {
|
||
|
|
open(my $l, q{-|}, q{go}, q{test}, q{-list}, q{^Fuzz}, $pkg)
|
||
|
|
or die qq{go test -list $pkg: $!};
|
||
|
|
my @targets = map { chomp; $_ } grep { m{^Fuzz} } <$l>;
|
||
|
|
close($l) or die qq{go test -list $pkg failed\n};
|
||
|
|
for my $t (@targets) {
|
||
|
|
print qq{=== fuzz $t in $pkg for {{fuzztime}} ===\n};
|
||
|
|
system(@fence, q{go}, q{test}, q{-run}, q{^$}, q{-fuzz}, qq{^$t\$},
|
||
|
|
q{-fuzztime}, q({{fuzztime}}), $pkg) == 0
|
||
|
|
or do { $bad = 1; print qq{FAILED: $t\n} };
|
||
|
|
}
|
||
|
|
}
|
||
|
|
exit($bad ? 1 : 0);
|
||
|
|
|
||
|
|
# Benchmark the working tree against the latest release tag, rewriting docs/benchmarks/release-vs-head.md.
|
||
|
|
bench-report benchtime="0.5s":
|
||
|
|
#!/usr/bin/env perl
|
||
|
|
use v5.40;
|
||
|
|
my $repo = `git rev-parse --show-toplevel`; chomp $repo;
|
||
|
|
my @tags = grep { chomp; $_ } `git tag --list v* --sort=-v:refname`;
|
||
|
|
@tags or die qq{no release tag in the repository\n};
|
||
|
|
my ($tag) = @tags;
|
||
|
|
my ($head, $dirty) = (`git rev-parse --short HEAD` =~ s/\s+\z//r);
|
||
|
|
$dirty = `git status --porcelain` eq q{} ? q{} : q{, dirty working tree};
|
||
|
|
chomp $head;
|
||
|
|
my $rel_dir = $repo =~ s{/[^/]+\z}{}r . qq{/tensor-bench-release};
|
||
|
|
END { system(q{git}, q{worktree}, q{remove}, q{--force}, $rel_dir) }
|
||
|
|
if (-d $rel_dir) { system(q{git}, q{worktree}, q{remove}, q{--force}, $rel_dir) }
|
||
|
|
system(q{git}, q{worktree}, q{add}, q{--detach}, $rel_dir, $tag) == 0
|
||
|
|
or die qq{worktree for $tag failed\n};
|
||
|
|
print qq{=== $tag ($rel_dir) against head $head$dirty ===\n};
|
||
|
|
my %wanted = map { $_ => 1 } split q{ }, q({{bench_set}});
|
||
|
|
my %map; # side -> name -> package
|
||
|
|
for my $side ([release => $rel_dir], [head => $repo]) {
|
||
|
|
my @pkgs = split /\n/, `go -C $side->[1] list ./...`;
|
||
|
|
for my $p (@pkgs) {
|
||
|
|
open(my $l, q{-|}, q{go}, q{-C}, $side->[1], q{test}, $p, q{-list}, q{Benchmark})
|
||
|
|
or die qq{go -C $side->[1] test -list $p: $!};
|
||
|
|
while (my $n = <$l>) {
|
||
|
|
next unless $n =~ m{^Benchmark(\w+)\s*\z};
|
||
|
|
chomp $n;
|
||
|
|
$map{$side->[0]}{$1} = $p if $wanted{$1};
|
||
|
|
}
|
||
|
|
close($l);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
my %pkgs; # side -> package -> [names both sides know]
|
||
|
|
for my $n (sort keys %{$map{release}}) {
|
||
|
|
next unless $map{head}{$n};
|
||
|
|
push @{$pkgs{release}{$map{release}{$n}}}, $n;
|
||
|
|
push @{$pkgs{head}{$map{head}{$n}}}, $n;
|
||
|
|
}
|
||
|
|
my (%ns, %bytes, %allocs);
|
||
|
|
for my $round (1 .. 4) {
|
||
|
|
my @sides = $round % 2 ? ([release => $rel_dir], [head => $repo])
|
||
|
|
: ([head => $repo], [release => $rel_dir]);
|
||
|
|
for my $s (@sides) {
|
||
|
|
my ($side, $dir) = @$s;
|
||
|
|
for my $p (sort keys %{$pkgs{$side}}) {
|
||
|
|
my @names = @{$pkgs{$side}{$p}};
|
||
|
|
my $sel = q{^(} . join(q{|}, map { qq{Benchmark$_} } @names) . q{)$};
|
||
|
|
open(my $r, q{-|}, q{go}, q{-C}, $dir, q{test}, $p, q{-run}, q{^$},
|
||
|
|
q{-bench}, $sel, q{-benchmem}, qq{-benchtime={{benchtime}}}, q{-count=2})
|
||
|
|
or die qq{go -C $dir bench $p: $!};
|
||
|
|
while (my $l = <$r>) {
|
||
|
|
next unless $l =~ m{^Benchmark(\w+)\S*\s+\d+\s+([\d.]+) ns/op(?:\s+(\d+) B/op(?:\s+(\d+) allocs/op)?)?};
|
||
|
|
my $n = $1;
|
||
|
|
next unless $wanted{$n} && $map{release}{$n} && $map{head}{$n};
|
||
|
|
push @{$ns{$n}{$side}}, $2;
|
||
|
|
push @{$bytes{$n}{$side}}, $3 if defined $3;
|
||
|
|
push @{$allocs{$n}{$side}}, $4 if defined $4;
|
||
|
|
}
|
||
|
|
close($r) or die qq{the bench run failed: $p ($side)\n};
|
||
|
|
}
|
||
|
|
}
|
||
|
|
print qq{=== round $round done ===\n};
|
||
|
|
}
|
||
|
|
sub median { my @v = sort { $a <=> $b } @_; my $m = int(@v / 2); @v % 2 ? $v[$m] : ($v[$m - 1] + $v[$m]) / 2 }
|
||
|
|
sub unit {
|
||
|
|
my ($ns) = @_;
|
||
|
|
return sprintf qq{%.2f ms}, $ns / 1e6 if $ns >= 1e6;
|
||
|
|
return sprintf qq{%.2f µs}, $ns / 1e3 if $ns >= 1e3;
|
||
|
|
return sprintf qq{%.0f ns}, $ns;
|
||
|
|
}
|
||
|
|
my (@rows, $faster, $slower, $same);
|
||
|
|
($faster, $slower, $same) = (0, 0, 0);
|
||
|
|
for my $n (sort keys %ns) {
|
||
|
|
next unless @{$ns{$n}{release}} && @{$ns{$n}{head}};
|
||
|
|
my ($mr, $mh) = (median(@{$ns{$n}{release}}), median(@{$ns{$n}{head}}));
|
||
|
|
my $f = $mr / $mh;
|
||
|
|
$f >= 1.05 ? $faster++ : $f <= 0.95 ? $slower++ : $same++;
|
||
|
|
my $bs = defined $bytes{$n}{release} && defined $bytes{$n}{head}
|
||
|
|
? sprintf qq{%d → %d}, median(@{$bytes{$n}{release}}), median(@{$bytes{$n}{head}}) : q{—};
|
||
|
|
my $as = defined $allocs{$n}{release} && defined $allocs{$n}{head}
|
||
|
|
? sprintf qq{%d → %d}, median(@{$allocs{$n}{release}}), median(@{$allocs{$n}{head}}) : q{—};
|
||
|
|
push @rows, [$n, unit($mr), unit($mh), sprintf(qq{%.2f}, $f), $bs, $as];
|
||
|
|
}
|
||
|
|
@rows or die qq{no benchmark measured on both sides\n};
|
||
|
|
@rows = sort { $b->[3] <=> $a->[3] } @rows;
|
||
|
|
open(my $ci, q{<}, q{/proc/cpuinfo}) or die qq{/proc/cpuinfo: $!};
|
||
|
|
my ($cpu_name) = grep { m{^model name} } <$ci>;
|
||
|
|
close($ci);
|
||
|
|
($cpu_name = $cpu_name // q{unknown CPU}) =~ s{^model name\s*:\s*}{};
|
||
|
|
chomp $cpu_name;
|
||
|
|
my $cores = do { open(my $c3, q{<}, q{/proc/cpuinfo}); grep { m{^processor} } <$c3> };
|
||
|
|
my $mem = do { open(my $m, q{<}, q{/proc/meminfo}); (grep { m{^MemTotal} } <$m>)[0] };
|
||
|
|
$mem =~ s{^MemTotal:\s+([0-9]+) kB.*}{sprintf qq{%.0f GiB}, $1 / 1048576}e;
|
||
|
|
chomp(my $go_v = `go version`); chomp(my $os = `uname -sr`); chomp($mem);
|
||
|
|
my $report = qq{$repo/docs/benchmarks/release-vs-head.md};
|
||
|
|
mkdir qq{$repo/docs/benchmarks};
|
||
|
|
open(my $o, q{>}, $report) or die qq{$report: $!};
|
||
|
|
print $o <<"EOF";
|
||
|
|
# Benchmark report: $tag against head $head
|
||
|
|
|
||
|
|
One live comparison, regenerated by `just bench-report` and never
|
||
|
|
accumulated: the newest release tag against the working tree, taken in
|
||
|
|
one interleaved session on an idle machine. Re-run it before a release
|
||
|
|
and commit the file the run writes.
|
||
|
|
|
||
|
|
## Environment
|
||
|
|
|
||
|
|
- CPU: $cpu_name, $cores logical cores
|
||
|
|
- Memory: $mem
|
||
|
|
- OS: $os
|
||
|
|
- Go: $go_v
|
||
|
|
- Build: the portable build, no pinned `GOAMD64` level, no `GOEXPERIMENT`
|
||
|
|
|
||
|
|
## Revisions
|
||
|
|
|
||
|
|
- Release: `$tag`
|
||
|
|
- Head: `$head`$dirty
|
||
|
|
- The set: the benchmarks `bench_set` names that both revisions carry;
|
||
|
|
a name either side lacks is left out, never counted as a result.
|
||
|
|
|
||
|
|
## Method
|
||
|
|
|
||
|
|
Four interleaved rounds, the side order alternating between rounds, two
|
||
|
|
counts per side per round, `-benchtime={{benchtime}}` with `-benchmem`,
|
||
|
|
medians over the eight samples a side collects. A factor at or above
|
||
|
|
1.05 is faster, at or below 0.95 slower, anything between noise. Time
|
||
|
|
of the release divides time of the head, so above one means the head is
|
||
|
|
faster.
|
||
|
|
|
||
|
|
## Results
|
||
|
|
|
||
|
|
| Benchmark | $tag | head $head | factor | B/op release → head | allocs/op release → head |
|
||
|
|
|---|---|---|---|---|---|
|
||
|
|
EOF
|
||
|
|
for my $r (@rows) { print $o qq{| `$r->[0]` | $r->[1] | $r->[2] | $r->[3]x | $r->[4] | $r->[5] |\n} }
|
||
|
|
print $o qq{\n## Summary\n\n$faster of }, scalar @rows,
|
||
|
|
qq{ benchmarks sit above the noise band (faster), $slower below it (slower), $same inside it.\n};
|
||
|
|
close($o) or die qq{$report close failed\n};
|
||
|
|
print qq{=== $report written: $faster faster, $slower slower, $same unchanged ===\n};
|
||
|
|
|