Files
gasm-sdk/verify/fuzz.go
T
2026-09-19 23:49:19 +02:00

592 lines
17 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package verify
import (
"encoding/hex"
"fmt"
"math/rand/v2"
"regexp"
"runtime"
"strconv"
"strings"
"unsafe"
)
// newRNG builds the deterministic generator for a fuzz seed. PCG seeds
// with two 64-bit words; deriving the second from the first keeps one seed
// one stream and rules out the all-zero seed. The sequences differ from
// the retired math/rand ones for the same seed, but remain reproducible
// run to run, which is the property the fuzzers rely on.
func newRNG(seed int64) *rand.Rand {
lo := uint64(seed)
return rand.New(rand.NewPCG(lo, ^lo))
}
// fillRandom fills buf from rng, eight bytes per draw. math/rand/v2
// dropped Read from *rand.Rand, and this loop keeps the byte sequence a
// pure function of the generator state.
func fillRandom(rng *rand.Rand, buf []byte) {
for off := 0; off < len(buf); off += 8 {
v := rng.Uint64()
for j := 0; j < 8 && off+j < len(buf); j++ {
buf[off+j] = byte(v >> (8 * uint(j)))
}
}
}
// FuzzResult reports the outcome of a differential fuzz campaign for one
// function.
type FuzzResult struct {
Func string
Iterations int
Matches int
Mismatches int
FirstFail string // description of the first mismatch ("" if none)
CrashInput []byte // input that caused the last crash/mismatch (nil if none)
}
// OK returns true when all iterations matched.
func (r FuzzResult) OK() bool { return r.Mismatches == 0 }
// String returns a human-readable summary.
func (r FuzzResult) String() string {
if r.OK() {
return fmt.Sprintf("%s: %d/%d iterations match", r.Func, r.Matches, r.Iterations)
}
s := fmt.Sprintf("%s: %d/%d match, %d MISMATCH: %s",
r.Func, r.Matches, r.Iterations, r.Mismatches, r.FirstFail)
if len(r.CrashInput) > 0 {
s += fmt.Sprintf("\n input: %x", r.CrashInput)
}
return s
}
// CorpusArg is one replayable argument of a corpus entry.
type CorpusArg struct {
Kind string `json:"kind"` // "slice", "string", "ptr", "int", "scalar"
Len int `json:"len,omitempty"` // slice: declared length in elements; string: length in bytes
Data string `json:"data,omitempty"` // slice/string/ptr: hex-encoded buffer content
Value string `json:"value,omitempty"` // int/scalar: decimal value
}
// CorpusEntry is a replayable fuzz input: the logical arguments of one
// generated call, stored as JSON. A raw argument block replays nowhere
// (its pointers point into mappings that died with the process), so the
// corpus records buffer contents and scalars instead and ReplayEntry
// rebuilds a live block from them.
type CorpusEntry struct {
Func string `json:"func"`
Args []CorpusArg `json:"args"`
}
// funcSig is a parsed // func signature from the assembly source.
type funcSig struct {
name string
params []param
results []param
}
type param struct {
name string
typ string // "[]byte", "[]int32", "int", "*[32]uint16", etc.
}
// funcSigRe matches the conventional "// func name(...)" comment.
var funcSigRe = regexp.MustCompile(`^//\s*func\s+(\w+)\(([^)]*)\)\s*(.*)$`)
// parseFuncSig extracts the function signature from a "// func ..." comment.
func parseFuncSig(comment string) (funcSig, bool) {
m := funcSigRe.FindStringSubmatch(strings.TrimSpace(comment))
if m == nil {
return funcSig{}, false
}
sig := funcSig{name: m[1],
params: parseParams(m[2])}
// Results may be "(a int, b int)" or "int" or "(int, error)".
res := strings.TrimSpace(m[3])
res = strings.TrimPrefix(res, "(")
res = strings.TrimSuffix(res, ")")
if res != "" {
sig.results = parseParams(res)
}
return sig, true
}
// parseParams splits "a []byte, b []int32" into typed parameters, handling
// shared types ("a, b []int32").
func parseParams(s string) []param {
s = strings.TrimSpace(s)
if s == "" {
return nil
}
fields := strings.Split(s, ",")
// First pass: extract the type from each field (if present).
types := make([]string, len(fields))
for i, field := range fields {
parts := strings.Fields(strings.TrimSpace(field))
if len(parts) >= 2 {
types[i] = parts[len(parts)-1]
}
}
// Propagate types backward: a field without a type inherits from the next
// field that has one (e.g. "dst" inherits "[]byte" from "src []byte").
for i := range fields {
if types[i] == "" {
for j := i + 1; j < len(fields); j++ {
if types[j] != "" {
types[i] = types[j]
break
}
}
}
}
var out []param
for i, field := range fields {
field = strings.TrimSpace(field)
if field == "" {
continue
}
parts := strings.Fields(field)
typ := types[i]
if typ == "" {
out = append(out, param{typ: parts[0]})
} else {
out = append(out, param{name: parts[0], typ: typ})
}
}
return out
}
// ExtractSignatures scans assembly source for "// func name(...)" comments
// that immediately precede a TEXT directive, and returns the parsed
// signatures keyed by the function's short name.
func ExtractSignatures(src string) map[string]funcSig {
lines := strings.Split(src, "\n")
sigs := make(map[string]funcSig)
var comments []string
for _, line := range lines {
trimmed := strings.TrimSpace(line)
if strings.HasPrefix(trimmed, "//") {
comments = append(comments, trimmed)
continue
}
if strings.HasPrefix(trimmed, "TEXT") {
// Search the comment block for the // func line.
for _, c := range comments {
if sig, ok := parseFuncSig(c); ok {
sigs[sig.name] = sig
break
}
}
comments = nil
continue
}
if trimmed != "" {
comments = nil
}
}
return sigs
}
// FuzzFunc runs a differential fuzz campaign: it JIT-executes both the
// gasm-assembled and the go-tool-asm-assembled versions of the named
// function with random inputs derived from the // func signature, and
// compares the output argument area bit-for-bit.
//
// The signature comment must appear immediately above the TEXT directive
// in the source (the conventional Go assembly layout).
//
// Not safe for concurrent use: only one JIT call may be in flight at a
// time, the trampolines keep the saved registers in package globals.
func (k *Kernel) FuzzFunc(name string, sig funcSig, goCode []byte, iterations int, seed int64) FuzzResult {
return k.FuzzFuncHook(name, sig, goCode, iterations, seed, nil)
}
// FuzzFuncHook is FuzzFunc with a hook invoked for every failing input (a
// crash or a mismatch), receiving a replayable corpus entry. A nil hook
// behaves exactly like FuzzFunc.
//
// Not safe for concurrent use: only one JIT call may be in flight at a
// time, the trampolines keep the saved registers in package globals.
func (k *Kernel) FuzzFuncHook(name string, sig funcSig, goCode []byte, iterations int, seed int64, onSave func(CorpusEntry)) FuzzResult {
result := FuzzResult{Func: name, Iterations: iterations}
rng := newRNG(seed)
// Map the Go-assembled code into a second executable region.
goExec, err := Map(goCode)
if err != nil {
result.Mismatches = iterations
result.FirstFail = fmt.Sprintf("map go code: %v", err)
return result
}
defer goExec.Unmap()
fl, err := k.Func(name)
if err != nil {
result.Mismatches = iterations
result.FirstFail = err.Error()
return result
}
// The result comparison below slices the parameter area off the
// argument block; a // func comment declaring more parameter bytes
// than the TEXT frame carries would slice past its end and panic.
// Fail the whole campaign with a clear message instead.
if ps := paramsSize(sig); ps > fl.Args {
result.Mismatches = iterations
result.FirstFail = fmt.Sprintf(
"signature declares %d parameter bytes, but the TEXT frame of %s carries %d argument bytes",
ps, name, fl.Args)
return result
}
for i := range iterations {
// Generate inputs and build TWO independent arg blocks (one per
// version) so that functions which write to their arguments
// (e.g. histogram increments) don't corrupt the other's input.
gasmArgs, goArgs, bufs, entry := genDualArgs(rng, sig, fl.Args)
entry.Func = name
// Save the current input for crash diagnostics.
result.CrashInput = gasmArgs
// Call the gasm version.
gasmOut, err := k.CallFunc(name, gasmArgs)
if err != nil {
result.Mismatches++
if result.FirstFail == "" {
result.FirstFail = fmt.Sprintf("iter %d: gasm call: %v", i, err)
}
if onSave != nil {
onSave(entry)
}
runtime.KeepAlive(bufs)
continue
}
// Call the Go version (same function, independent buffers).
goOut, err := Call(goExec.FuncAddr(0), goArgs)
if err != nil {
result.Mismatches++
if result.FirstFail == "" {
result.FirstFail = fmt.Sprintf("iter %d: go call: %v", i, err)
}
if onSave != nil {
onSave(entry)
}
runtime.KeepAlive(bufs)
continue
}
// Compare only the result area (after all input parameters).
// Pointers in the arg block differ (separate buffers), so we
// compare from resultOff to the end.
resultOff := paramsSize(sig)
gasmRes := gasmOut[resultOff:]
goRes := goOut[resultOff:]
if !equalBytes(gasmRes, goRes) {
result.Mismatches++
if result.FirstFail == "" {
result.FirstFail = fmt.Sprintf("iter %d: output mismatch at result offset %d", i, resultOff)
}
if onSave != nil {
onSave(entry)
}
} else {
result.Matches++
}
runtime.KeepAlive(bufs)
}
return result
}
// genDualArgs generates two independent ABI0 argument blocks (for gasm and
// go) with identical logical content but separate backing buffers, so that
// functions which write to their arguments don't corrupt the other's input.
func genDualArgs(rng *rand.Rand, sig funcSig, argSize int) (gasmArgs, goArgs []byte, bufs [][]byte, entry CorpusEntry) {
gasmArgs = make([]byte, argSize)
goArgs = make([]byte, argSize)
off := 0
sliceIdx := 0
for _, p := range sig.params {
switch {
case strings.HasPrefix(p.typ, "[]"):
elemSize := elemSizeFor(p.typ)
n := 1 + rng.IntN(127)
var declaredLen int
if sliceIdx == 0 {
declaredLen = n
} else {
declaredLen = n + 512
}
// Allocate a buffer comfortably larger than declaredLen*elemSize so
// that SIMD over-reads and functions that write slightly past len
// (e.g. decoders that trust len(src)) never touch unmapped memory.
bufBytes := declaredLen*elemSize + 8192
// Two independent buffers with identical random content.
buf1 := make([]byte, bufBytes)
buf2 := make([]byte, bufBytes)
fillRandom(rng, buf1[:n*elemSize])
copy(buf2, buf1)
bufs = append(bufs, buf1, buf2)
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
// len and cap both equal declaredLen, the buffer is guaranteed
// to hold at least declaredLen elements plus safety margin.
putU64(gasmArgs, off+8, uint64(declaredLen))
putU64(gasmArgs, off+16, uint64(declaredLen))
putU64(goArgs, off+8, uint64(declaredLen))
putU64(goArgs, off+16, uint64(declaredLen))
off += 24
sliceIdx++
entry.Args = append(entry.Args, CorpusArg{
Kind: "slice",
Len: declaredLen,
Data: hex.EncodeToString(buf1[:n*elemSize]),
})
case p.typ == "string":
// An ABI0 string is a two-word header (data pointer +
// length); a random pointer would fault kernels that read
// the string, so the header points at a real buffer with
// the same safety margin slices get.
n := 1 + rng.IntN(127)
buf1 := make([]byte, n+8192)
buf2 := make([]byte, n+8192)
fillRandom(rng, buf1[:n])
copy(buf2, buf1)
bufs = append(bufs, buf1, buf2)
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
putU64(gasmArgs, off+8, uint64(n))
putU64(goArgs, off+8, uint64(n))
off += 16
entry.Args = append(entry.Args, CorpusArg{
Kind: "string",
Len: n,
Data: hex.EncodeToString(buf1[:n]),
})
case strings.HasPrefix(p.typ, "*["):
nElem := arrayLen(p.typ)
elem := elemSizeFor("[]" + p.typ[strings.Index(p.typ, "]")+1:])
size := max(nElem*elem, 8)
buf1 := make([]byte, size)
buf2 := make([]byte, size)
fillRandom(rng, buf1)
copy(buf2, buf1)
bufs = append(bufs, buf1, buf2)
putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0]))
putPtr(goArgs, off, unsafe.Pointer(&buf2[0]))
off += 8
entry.Args = append(entry.Args, CorpusArg{
Kind: "ptr",
Data: hex.EncodeToString(buf1),
})
case p.typ == "int" || p.typ == "uint" || p.typ == "int64" || p.typ == "uint64":
v := uint64(rng.IntN(256))
putU64(gasmArgs, off, v)
putU64(goArgs, off, v)
off += 8
entry.Args = append(entry.Args, CorpusArg{Kind: "int", Value: strconv.FormatUint(v, 10)})
case p.typ == "complex64", p.typ == "complex128":
// complex64 is two float32s (8 bytes), complex128 two
// float64s (16): plain data words to the marshaller, one
// corpus scalar per word so replay rebuilds them exactly.
for range paramSize(p.typ) / 8 {
v := rng.Uint64()
putU64(gasmArgs, off, v)
putU64(goArgs, off, v)
off += 8
entry.Args = append(entry.Args, CorpusArg{Kind: "scalar", Value: strconv.FormatUint(v, 10)})
}
default:
v := rng.Uint64()
putU64(gasmArgs, off, v)
putU64(goArgs, off, v)
off += 8
entry.Args = append(entry.Args, CorpusArg{Kind: "scalar", Value: strconv.FormatUint(v, 10)})
}
}
return gasmArgs, goArgs, bufs, entry
}
// ReplayEntry rebuilds the argument block of a corpus entry and invokes the
// named function once, returning the argument block after the call. Slice
// and string buffers get the same safety padding the fuzzer uses, so
// over-reads that were harmless during the original run stay harmless on
// replay.
//
// Not safe for concurrent use: only one JIT call may be in flight at a
// time, the trampolines keep the saved registers in package globals.
func (k *Kernel) ReplayEntry(name string, e CorpusEntry) ([]byte, error) {
fl, err := k.Func(name)
if err != nil {
return nil, err
}
args := make([]byte, fl.Args)
var bufs [][]byte
off := 0
for _, a := range e.Args {
switch a.Kind {
case "slice":
data, err := hex.DecodeString(a.Data)
if err != nil {
return nil, fmt.Errorf("corpus: slice data: %w", err)
}
buf := make([]byte, len(data)+8192)
copy(buf, data)
bufs = append(bufs, buf)
if off+24 > len(args) {
return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name)
}
putPtr(args, off, unsafe.Pointer(&buf[0]))
putU64(args, off+8, uint64(a.Len))
putU64(args, off+16, uint64(a.Len))
off += 24
case "string":
data, err := hex.DecodeString(a.Data)
if err != nil {
return nil, fmt.Errorf("corpus: string data: %w", err)
}
buf := make([]byte, len(data)+8192)
copy(buf, data)
bufs = append(bufs, buf)
if off+16 > len(args) {
return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name)
}
putPtr(args, off, unsafe.Pointer(&buf[0]))
putU64(args, off+8, uint64(a.Len))
off += 16
case "ptr":
data, err := hex.DecodeString(a.Data)
if err != nil {
return nil, fmt.Errorf("corpus: ptr data: %w", err)
}
buf := make([]byte, max(len(data), 8))
copy(buf, data)
bufs = append(bufs, buf)
if off+8 > len(args) {
return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name)
}
putPtr(args, off, unsafe.Pointer(&buf[0]))
off += 8
default: // "int", "scalar"
v, err := strconv.ParseUint(a.Value, 10, 64)
if err != nil {
return nil, fmt.Errorf("corpus: %s value: %w", a.Kind, err)
}
if off+8 > len(args) {
return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name)
}
putU64(args, off, v)
off += 8
}
}
out, err := k.CallFunc(name, args)
runtime.KeepAlive(bufs)
return out, err
}
func elemSizeFor(sliceType string) int {
switch strings.TrimPrefix(sliceType, "[]") {
case "byte", "uint8", "int8":
return 1
case "uint16", "int16":
return 2
case "uint32", "int32", "float32":
return 4
case "uint64", "int64", "float64":
return 8
default:
return 8
}
}
// paramsSize returns the ABI0 stack size occupied by the input parameters.
func paramsSize(sig funcSig) int {
size := 0
for _, p := range sig.params {
switch {
case strings.HasPrefix(p.typ, "[]"):
size += 24 // slice header
case strings.HasPrefix(p.typ, "*["):
size += 8 // pointer
case p.typ == "bool":
size += 1
case p.typ == "string":
size += 16 // data pointer + length
case p.typ == "complex64":
size += 8 // two float32s
case p.typ == "complex128":
size += 16 // two float64s
default:
size += 8 // int, uint, etc.
}
}
return size
}
func arrayLen(typ string) int {
// "*[32]uint16" → 32
start := strings.Index(typ, "[")
end := strings.Index(typ, "]")
if start < 0 || end < 0 || end <= start {
return 1
}
n, _ := strconv.Atoi(typ[start+1 : end])
if n <= 0 {
n = 1
}
return n
}
func putPtr(buf []byte, off int, p unsafe.Pointer) {
if off+8 <= len(buf) {
u64 := uint64(uintptr(p))
buf[off] = byte(u64)
buf[off+1] = byte(u64 >> 8)
buf[off+2] = byte(u64 >> 16)
buf[off+3] = byte(u64 >> 24)
buf[off+4] = byte(u64 >> 32)
buf[off+5] = byte(u64 >> 40)
buf[off+6] = byte(u64 >> 48)
buf[off+7] = byte(u64 >> 56)
}
}
func putU64(buf []byte, off int, v uint64) {
if off+8 <= len(buf) {
buf[off] = byte(v)
buf[off+1] = byte(v >> 8)
buf[off+2] = byte(v >> 16)
buf[off+3] = byte(v >> 24)
buf[off+4] = byte(v >> 32)
buf[off+5] = byte(v >> 40)
buf[off+6] = byte(v >> 48)
buf[off+7] = byte(v >> 56)
}
}
func equalBytes(a, b []byte) bool {
if len(a) != len(b) {
return false
}
for i := range a {
if a[i] != b[i] {
return false
}
}
return true
}