// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package verify import ( "encoding/hex" "fmt" "math/rand/v2" "regexp" "runtime" "strconv" "strings" "unsafe" ) // newRNG builds the deterministic generator for a fuzz seed. PCG seeds // with two 64-bit words; deriving the second from the first keeps one seed // one stream and rules out the all-zero seed. The sequences differ from // the retired math/rand ones for the same seed, but remain reproducible // run to run, which is the property the fuzzers rely on. func newRNG(seed int64) *rand.Rand { lo := uint64(seed) return rand.New(rand.NewPCG(lo, ^lo)) } // fillRandom fills buf from rng, eight bytes per draw. math/rand/v2 // dropped Read from *rand.Rand, and this loop keeps the byte sequence a // pure function of the generator state. func fillRandom(rng *rand.Rand, buf []byte) { for off := 0; off < len(buf); off += 8 { v := rng.Uint64() for j := 0; j < 8 && off+j < len(buf); j++ { buf[off+j] = byte(v >> (8 * uint(j))) } } } // FuzzResult reports the outcome of a differential fuzz campaign for one // function. type FuzzResult struct { Func string Iterations int Matches int Mismatches int FirstFail string // description of the first mismatch ("" if none) CrashInput []byte // input that caused the last crash/mismatch (nil if none) } // OK returns true when all iterations matched. func (r FuzzResult) OK() bool { return r.Mismatches == 0 } // String returns a human-readable summary. func (r FuzzResult) String() string { if r.OK() { return fmt.Sprintf("%s: %d/%d iterations match", r.Func, r.Matches, r.Iterations) } s := fmt.Sprintf("%s: %d/%d match, %d MISMATCH: %s", r.Func, r.Matches, r.Iterations, r.Mismatches, r.FirstFail) if len(r.CrashInput) > 0 { s += fmt.Sprintf("\n input: %x", r.CrashInput) } return s } // CorpusArg is one replayable argument of a corpus entry. type CorpusArg struct { Kind string `json:"kind"` // "slice", "string", "ptr", "int", "scalar" Len int `json:"len,omitempty"` // slice: declared length in elements; string: length in bytes Data string `json:"data,omitempty"` // slice/string/ptr: hex-encoded buffer content Value string `json:"value,omitempty"` // int/scalar: decimal value } // CorpusEntry is a replayable fuzz input: the logical arguments of one // generated call, stored as JSON. A raw argument block replays nowhere // (its pointers point into mappings that died with the process), so the // corpus records buffer contents and scalars instead and ReplayEntry // rebuilds a live block from them. type CorpusEntry struct { Func string `json:"func"` Args []CorpusArg `json:"args"` } // funcSig is a parsed // func signature from the assembly source. type funcSig struct { name string params []param results []param } type param struct { name string typ string // "[]byte", "[]int32", "int", "*[32]uint16", etc. } // funcSigRe matches the conventional "// func name(...)" comment. var funcSigRe = regexp.MustCompile(`^//\s*func\s+(\w+)\(([^)]*)\)\s*(.*)$`) // parseFuncSig extracts the function signature from a "// func ..." comment. func parseFuncSig(comment string) (funcSig, bool) { m := funcSigRe.FindStringSubmatch(strings.TrimSpace(comment)) if m == nil { return funcSig{}, false } sig := funcSig{name: m[1], params: parseParams(m[2])} // Results may be "(a int, b int)" or "int" or "(int, error)". res := strings.TrimSpace(m[3]) res = strings.TrimPrefix(res, "(") res = strings.TrimSuffix(res, ")") if res != "" { sig.results = parseParams(res) } return sig, true } // parseParams splits "a []byte, b []int32" into typed parameters, handling // shared types ("a, b []int32"). func parseParams(s string) []param { s = strings.TrimSpace(s) if s == "" { return nil } fields := strings.Split(s, ",") // First pass: extract the type from each field (if present). types := make([]string, len(fields)) for i, field := range fields { parts := strings.Fields(strings.TrimSpace(field)) if len(parts) >= 2 { types[i] = parts[len(parts)-1] } } // Propagate types backward: a field without a type inherits from the next // field that has one (e.g. "dst" inherits "[]byte" from "src []byte"). for i := range fields { if types[i] == "" { for j := i + 1; j < len(fields); j++ { if types[j] != "" { types[i] = types[j] break } } } } var out []param for i, field := range fields { field = strings.TrimSpace(field) if field == "" { continue } parts := strings.Fields(field) typ := types[i] if typ == "" { out = append(out, param{typ: parts[0]}) } else { out = append(out, param{name: parts[0], typ: typ}) } } return out } // ExtractSignatures scans assembly source for "// func name(...)" comments // that immediately precede a TEXT directive, and returns the parsed // signatures keyed by the function's short name. func ExtractSignatures(src string) map[string]funcSig { lines := strings.Split(src, "\n") sigs := make(map[string]funcSig) var comments []string for _, line := range lines { trimmed := strings.TrimSpace(line) if strings.HasPrefix(trimmed, "//") { comments = append(comments, trimmed) continue } if strings.HasPrefix(trimmed, "TEXT") { // Search the comment block for the // func line. for _, c := range comments { if sig, ok := parseFuncSig(c); ok { sigs[sig.name] = sig break } } comments = nil continue } if trimmed != "" { comments = nil } } return sigs } // FuzzFunc runs a differential fuzz campaign: it JIT-executes both the // gasm-assembled and the go-tool-asm-assembled versions of the named // function with random inputs derived from the // func signature, and // compares the output argument area bit-for-bit. // // The signature comment must appear immediately above the TEXT directive // in the source (the conventional Go assembly layout). // // Not safe for concurrent use: only one JIT call may be in flight at a // time, the trampolines keep the saved registers in package globals. func (k *Kernel) FuzzFunc(name string, sig funcSig, goCode []byte, iterations int, seed int64) FuzzResult { return k.FuzzFuncHook(name, sig, goCode, iterations, seed, nil) } // FuzzFuncHook is FuzzFunc with a hook invoked for every failing input (a // crash or a mismatch), receiving a replayable corpus entry. A nil hook // behaves exactly like FuzzFunc. // // Not safe for concurrent use: only one JIT call may be in flight at a // time, the trampolines keep the saved registers in package globals. func (k *Kernel) FuzzFuncHook(name string, sig funcSig, goCode []byte, iterations int, seed int64, onSave func(CorpusEntry)) FuzzResult { result := FuzzResult{Func: name, Iterations: iterations} rng := newRNG(seed) // Map the Go-assembled code into a second executable region. goExec, err := Map(goCode) if err != nil { result.Mismatches = iterations result.FirstFail = fmt.Sprintf("map go code: %v", err) return result } defer goExec.Unmap() fl, err := k.Func(name) if err != nil { result.Mismatches = iterations result.FirstFail = err.Error() return result } // The result comparison below slices the parameter area off the // argument block; a // func comment declaring more parameter bytes // than the TEXT frame carries would slice past its end and panic. // Fail the whole campaign with a clear message instead. if ps := paramsSize(sig); ps > fl.Args { result.Mismatches = iterations result.FirstFail = fmt.Sprintf( "signature declares %d parameter bytes, but the TEXT frame of %s carries %d argument bytes", ps, name, fl.Args) return result } for i := range iterations { // Generate inputs and build TWO independent arg blocks (one per // version) so that functions which write to their arguments // (e.g. histogram increments) don't corrupt the other's input. gasmArgs, goArgs, bufs, entry := genDualArgs(rng, sig, fl.Args) entry.Func = name // Save the current input for crash diagnostics. result.CrashInput = gasmArgs // Call the gasm version. gasmOut, err := k.CallFunc(name, gasmArgs) if err != nil { result.Mismatches++ if result.FirstFail == "" { result.FirstFail = fmt.Sprintf("iter %d: gasm call: %v", i, err) } if onSave != nil { onSave(entry) } runtime.KeepAlive(bufs) continue } // Call the Go version (same function, independent buffers). goOut, err := Call(goExec.FuncAddr(0), goArgs) if err != nil { result.Mismatches++ if result.FirstFail == "" { result.FirstFail = fmt.Sprintf("iter %d: go call: %v", i, err) } if onSave != nil { onSave(entry) } runtime.KeepAlive(bufs) continue } // Compare only the result area (after all input parameters). // Pointers in the arg block differ (separate buffers), so we // compare from resultOff to the end. resultOff := paramsSize(sig) gasmRes := gasmOut[resultOff:] goRes := goOut[resultOff:] if !equalBytes(gasmRes, goRes) { result.Mismatches++ if result.FirstFail == "" { result.FirstFail = fmt.Sprintf("iter %d: output mismatch at result offset %d", i, resultOff) } if onSave != nil { onSave(entry) } } else { result.Matches++ } runtime.KeepAlive(bufs) } return result } // genDualArgs generates two independent ABI0 argument blocks (for gasm and // go) with identical logical content but separate backing buffers, so that // functions which write to their arguments don't corrupt the other's input. func genDualArgs(rng *rand.Rand, sig funcSig, argSize int) (gasmArgs, goArgs []byte, bufs [][]byte, entry CorpusEntry) { gasmArgs = make([]byte, argSize) goArgs = make([]byte, argSize) off := 0 sliceIdx := 0 for _, p := range sig.params { switch { case strings.HasPrefix(p.typ, "[]"): elemSize := elemSizeFor(p.typ) n := 1 + rng.IntN(127) var declaredLen int if sliceIdx == 0 { declaredLen = n } else { declaredLen = n + 512 } // Allocate a buffer comfortably larger than declaredLen*elemSize so // that SIMD over-reads and functions that write slightly past len // (e.g. decoders that trust len(src)) never touch unmapped memory. bufBytes := declaredLen*elemSize + 8192 // Two independent buffers with identical random content. buf1 := make([]byte, bufBytes) buf2 := make([]byte, bufBytes) fillRandom(rng, buf1[:n*elemSize]) copy(buf2, buf1) bufs = append(bufs, buf1, buf2) putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0])) putPtr(goArgs, off, unsafe.Pointer(&buf2[0])) // len and cap both equal declaredLen, the buffer is guaranteed // to hold at least declaredLen elements plus safety margin. putU64(gasmArgs, off+8, uint64(declaredLen)) putU64(gasmArgs, off+16, uint64(declaredLen)) putU64(goArgs, off+8, uint64(declaredLen)) putU64(goArgs, off+16, uint64(declaredLen)) off += 24 sliceIdx++ entry.Args = append(entry.Args, CorpusArg{ Kind: "slice", Len: declaredLen, Data: hex.EncodeToString(buf1[:n*elemSize]), }) case p.typ == "string": // An ABI0 string is a two-word header (data pointer + // length); a random pointer would fault kernels that read // the string, so the header points at a real buffer with // the same safety margin slices get. n := 1 + rng.IntN(127) buf1 := make([]byte, n+8192) buf2 := make([]byte, n+8192) fillRandom(rng, buf1[:n]) copy(buf2, buf1) bufs = append(bufs, buf1, buf2) putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0])) putPtr(goArgs, off, unsafe.Pointer(&buf2[0])) putU64(gasmArgs, off+8, uint64(n)) putU64(goArgs, off+8, uint64(n)) off += 16 entry.Args = append(entry.Args, CorpusArg{ Kind: "string", Len: n, Data: hex.EncodeToString(buf1[:n]), }) case strings.HasPrefix(p.typ, "*["): nElem := arrayLen(p.typ) elem := elemSizeFor("[]" + p.typ[strings.Index(p.typ, "]")+1:]) size := max(nElem*elem, 8) buf1 := make([]byte, size) buf2 := make([]byte, size) fillRandom(rng, buf1) copy(buf2, buf1) bufs = append(bufs, buf1, buf2) putPtr(gasmArgs, off, unsafe.Pointer(&buf1[0])) putPtr(goArgs, off, unsafe.Pointer(&buf2[0])) off += 8 entry.Args = append(entry.Args, CorpusArg{ Kind: "ptr", Data: hex.EncodeToString(buf1), }) case p.typ == "int" || p.typ == "uint" || p.typ == "int64" || p.typ == "uint64": v := uint64(rng.IntN(256)) putU64(gasmArgs, off, v) putU64(goArgs, off, v) off += 8 entry.Args = append(entry.Args, CorpusArg{Kind: "int", Value: strconv.FormatUint(v, 10)}) case p.typ == "complex64", p.typ == "complex128": // complex64 is two float32s (8 bytes), complex128 two // float64s (16): plain data words to the marshaller, one // corpus scalar per word so replay rebuilds them exactly. for range paramSize(p.typ) / 8 { v := rng.Uint64() putU64(gasmArgs, off, v) putU64(goArgs, off, v) off += 8 entry.Args = append(entry.Args, CorpusArg{Kind: "scalar", Value: strconv.FormatUint(v, 10)}) } default: v := rng.Uint64() putU64(gasmArgs, off, v) putU64(goArgs, off, v) off += 8 entry.Args = append(entry.Args, CorpusArg{Kind: "scalar", Value: strconv.FormatUint(v, 10)}) } } return gasmArgs, goArgs, bufs, entry } // ReplayEntry rebuilds the argument block of a corpus entry and invokes the // named function once, returning the argument block after the call. Slice // and string buffers get the same safety padding the fuzzer uses, so // over-reads that were harmless during the original run stay harmless on // replay. // // Not safe for concurrent use: only one JIT call may be in flight at a // time, the trampolines keep the saved registers in package globals. func (k *Kernel) ReplayEntry(name string, e CorpusEntry) ([]byte, error) { fl, err := k.Func(name) if err != nil { return nil, err } args := make([]byte, fl.Args) var bufs [][]byte off := 0 for _, a := range e.Args { switch a.Kind { case "slice": data, err := hex.DecodeString(a.Data) if err != nil { return nil, fmt.Errorf("corpus: slice data: %w", err) } buf := make([]byte, len(data)+8192) copy(buf, data) bufs = append(bufs, buf) if off+24 > len(args) { return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name) } putPtr(args, off, unsafe.Pointer(&buf[0])) putU64(args, off+8, uint64(a.Len)) putU64(args, off+16, uint64(a.Len)) off += 24 case "string": data, err := hex.DecodeString(a.Data) if err != nil { return nil, fmt.Errorf("corpus: string data: %w", err) } buf := make([]byte, len(data)+8192) copy(buf, data) bufs = append(bufs, buf) if off+16 > len(args) { return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name) } putPtr(args, off, unsafe.Pointer(&buf[0])) putU64(args, off+8, uint64(a.Len)) off += 16 case "ptr": data, err := hex.DecodeString(a.Data) if err != nil { return nil, fmt.Errorf("corpus: ptr data: %w", err) } buf := make([]byte, max(len(data), 8)) copy(buf, data) bufs = append(bufs, buf) if off+8 > len(args) { return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name) } putPtr(args, off, unsafe.Pointer(&buf[0])) off += 8 default: // "int", "scalar" v, err := strconv.ParseUint(a.Value, 10, 64) if err != nil { return nil, fmt.Errorf("corpus: %s value: %w", a.Kind, err) } if off+8 > len(args) { return nil, fmt.Errorf("corpus: entry does not fit the argument block of %s", name) } putU64(args, off, v) off += 8 } } out, err := k.CallFunc(name, args) runtime.KeepAlive(bufs) return out, err } func elemSizeFor(sliceType string) int { switch strings.TrimPrefix(sliceType, "[]") { case "byte", "uint8", "int8": return 1 case "uint16", "int16": return 2 case "uint32", "int32", "float32": return 4 case "uint64", "int64", "float64": return 8 default: return 8 } } // paramsSize returns the ABI0 stack size occupied by the input parameters. func paramsSize(sig funcSig) int { size := 0 for _, p := range sig.params { switch { case strings.HasPrefix(p.typ, "[]"): size += 24 // slice header case strings.HasPrefix(p.typ, "*["): size += 8 // pointer case p.typ == "bool": size += 1 case p.typ == "string": size += 16 // data pointer + length case p.typ == "complex64": size += 8 // two float32s case p.typ == "complex128": size += 16 // two float64s default: size += 8 // int, uint, etc. } } return size } func arrayLen(typ string) int { // "*[32]uint16" → 32 start := strings.Index(typ, "[") end := strings.Index(typ, "]") if start < 0 || end < 0 || end <= start { return 1 } n, _ := strconv.Atoi(typ[start+1 : end]) if n <= 0 { n = 1 } return n } func putPtr(buf []byte, off int, p unsafe.Pointer) { if off+8 <= len(buf) { u64 := uint64(uintptr(p)) buf[off] = byte(u64) buf[off+1] = byte(u64 >> 8) buf[off+2] = byte(u64 >> 16) buf[off+3] = byte(u64 >> 24) buf[off+4] = byte(u64 >> 32) buf[off+5] = byte(u64 >> 40) buf[off+6] = byte(u64 >> 48) buf[off+7] = byte(u64 >> 56) } } func putU64(buf []byte, off int, v uint64) { if off+8 <= len(buf) { buf[off] = byte(v) buf[off+1] = byte(v >> 8) buf[off+2] = byte(v >> 16) buf[off+3] = byte(v >> 24) buf[off+4] = byte(v >> 32) buf[off+5] = byte(v >> 40) buf[off+6] = byte(v >> 48) buf[off+7] = byte(v >> 56) } } func equalBytes(a, b []byte) bool { if len(a) != len(b) { return false } for i := range a { if a[i] != b[i] { return false } } return true }