2026-08-20 23:57:10 +02:00
|
|
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
|
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
|
|
|
|
|
|
//go:build linux
|
|
|
|
|
|
|
|
|
|
package debug
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
"fmt"
|
|
|
|
|
"os"
|
|
|
|
|
"os/exec"
|
|
|
|
|
"path/filepath"
|
2026-08-30 22:05:05 +02:00
|
|
|
"runtime"
|
2026-10-07 13:46:18 +02:00
|
|
|
"strconv"
|
2026-08-20 23:57:10 +02:00
|
|
|
"strings"
|
|
|
|
|
"syscall"
|
|
|
|
|
"time"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
// Session is a ptrace debugging session controlling one debuggee process.
|
|
|
|
|
type Session struct {
|
2026-10-07 13:46:18 +02:00
|
|
|
pid int
|
|
|
|
|
cmd *exec.Cmd
|
|
|
|
|
stopped bool
|
|
|
|
|
exited bool
|
|
|
|
|
// tid is the thread the trace relation actually landed on. The
|
|
|
|
|
// debuggee calls PTRACE_TRACEME from its target-mode goroutine, and the
|
|
|
|
|
// Go runtime may have migrated that goroutine off the process leader
|
|
|
|
|
// before the call, so the traced thread is not always pid (the leader).
|
|
|
|
|
// Every ptrace request and every wait must address the traced thread,
|
|
|
|
|
// while process-wide operations (signals and /proc/pid entries) keep
|
|
|
|
|
// using pid. It equals pid unless the handshake reported otherwise.
|
|
|
|
|
tid int
|
|
|
|
|
tracetidSeen bool // the handshake's traced-thread notice has been read
|
|
|
|
|
codeBase uint64 // base address of the JIT code in the debuggee
|
|
|
|
|
tmpDir string // scratch directory of the session, removed on Kill
|
|
|
|
|
wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 DBGWVR0-15)
|
2026-09-19 23:49:19 +02:00
|
|
|
// lastSignal holds the signal of the most recent stop when that stop
|
|
|
|
|
// was a genuine signal-delivery-stop the caller must see (a fault such
|
|
|
|
|
// as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps,
|
|
|
|
|
// single-steps, SIGSTOP and suppressed runtime signals.
|
|
|
|
|
lastSignal syscall.Signal
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Launch starts the debuggee subprocess (gasm debug --target ...) and
|
|
|
|
|
// attaches to it via ptrace.
|
|
|
|
|
func Launch(gasmBin, asmPath, funcName string, args []byte) (*Session, error) {
|
|
|
|
|
sess, _, err := LaunchWithBuffers(gasmBin, asmPath, funcName, args, "")
|
|
|
|
|
return sess, err
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// LaunchWithBuffers is like Launch but also allocates buffers in the debuggee.
|
2026-08-30 22:05:05 +02:00
|
|
|
//
|
|
|
|
|
// It pins the calling goroutine to its OS thread and leaves it pinned: the
|
|
|
|
|
// debuggee's PTRACE_TRACEME binds the tracer relation to the forking thread,
|
|
|
|
|
// and every ptrace request on the session must come from that same thread.
|
|
|
|
|
// All Session methods must therefore be called from the goroutine that
|
|
|
|
|
// launched the session (the REPL and coverage loops do exactly that).
|
2026-08-20 23:57:10 +02:00
|
|
|
func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec string) (*Session, []uint64, error) {
|
2026-08-30 22:05:05 +02:00
|
|
|
runtime.LockOSThread() // ptrace requests must stay on the forking thread
|
2026-08-20 23:57:10 +02:00
|
|
|
self, err := os.Executable()
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, nil, fmt.Errorf("debug: cannot find gasm binary: %w", err)
|
|
|
|
|
}
|
|
|
|
|
if gasmBin != "" {
|
|
|
|
|
self = gasmBin
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
tmpDir, err := os.MkdirTemp("", "gasm-debug-*")
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, nil, fmt.Errorf("debug: tempdir: %w", err)
|
|
|
|
|
}
|
|
|
|
|
argsFile := filepath.Join(tmpDir, "args.bin")
|
|
|
|
|
if err := os.WriteFile(argsFile, args, 0o644); err != nil {
|
|
|
|
|
os.RemoveAll(tmpDir)
|
|
|
|
|
return nil, nil, fmt.Errorf("debug: write args: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if bufSpec != "" {
|
|
|
|
|
if err := os.WriteFile(filepath.Join(tmpDir, "bufspec"), []byte(bufSpec), 0o644); err != nil {
|
|
|
|
|
os.RemoveAll(tmpDir)
|
|
|
|
|
return nil, nil, fmt.Errorf("debug: write bufspec: %w", err)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-30 11:00:40 +02:00
|
|
|
cmd := exec.Command(self, "debug", "--func", funcName, "--args", argsFile, asmPath)
|
|
|
|
|
cmd.Env = append(os.Environ(), "GASM_DEBUG_TARGET=1", "GASM_DEBUG_TMP="+tmpDir)
|
2026-08-20 23:57:10 +02:00
|
|
|
cmd.Stdout = nil
|
|
|
|
|
cmd.Stderr = os.Stderr
|
|
|
|
|
cmd.SysProcAttr = &syscall.SysProcAttr{}
|
|
|
|
|
|
|
|
|
|
if err := cmd.Start(); err != nil {
|
|
|
|
|
os.RemoveAll(tmpDir)
|
|
|
|
|
return nil, nil, fmt.Errorf("debug: start debuggee: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-19 23:49:19 +02:00
|
|
|
s := &Session{pid: cmd.Process.Pid, cmd: cmd, tmpDir: tmpDir}
|
2026-08-20 23:57:10 +02:00
|
|
|
|
|
|
|
|
readyFile := filepath.Join(tmpDir, "ready")
|
2026-08-29 15:40:31 +02:00
|
|
|
for range 500 {
|
2026-08-20 23:57:10 +02:00
|
|
|
if _, err := os.Stat(readyFile); err == nil {
|
|
|
|
|
break
|
|
|
|
|
}
|
2026-10-02 00:40:54 +02:00
|
|
|
// A debuggee that died before signalling readiness (unknown
|
|
|
|
|
// function, unparseable source) writes its failure notice to the
|
|
|
|
|
// handshake directory; read it and fail fast. The poll never
|
|
|
|
|
// waits on the child: a wait here could consume the SIGSTOP park
|
|
|
|
|
// that waitStopped below must receive, hanging the launch.
|
|
|
|
|
if err := s.deadReason(); err != nil {
|
|
|
|
|
cmd.Wait()
|
|
|
|
|
os.RemoveAll(tmpDir)
|
|
|
|
|
return nil, nil, err
|
|
|
|
|
}
|
2026-08-20 23:57:10 +02:00
|
|
|
time.Sleep(5 * time.Millisecond)
|
|
|
|
|
}
|
2026-08-30 22:05:05 +02:00
|
|
|
|
2026-10-07 13:46:18 +02:00
|
|
|
// The debuggee reports which thread called PTRACE_TRACEME in the
|
|
|
|
|
// handshake. The trace relation binds to that thread, so it is the
|
|
|
|
|
// one every ptrace request and every wait below must address; pid (the
|
|
|
|
|
// leader) stays the target for process-wide signals and /proc entries.
|
|
|
|
|
// A starved debuggee may deliver the notice only later; waitStopped
|
|
|
|
|
// adopts it the moment it appears.
|
|
|
|
|
s.adoptTracedThread()
|
|
|
|
|
|
2026-08-30 22:05:05 +02:00
|
|
|
// The debuggee parks itself with SIGSTOP once the JIT code is mapped.
|
|
|
|
|
// A Go tracee also reports SIGURG preemption as signal-delivery-stops,
|
|
|
|
|
// so the wait loops until a stop the debugger cares about instead of
|
|
|
|
|
// assuming the first event is the SIGSTOP.
|
2026-10-07 13:46:18 +02:00
|
|
|
if _, err := s.waitStopped(false); err != nil {
|
2026-08-20 23:57:10 +02:00
|
|
|
cmd.Process.Kill()
|
|
|
|
|
os.RemoveAll(tmpDir)
|
2026-08-30 22:05:05 +02:00
|
|
|
return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err)
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
s.stopped = true
|
|
|
|
|
|
2026-08-30 22:05:05 +02:00
|
|
|
// The debuggee reports its JIT mapping in the codebase file; that is the
|
|
|
|
|
// exact region the kernel was written to. Scanning /proc/pid/maps for
|
|
|
|
|
// any RWX region is only the fallback.
|
|
|
|
|
if data, err := os.ReadFile(filepath.Join(tmpDir, "codebase")); err == nil {
|
|
|
|
|
fmt.Sscanf(string(data), "%d", &s.codeBase)
|
|
|
|
|
}
|
2026-08-20 23:57:10 +02:00
|
|
|
if s.codeBase == 0 {
|
2026-08-30 22:05:05 +02:00
|
|
|
s.codeBase = findRWXMapping(s.pid)
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
var bufAddrs []uint64
|
|
|
|
|
if bufSpec != "" {
|
|
|
|
|
addrFile := filepath.Join(tmpDir, "bufaddrs")
|
|
|
|
|
if data, err := os.ReadFile(addrFile); err == nil {
|
2026-08-29 15:40:31 +02:00
|
|
|
for line := range strings.SplitSeq(strings.TrimSpace(string(data)), "\n") {
|
2026-08-20 23:57:10 +02:00
|
|
|
var addr uint64
|
|
|
|
|
if _, err := fmt.Sscanf(line, "%d", &addr); err == nil {
|
|
|
|
|
bufAddrs = append(bufAddrs, addr)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return s, bufAddrs, nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-02 00:40:54 +02:00
|
|
|
// deadReason reports the debuggee's own failure notice, the file its
|
|
|
|
|
// failure paths write before exiting. A debuggee killed without a notice
|
|
|
|
|
// (a crash, SIGKILL) surfaces through waitStopped after the poll instead,
|
|
|
|
|
// which is why the poll's budget stays finite.
|
|
|
|
|
func (s *Session) deadReason() error {
|
|
|
|
|
data, err := os.ReadFile(filepath.Join(s.tmpDir, "dead"))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
return fmt.Errorf("debug: debuggee failed before signalling readiness: %s", strings.TrimSpace(string(data)))
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-30 22:05:05 +02:00
|
|
|
// waitStopped consumes ptrace-stop events until one the debugger cares
|
2026-09-19 23:49:19 +02:00
|
|
|
// about arrives: SIGTRAP (a breakpoint or a completed single-step), the
|
|
|
|
|
// debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's
|
|
|
|
|
// runtime raises SIGURG for asynchronous preemption, and every signal on a
|
|
|
|
|
// traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed
|
2026-10-07 13:46:18 +02:00
|
|
|
// and the tracee resumed without it, with the request the caller issued: a
|
|
|
|
|
// single-step cancelled by the arriving signal must be re-issued, or the
|
|
|
|
|
// tracee would run uncontrolled past the one instruction it was told to
|
|
|
|
|
// execute. Every other signal (SIGSEGV, SIGBUS, SIGFPE, SIGILL, ...) is
|
|
|
|
|
// returned to the caller: resuming with signal 0 would restart the faulting
|
|
|
|
|
// instruction and fault forever, so a faulting kernel must surface as a stop
|
|
|
|
|
// the caller reports. Runtime noise is also why a single wait can return in
|
|
|
|
|
// the middle of runtime code and a resume can then fail: the event stream
|
|
|
|
|
// must be drained by the tracer.
|
|
|
|
|
//
|
|
|
|
|
// The wait itself polls without blocking: a launch whose readiness poll
|
|
|
|
|
// exhausted its budget may still be waiting for a starved debuggee to call
|
|
|
|
|
// PTRACE_TRACEME, and the relation may then land on a thread other than the
|
|
|
|
|
// leader, so the wait adopts the handshake's traced-thread notice the
|
|
|
|
|
// moment it appears instead of starving on the wrong target forever.
|
|
|
|
|
func (s *Session) waitStopped(stepping bool) (syscall.Signal, error) {
|
|
|
|
|
request, name := uintptr(syscall.PTRACE_CONT), "PTRACE_CONT"
|
|
|
|
|
if stepping {
|
|
|
|
|
request, name = uintptr(syscall.PTRACE_SINGLESTEP), "PTRACE_SINGLESTEP"
|
|
|
|
|
}
|
2026-08-30 22:05:05 +02:00
|
|
|
for {
|
|
|
|
|
var ws syscall.WaitStatus
|
2026-10-07 13:46:18 +02:00
|
|
|
wpid, err := syscall.Wait4(s.tid, &ws, syscall.WUNTRACED|syscall.WNOHANG, nil)
|
|
|
|
|
if err == syscall.EINTR {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
if err != nil {
|
2026-08-30 22:05:05 +02:00
|
|
|
return 0, err
|
|
|
|
|
}
|
2026-10-07 13:46:18 +02:00
|
|
|
if wpid == 0 {
|
|
|
|
|
s.adoptTracedThread()
|
|
|
|
|
time.Sleep(time.Millisecond)
|
|
|
|
|
continue
|
|
|
|
|
}
|
2026-08-30 22:05:05 +02:00
|
|
|
if ws.Exited() {
|
|
|
|
|
s.exited = true
|
|
|
|
|
return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus())
|
|
|
|
|
}
|
|
|
|
|
if ws.Signaled() {
|
|
|
|
|
s.exited = true
|
|
|
|
|
return 0, fmt.Errorf("debuggee killed by signal %v", ws.Signal())
|
|
|
|
|
}
|
|
|
|
|
switch sig := ws.StopSignal(); sig {
|
|
|
|
|
case syscall.SIGTRAP, syscall.SIGSTOP:
|
|
|
|
|
s.stopped = true
|
2026-09-19 23:49:19 +02:00
|
|
|
s.lastSignal = 0
|
2026-08-30 22:05:05 +02:00
|
|
|
return sig, nil
|
2026-09-19 23:49:19 +02:00
|
|
|
case syscall.SIGURG:
|
|
|
|
|
// Go runtime asynchronous preemption: resume the tracee
|
|
|
|
|
// without delivering the signal.
|
|
|
|
|
s.lastSignal = 0
|
2026-10-07 13:46:18 +02:00
|
|
|
if err := s.resume(request, name); err != nil {
|
|
|
|
|
return 0, err
|
2026-08-30 22:05:05 +02:00
|
|
|
}
|
2026-09-19 23:49:19 +02:00
|
|
|
default:
|
|
|
|
|
// A genuine signal-delivery-stop. Report it; the caller
|
|
|
|
|
// decides how to proceed.
|
|
|
|
|
s.stopped = true
|
|
|
|
|
s.lastSignal = sig
|
|
|
|
|
return sig, nil
|
2026-08-30 22:05:05 +02:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-07 13:46:18 +02:00
|
|
|
// adoptTracedThread reads the handshake's traced-thread notice once and
|
|
|
|
|
// retargets the session's ptrace requests and waits at the thread that
|
|
|
|
|
// called PTRACE_TRACEME. Until the notice exists the debuggee has not
|
|
|
|
|
// reached PTRACE_TRACEME, and the leader stays the target.
|
|
|
|
|
func (s *Session) adoptTracedThread() {
|
|
|
|
|
if s.tracetidSeen {
|
|
|
|
|
return
|
|
|
|
|
}
|
|
|
|
|
data, err := os.ReadFile(filepath.Join(s.tmpDir, "tracetid"))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return
|
|
|
|
|
}
|
|
|
|
|
s.tracetidSeen = true
|
|
|
|
|
if v, perr := strconv.Atoi(strings.TrimSpace(string(data))); perr == nil && v > 0 {
|
|
|
|
|
s.tid = v
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-19 23:49:19 +02:00
|
|
|
// LastSignal returns the signal of the most recent stop when that stop was
|
|
|
|
|
// a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL
|
|
|
|
|
// or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and
|
|
|
|
|
// suppressed runtime signals.
|
|
|
|
|
func (s *Session) LastSignal() syscall.Signal { return s.lastSignal }
|
|
|
|
|
|
2026-08-20 23:57:10 +02:00
|
|
|
// Peek reads a word (8 bytes) from the debuggee's memory at addr.
|
|
|
|
|
func (s *Session) Peek(addr uint64) (uint64, error) {
|
|
|
|
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return 0, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
|
|
|
|
}
|
|
|
|
|
defer mem.Close()
|
|
|
|
|
buf := make([]byte, 8)
|
|
|
|
|
if _, err := mem.ReadAt(buf, int64(addr)); err != nil {
|
|
|
|
|
return 0, fmt.Errorf("debug: read mem %#x: %w", addr, err)
|
|
|
|
|
}
|
|
|
|
|
return uint64(buf[0]) | uint64(buf[1])<<8 | uint64(buf[2])<<16 | uint64(buf[3])<<24 |
|
|
|
|
|
uint64(buf[4])<<32 | uint64(buf[5])<<40 | uint64(buf[6])<<48 | uint64(buf[7])<<56, nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Poke writes a word (8 bytes) to the debuggee's memory at addr.
|
|
|
|
|
func (s *Session) Poke(addr, val uint64) error {
|
|
|
|
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
|
|
|
|
}
|
|
|
|
|
defer mem.Close()
|
|
|
|
|
buf := []byte{byte(val), byte(val >> 8), byte(val >> 16), byte(val >> 24),
|
|
|
|
|
byte(val >> 32), byte(val >> 40), byte(val >> 48), byte(val >> 56)}
|
|
|
|
|
if _, err := mem.WriteAt(buf, int64(addr)); err != nil {
|
|
|
|
|
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
|
|
|
|
|
}
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-02 00:40:54 +02:00
|
|
|
// ReadMemory reads len bytes from the debuggee's memory at addr. The read
|
|
|
|
|
// covers exactly the requested range: the old word-at-a-time loop read a
|
|
|
|
|
// whole 8-byte word for the final partial word, so a request that ended
|
|
|
|
|
// inside the last mapped page failed whenever the following page was
|
|
|
|
|
// unmapped, even though every requested byte was readable.
|
2026-08-20 23:57:10 +02:00
|
|
|
func (s *Session) ReadMemory(addr uint64, length int) ([]byte, error) {
|
|
|
|
|
out := make([]byte, length)
|
2026-10-02 00:40:54 +02:00
|
|
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_RDONLY, 0)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return out, fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
|
|
|
|
}
|
|
|
|
|
defer mem.Close()
|
|
|
|
|
n, err := mem.ReadAt(out, int64(addr))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return out[:n], fmt.Errorf("debug: read mem %#x: %w", addr, err)
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
return out, nil
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-02 00:40:54 +02:00
|
|
|
// WriteMemory writes bytes to the debuggee's memory at addr. The write
|
|
|
|
|
// covers exactly the given bytes: /proc/pid/mem accepts writes of any
|
|
|
|
|
// length at any offset, so the word loop's read-modify-write of the final
|
|
|
|
|
// partial word (which read past the requested range and failed on an
|
|
|
|
|
// unmapped following page) is unnecessary.
|
2026-08-20 23:57:10 +02:00
|
|
|
func (s *Session) WriteMemory(addr uint64, data []byte) error {
|
2026-10-02 00:40:54 +02:00
|
|
|
if len(data) == 0 {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
mem, err := os.OpenFile(fmt.Sprintf("/proc/%d/mem", s.pid), os.O_WRONLY, 0)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return fmt.Errorf("debug: open /proc/%d/mem: %w", s.pid, err)
|
|
|
|
|
}
|
|
|
|
|
defer mem.Close()
|
|
|
|
|
if _, err := mem.WriteAt(data, int64(addr)); err != nil {
|
|
|
|
|
return fmt.Errorf("debug: write mem %#x: %w", addr, err)
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Step executes a single instruction in the debuggee.
|
|
|
|
|
func (s *Session) Step() error {
|
|
|
|
|
if s.exited {
|
|
|
|
|
return fmt.Errorf("debug: debuggee has exited")
|
|
|
|
|
}
|
2026-10-07 13:46:18 +02:00
|
|
|
if err := s.resume(syscall.PTRACE_SINGLESTEP, "PTRACE_SINGLESTEP"); err != nil {
|
|
|
|
|
return err
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
2026-10-07 13:46:18 +02:00
|
|
|
_, err := s.waitStopped(true)
|
2026-08-30 22:05:05 +02:00
|
|
|
return err
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Continue resumes execution until the next breakpoint or exit.
|
|
|
|
|
func (s *Session) Continue() error {
|
|
|
|
|
if s.exited {
|
|
|
|
|
return fmt.Errorf("debug: debuggee has exited")
|
|
|
|
|
}
|
2026-10-07 13:46:18 +02:00
|
|
|
if err := s.resume(syscall.PTRACE_CONT, "PTRACE_CONT"); err != nil {
|
|
|
|
|
return err
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
2026-10-07 13:46:18 +02:00
|
|
|
_, err := s.waitStopped(false)
|
2026-08-30 22:05:05 +02:00
|
|
|
return err
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
|
2026-10-07 13:46:18 +02:00
|
|
|
// resume restarts the stopped tracee with the given ptrace request. A
|
|
|
|
|
// tracee that slipped into a group-stop (a stop signal delivered to one of
|
|
|
|
|
// its untraced sibling threads stops the whole group without leaving a
|
|
|
|
|
// ptrace-stop to resume) is rejected by the kernel with ESRCH; SIGCONT
|
|
|
|
|
// lifts a group-stop, so one retry recovers the session when the tracee is
|
|
|
|
|
// merely group-stopped and lets a tracee past any lost trace relation run
|
|
|
|
|
// to its exit instead of wedging the session.
|
|
|
|
|
func (s *Session) resume(request uintptr, name string) error {
|
|
|
|
|
for range 2 {
|
|
|
|
|
_, _, errno := syscall.Syscall6(
|
|
|
|
|
syscall.SYS_PTRACE,
|
|
|
|
|
request,
|
|
|
|
|
uintptr(s.tid),
|
|
|
|
|
0, 0, 0, 0,
|
|
|
|
|
)
|
|
|
|
|
if errno == 0 {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
if errno != syscall.ESRCH {
|
|
|
|
|
return fmt.Errorf("debug: %s: %w", name, errno)
|
|
|
|
|
}
|
|
|
|
|
syscall.Kill(s.pid, syscall.SIGCONT)
|
|
|
|
|
}
|
|
|
|
|
return fmt.Errorf("debug: %s: %w", name, syscall.ESRCH)
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-20 23:57:10 +02:00
|
|
|
// Exited returns true if the debuggee has terminated.
|
|
|
|
|
func (s *Session) Exited() bool { return s.exited }
|
|
|
|
|
|
|
|
|
|
// Pid returns the debuggee's process ID.
|
|
|
|
|
func (s *Session) Pid() int { return s.pid }
|
|
|
|
|
|
|
|
|
|
// CodeBase returns the base address of the JIT code in the debuggee.
|
|
|
|
|
func (s *Session) CodeBase() uint64 { return s.codeBase }
|
|
|
|
|
|
2026-10-07 13:46:18 +02:00
|
|
|
// killReapBudget bounds how long Kill waits for the SIGKILLed debuggee to
|
|
|
|
|
// become waitable. The normal case reaps in single-digit milliseconds; the
|
|
|
|
|
// budget only matters for a tracee wedged past any resume, where hanging the
|
|
|
|
|
// caller forever would repeat the very defect Kill exists to end.
|
|
|
|
|
const killReapBudget = 2 * time.Second
|
|
|
|
|
|
2026-09-19 23:49:19 +02:00
|
|
|
// Kill terminates the debuggee and removes the session's scratch
|
|
|
|
|
// directory, so a successful session leaves no gasm-debug-* debris behind.
|
2026-10-07 13:46:18 +02:00
|
|
|
//
|
|
|
|
|
// The sequence must return for a debuggee in any state: running, parked in
|
|
|
|
|
// a ptrace-stop, already dead, or already reaped by an earlier wait.
|
|
|
|
|
// SIGKILL alone does not reliably wake a traced, stopped child (the stop can
|
|
|
|
|
// outrank the kill on the tracee's way out), and a blocking Wait4 on such a
|
|
|
|
|
// tracee then never returns, so the tracee is resumed first, killed second
|
|
|
|
|
// and reaped through a non-blocking wait loop that treats ECHILD as done.
|
2026-08-20 23:57:10 +02:00
|
|
|
func (s *Session) Kill() {
|
|
|
|
|
if !s.exited {
|
2026-10-07 13:46:18 +02:00
|
|
|
s.resumeTracee()
|
2026-08-20 23:57:10 +02:00
|
|
|
syscall.Kill(s.pid, syscall.SIGKILL)
|
2026-10-07 13:46:18 +02:00
|
|
|
if s.reapTracee() {
|
|
|
|
|
// cmd.Wait collects the exec.Cmd bookkeeping; the child is
|
|
|
|
|
// gone, so it returns at once. A tracee wedged past the reap
|
|
|
|
|
// budget is left to the kernel instead: waiting here would
|
|
|
|
|
// hang the caller exactly the way the blocking Wait4 did.
|
|
|
|
|
s.waitCmd()
|
|
|
|
|
}
|
2026-08-20 23:57:10 +02:00
|
|
|
s.exited = true
|
2026-10-07 13:46:18 +02:00
|
|
|
} else {
|
|
|
|
|
s.waitCmd()
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
2026-09-19 23:49:19 +02:00
|
|
|
if s.tmpDir != "" {
|
|
|
|
|
os.RemoveAll(s.tmpDir)
|
|
|
|
|
s.tmpDir = ""
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-07 13:46:18 +02:00
|
|
|
// waitCmd collects the exec.Cmd bookkeeping, bounded by the same budget as
|
|
|
|
|
// the raw reap: a debuggee whose thread group stays wedged past the kill is
|
|
|
|
|
// left to the kernel rather than allowed to hang the caller here.
|
|
|
|
|
func (s *Session) waitCmd() {
|
|
|
|
|
if s.cmd == nil || s.cmd.Process == nil {
|
|
|
|
|
return
|
|
|
|
|
}
|
|
|
|
|
done := make(chan struct{})
|
|
|
|
|
go func() {
|
|
|
|
|
s.cmd.Wait()
|
|
|
|
|
close(done)
|
|
|
|
|
}()
|
|
|
|
|
select {
|
|
|
|
|
case <-done:
|
|
|
|
|
case <-time.After(killReapBudget):
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// resumeTracee wakes the tracee out of any ptrace-stop, ignoring failures:
|
|
|
|
|
// ESRCH means the tracee is already gone or running, and either way the
|
|
|
|
|
// SIGKILL that follows needs no help from this side.
|
|
|
|
|
func (s *Session) resumeTracee() {
|
|
|
|
|
syscall.Syscall6(
|
|
|
|
|
syscall.SYS_PTRACE,
|
|
|
|
|
uintptr(syscall.PTRACE_CONT),
|
|
|
|
|
uintptr(s.tid),
|
|
|
|
|
0, 0, 0, 0,
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// reapTracee collects the killed debuggee with non-blocking waits until it
|
|
|
|
|
// is reaped, gone (ECHILD: no longer a waitable child, someone collected it
|
|
|
|
|
// already) or past the budget. It reports whether the debuggee was
|
|
|
|
|
// collected or is gone; false means the tracee stayed wedged and the caller
|
|
|
|
|
// must not block on it.
|
|
|
|
|
func (s *Session) reapTracee() bool {
|
|
|
|
|
half := time.Now().Add(killReapBudget / 2)
|
|
|
|
|
deadline := time.Now().Add(killReapBudget)
|
|
|
|
|
prodded := false
|
|
|
|
|
for {
|
|
|
|
|
var ws syscall.WaitStatus
|
|
|
|
|
wpid, err := syscall.Wait4(s.tid, &ws, syscall.WNOHANG, nil)
|
|
|
|
|
if err == syscall.EINTR {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
if err != nil {
|
|
|
|
|
// ECHILD and any other wait error: the debuggee is not ours to
|
|
|
|
|
// wait for any more. Done.
|
|
|
|
|
return true
|
|
|
|
|
}
|
|
|
|
|
if wpid == s.tid {
|
|
|
|
|
if ws.Exited() || ws.Signaled() {
|
|
|
|
|
return true // reaped
|
|
|
|
|
}
|
|
|
|
|
// A stop, not a death: the tracee is still alive with the
|
|
|
|
|
// SIGKILL pending. Keep polling rather than treating the
|
|
|
|
|
// consumed stop as an exit, which would hang the caller on
|
|
|
|
|
// the cmd.Wait that follows.
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
now := time.Now()
|
|
|
|
|
if now.After(deadline) {
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
if !prodded && now.After(half) {
|
|
|
|
|
// Half the budget gone without a collectable exit: poke the
|
|
|
|
|
// tracee once more, the way the first resume should have.
|
|
|
|
|
s.resumeTracee()
|
|
|
|
|
syscall.Kill(s.pid, syscall.SIGKILL)
|
|
|
|
|
prodded = true
|
|
|
|
|
}
|
|
|
|
|
time.Sleep(time.Millisecond)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-09-19 23:49:19 +02:00
|
|
|
// execRange is one executable mapping of the debuggee.
|
|
|
|
|
type execRange struct {
|
|
|
|
|
lo, hi uint64
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// execRanges parses the debuggee's executable mappings from /proc/pid/maps.
|
|
|
|
|
func execRanges(pid int) []execRange {
|
|
|
|
|
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil
|
|
|
|
|
}
|
|
|
|
|
var out []execRange
|
|
|
|
|
for line := range strings.SplitSeq(string(data), "\n") {
|
|
|
|
|
fields := strings.Fields(line)
|
|
|
|
|
if len(fields) < 2 || !strings.Contains(fields[1], "x") {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
var lo, hi uint64
|
|
|
|
|
if _, err := fmt.Sscanf(fields[0], "%x-%x", &lo, &hi); err == nil {
|
|
|
|
|
out = append(out, execRange{lo, hi})
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return out
|
2026-08-20 23:57:10 +02:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// findRWXMapping reads /proc/pid/maps and returns the base address of the
|
|
|
|
|
// first read-write-execute mapping (the JIT code region).
|
|
|
|
|
func findRWXMapping(pid int) uint64 {
|
|
|
|
|
data, err := os.ReadFile(fmt.Sprintf("/proc/%d/maps", pid))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return 0
|
|
|
|
|
}
|
2026-08-29 15:40:31 +02:00
|
|
|
for line := range strings.SplitSeq(string(data), "\n") {
|
2026-08-20 23:57:10 +02:00
|
|
|
fields := strings.Fields(line)
|
|
|
|
|
if len(fields) < 2 {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
perms := fields[1]
|
|
|
|
|
if len(perms) >= 3 && perms[0] == 'r' && perms[1] == 'w' && perms[2] == 'x' {
|
|
|
|
|
var start uint64
|
|
|
|
|
fmt.Sscanf(fields[0], "%x-", &start)
|
|
|
|
|
return start
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
return 0
|
|
|
|
|
}
|