From b5d6f1b46af84fd8953b39e09d986797180d27fe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Wed, 7 Oct 2026 13:46:18 +0200 Subject: [PATCH] fix(debug): target the traced thread and keep the kill from ever blocking The Go runtime can migrate the debuggee's target-mode goroutine off the process leader before PTRACE_TRACEME, which left the trace relation on a thread the session never addressed: its stops starved the waits on the leader, and a kill sequence that resumed nothing and then blocked in Wait4 hung the whole package. The debuggee now reports the traced thread in the launch handshake and parks with a thread-directed stop, every ptrace request and wait addresses that thread, a SIGURG arriving on a single-step resumes it as a single-step again instead of letting the tracee run uncontrolled, a resume rejected with ESRCH lifts a group-stop with SIGCONT and retries once, and Kill resumes, kills and reaps through non-blocking waits so it returns for a tracee in any state. Assisted-by: GLM 5.3 --- debug/ptrace_linux.go | 251 ++++++++++++++++++++++++------ debug/ptrace_linux_amd64.go | 8 +- debug/ptrace_linux_arm64.go | 6 +- debug/ptrace_linux_loong64.go | 6 +- debug/ptrace_linux_riscv64.go | 6 +- debug/stopinfo_linux.go | 2 +- debug/target_linux.go | 12 ++ debug/target_linux_amd64.go | 7 +- debug/target_linux_arm64.go | 7 +- debug/target_linux_loong64.go | 7 +- debug/target_linux_riscv64.go | 7 +- debug/watchpoint_linux_amd64.go | 18 +-- debug/watchpoint_linux_arm64.go | 4 +- debug/watchpoint_linux_loong64.go | 4 +- 14 files changed, 265 insertions(+), 80 deletions(-) diff --git a/debug/ptrace_linux.go b/debug/ptrace_linux.go index 3369823..10fd07d 100644 --- a/debug/ptrace_linux.go +++ b/debug/ptrace_linux.go @@ -11,6 +11,7 @@ import ( "os/exec" "path/filepath" "runtime" + "strconv" "strings" "syscall" "time" @@ -18,13 +19,22 @@ import ( // Session is a ptrace debugging session controlling one debuggee process. type Session struct { - pid int - cmd *exec.Cmd - stopped bool - exited bool - codeBase uint64 // base address of the JIT code in the debuggee - tmpDir string // scratch directory of the session, removed on Kill - wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 DBGWVR0-15) + pid int + cmd *exec.Cmd + stopped bool + exited bool + // tid is the thread the trace relation actually landed on. The + // debuggee calls PTRACE_TRACEME from its target-mode goroutine, and the + // Go runtime may have migrated that goroutine off the process leader + // before the call, so the traced thread is not always pid (the leader). + // Every ptrace request and every wait must address the traced thread, + // while process-wide operations (signals and /proc/pid entries) keep + // using pid. It equals pid unless the handshake reported otherwise. + tid int + tracetidSeen bool // the handshake's traced-thread notice has been read + codeBase uint64 // base address of the JIT code in the debuggee + tmpDir string // scratch directory of the session, removed on Kill + wpSlots [16]bool // hardware watchpoint slots in use (DR0-DR3, arm64 DBGWVR0-15) // lastSignal holds the signal of the most recent stop when that stop // was a genuine signal-delivery-stop the caller must see (a fault such // as SIGSEGV, SIGBUS, SIGFPE or SIGILL); 0 for breakpoint traps, @@ -104,11 +114,19 @@ func LaunchWithBuffers(gasmBin, asmPath, funcName string, args []byte, bufSpec s time.Sleep(5 * time.Millisecond) } + // The debuggee reports which thread called PTRACE_TRACEME in the + // handshake. The trace relation binds to that thread, so it is the + // one every ptrace request and every wait below must address; pid (the + // leader) stays the target for process-wide signals and /proc entries. + // A starved debuggee may deliver the notice only later; waitStopped + // adopts it the moment it appears. + s.adoptTracedThread() + // The debuggee parks itself with SIGSTOP once the JIT code is mapped. // A Go tracee also reports SIGURG preemption as signal-delivery-stops, // so the wait loops until a stop the debugger cares about instead of // assuming the first event is the SIGSTOP. - if _, err := s.waitStopped(); err != nil { + if _, err := s.waitStopped(false); err != nil { cmd.Process.Kill() os.RemoveAll(tmpDir) return nil, nil, fmt.Errorf("debug: wait for debuggee: %w", err) @@ -158,18 +176,40 @@ func (s *Session) deadReason() error { // debuggee's own SIGSTOP, or a genuine signal-delivery-stop. A Go tracee's // runtime raises SIGURG for asynchronous preemption, and every signal on a // traced thread surfaces as a signal-delivery-stop, so SIGURG is suppressed -// and the tracee resumed without it. Every other signal (SIGSEGV, SIGBUS, -// SIGFPE, SIGILL, ...) is returned to the caller: resuming with signal 0 -// would restart the faulting instruction and fault forever, so a faulting -// kernel must surface as a stop the caller reports. Runtime noise is also -// why a single wait can return in the middle of runtime code and a resume -// can then fail: the event stream must be drained by the tracer. -func (s *Session) waitStopped() (syscall.Signal, error) { +// and the tracee resumed without it, with the request the caller issued: a +// single-step cancelled by the arriving signal must be re-issued, or the +// tracee would run uncontrolled past the one instruction it was told to +// execute. Every other signal (SIGSEGV, SIGBUS, SIGFPE, SIGILL, ...) is +// returned to the caller: resuming with signal 0 would restart the faulting +// instruction and fault forever, so a faulting kernel must surface as a stop +// the caller reports. Runtime noise is also why a single wait can return in +// the middle of runtime code and a resume can then fail: the event stream +// must be drained by the tracer. +// +// The wait itself polls without blocking: a launch whose readiness poll +// exhausted its budget may still be waiting for a starved debuggee to call +// PTRACE_TRACEME, and the relation may then land on a thread other than the +// leader, so the wait adopts the handshake's traced-thread notice the +// moment it appears instead of starving on the wrong target forever. +func (s *Session) waitStopped(stepping bool) (syscall.Signal, error) { + request, name := uintptr(syscall.PTRACE_CONT), "PTRACE_CONT" + if stepping { + request, name = uintptr(syscall.PTRACE_SINGLESTEP), "PTRACE_SINGLESTEP" + } for { var ws syscall.WaitStatus - if _, err := syscall.Wait4(s.pid, &ws, syscall.WUNTRACED, nil); err != nil { + wpid, err := syscall.Wait4(s.tid, &ws, syscall.WUNTRACED|syscall.WNOHANG, nil) + if err == syscall.EINTR { + continue + } + if err != nil { return 0, err } + if wpid == 0 { + s.adoptTracedThread() + time.Sleep(time.Millisecond) + continue + } if ws.Exited() { s.exited = true return 0, fmt.Errorf("debuggee exited with status %d", ws.ExitStatus()) @@ -187,13 +227,8 @@ func (s *Session) waitStopped() (syscall.Signal, error) { // Go runtime asynchronous preemption: resume the tracee // without delivering the signal. s.lastSignal = 0 - if _, _, errno := syscall.Syscall6( - syscall.SYS_PTRACE, - uintptr(syscall.PTRACE_CONT), - uintptr(s.pid), - 0, 0, 0, 0, - ); errno != 0 { - return 0, fmt.Errorf("debug: PTRACE_CONT: %w", errno) + if err := s.resume(request, name); err != nil { + return 0, err } default: // A genuine signal-delivery-stop. Report it; the caller @@ -205,6 +240,24 @@ func (s *Session) waitStopped() (syscall.Signal, error) { } } +// adoptTracedThread reads the handshake's traced-thread notice once and +// retargets the session's ptrace requests and waits at the thread that +// called PTRACE_TRACEME. Until the notice exists the debuggee has not +// reached PTRACE_TRACEME, and the leader stays the target. +func (s *Session) adoptTracedThread() { + if s.tracetidSeen { + return + } + data, err := os.ReadFile(filepath.Join(s.tmpDir, "tracetid")) + if err != nil { + return + } + s.tracetidSeen = true + if v, perr := strconv.Atoi(strings.TrimSpace(string(data))); perr == nil && v > 0 { + s.tid = v + } +} + // LastSignal returns the signal of the most recent stop when that stop was // a genuine signal-delivery-stop (a fault such as SIGSEGV, SIGFPE, SIGILL // or SIGBUS), and 0 for breakpoint traps, single-steps, SIGSTOP and @@ -285,16 +338,10 @@ func (s *Session) Step() error { if s.exited { return fmt.Errorf("debug: debuggee has exited") } - _, _, errno := syscall.Syscall6( - syscall.SYS_PTRACE, - uintptr(syscall.PTRACE_SINGLESTEP), - uintptr(s.pid), - 0, 0, 0, 0, - ) - if errno != 0 { - return fmt.Errorf("debug: PTRACE_SINGLESTEP: %w", errno) + if err := s.resume(syscall.PTRACE_SINGLESTEP, "PTRACE_SINGLESTEP"); err != nil { + return err } - _, err := s.waitStopped() + _, err := s.waitStopped(true) return err } @@ -303,19 +350,39 @@ func (s *Session) Continue() error { if s.exited { return fmt.Errorf("debug: debuggee has exited") } - _, _, errno := syscall.Syscall6( - syscall.SYS_PTRACE, - uintptr(syscall.PTRACE_CONT), - uintptr(s.pid), - 0, 0, 0, 0, - ) - if errno != 0 { - return fmt.Errorf("debug: PTRACE_CONT: %w", errno) + if err := s.resume(syscall.PTRACE_CONT, "PTRACE_CONT"); err != nil { + return err } - _, err := s.waitStopped() + _, err := s.waitStopped(false) return err } +// resume restarts the stopped tracee with the given ptrace request. A +// tracee that slipped into a group-stop (a stop signal delivered to one of +// its untraced sibling threads stops the whole group without leaving a +// ptrace-stop to resume) is rejected by the kernel with ESRCH; SIGCONT +// lifts a group-stop, so one retry recovers the session when the tracee is +// merely group-stopped and lets a tracee past any lost trace relation run +// to its exit instead of wedging the session. +func (s *Session) resume(request uintptr, name string) error { + for range 2 { + _, _, errno := syscall.Syscall6( + syscall.SYS_PTRACE, + request, + uintptr(s.tid), + 0, 0, 0, 0, + ) + if errno == 0 { + return nil + } + if errno != syscall.ESRCH { + return fmt.Errorf("debug: %s: %w", name, errno) + } + syscall.Kill(s.pid, syscall.SIGCONT) + } + return fmt.Errorf("debug: %s: %w", name, syscall.ESRCH) +} + // Exited returns true if the debuggee has terminated. func (s *Session) Exited() bool { return s.exited } @@ -325,16 +392,35 @@ func (s *Session) Pid() int { return s.pid } // CodeBase returns the base address of the JIT code in the debuggee. func (s *Session) CodeBase() uint64 { return s.codeBase } +// killReapBudget bounds how long Kill waits for the SIGKILLed debuggee to +// become waitable. The normal case reaps in single-digit milliseconds; the +// budget only matters for a tracee wedged past any resume, where hanging the +// caller forever would repeat the very defect Kill exists to end. +const killReapBudget = 2 * time.Second + // Kill terminates the debuggee and removes the session's scratch // directory, so a successful session leaves no gasm-debug-* debris behind. +// +// The sequence must return for a debuggee in any state: running, parked in +// a ptrace-stop, already dead, or already reaped by an earlier wait. +// SIGKILL alone does not reliably wake a traced, stopped child (the stop can +// outrank the kill on the tracee's way out), and a blocking Wait4 on such a +// tracee then never returns, so the tracee is resumed first, killed second +// and reaped through a non-blocking wait loop that treats ECHILD as done. func (s *Session) Kill() { if !s.exited { + s.resumeTracee() syscall.Kill(s.pid, syscall.SIGKILL) - syscall.Wait4(s.pid, nil, 0, nil) + if s.reapTracee() { + // cmd.Wait collects the exec.Cmd bookkeeping; the child is + // gone, so it returns at once. A tracee wedged past the reap + // budget is left to the kernel instead: waiting here would + // hang the caller exactly the way the blocking Wait4 did. + s.waitCmd() + } s.exited = true - } - if s.cmd != nil && s.cmd.Process != nil { - s.cmd.Wait() + } else { + s.waitCmd() } if s.tmpDir != "" { os.RemoveAll(s.tmpDir) @@ -342,6 +428,81 @@ func (s *Session) Kill() { } } +// waitCmd collects the exec.Cmd bookkeeping, bounded by the same budget as +// the raw reap: a debuggee whose thread group stays wedged past the kill is +// left to the kernel rather than allowed to hang the caller here. +func (s *Session) waitCmd() { + if s.cmd == nil || s.cmd.Process == nil { + return + } + done := make(chan struct{}) + go func() { + s.cmd.Wait() + close(done) + }() + select { + case <-done: + case <-time.After(killReapBudget): + } +} + +// resumeTracee wakes the tracee out of any ptrace-stop, ignoring failures: +// ESRCH means the tracee is already gone or running, and either way the +// SIGKILL that follows needs no help from this side. +func (s *Session) resumeTracee() { + syscall.Syscall6( + syscall.SYS_PTRACE, + uintptr(syscall.PTRACE_CONT), + uintptr(s.tid), + 0, 0, 0, 0, + ) +} + +// reapTracee collects the killed debuggee with non-blocking waits until it +// is reaped, gone (ECHILD: no longer a waitable child, someone collected it +// already) or past the budget. It reports whether the debuggee was +// collected or is gone; false means the tracee stayed wedged and the caller +// must not block on it. +func (s *Session) reapTracee() bool { + half := time.Now().Add(killReapBudget / 2) + deadline := time.Now().Add(killReapBudget) + prodded := false + for { + var ws syscall.WaitStatus + wpid, err := syscall.Wait4(s.tid, &ws, syscall.WNOHANG, nil) + if err == syscall.EINTR { + continue + } + if err != nil { + // ECHILD and any other wait error: the debuggee is not ours to + // wait for any more. Done. + return true + } + if wpid == s.tid { + if ws.Exited() || ws.Signaled() { + return true // reaped + } + // A stop, not a death: the tracee is still alive with the + // SIGKILL pending. Keep polling rather than treating the + // consumed stop as an exit, which would hang the caller on + // the cmd.Wait that follows. + continue + } + now := time.Now() + if now.After(deadline) { + return false + } + if !prodded && now.After(half) { + // Half the budget gone without a collectable exit: poke the + // tracee once more, the way the first resume should have. + s.resumeTracee() + syscall.Kill(s.pid, syscall.SIGKILL) + prodded = true + } + time.Sleep(time.Millisecond) + } +} + // execRange is one executable mapping of the debuggee. type execRange struct { lo, hi uint64 diff --git a/debug/ptrace_linux_amd64.go b/debug/ptrace_linux_amd64.go index e98b4cc..0e1ebfb 100644 --- a/debug/ptrace_linux_amd64.go +++ b/debug/ptrace_linux_amd64.go @@ -18,7 +18,7 @@ func (s *Session) GetRegs() (Regs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(®s)), 0, 0, @@ -34,7 +34,7 @@ func (s *Session) SetRegs(regs *Regs) error { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_SETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(regs)), 0, 0, @@ -71,7 +71,7 @@ func (s *Session) GetFPRegs() (FPRegs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETFPREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(&fp)), 0, 0, @@ -125,7 +125,7 @@ func (s *Session) GetVectorRegs() (VectorRegs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGSET), - uintptr(s.pid), + uintptr(s.tid), uintptr(ntX86XState), uintptr(unsafe.Pointer(&iovec)), 0, 0, diff --git a/debug/ptrace_linux_arm64.go b/debug/ptrace_linux_arm64.go index c497d85..a718cb0 100644 --- a/debug/ptrace_linux_arm64.go +++ b/debug/ptrace_linux_arm64.go @@ -17,7 +17,7 @@ func (s *Session) GetRegs() (Regs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(®s)), 0, 0, @@ -33,7 +33,7 @@ func (s *Session) SetRegs(regs *Regs) error { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_SETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(regs)), 0, 0, @@ -66,7 +66,7 @@ func (s *Session) GetFPRegs() (FPRegs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGSET), - uintptr(s.pid), + uintptr(s.tid), uintptr(ntPrFPREG), uintptr(unsafe.Pointer(&iovec)), 0, 0, diff --git a/debug/ptrace_linux_loong64.go b/debug/ptrace_linux_loong64.go index d025794..68af49d 100644 --- a/debug/ptrace_linux_loong64.go +++ b/debug/ptrace_linux_loong64.go @@ -17,7 +17,7 @@ func (s *Session) GetRegs() (Regs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(®s)), 0, 0, @@ -33,7 +33,7 @@ func (s *Session) SetRegs(regs *Regs) error { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_SETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(regs)), 0, 0, @@ -67,7 +67,7 @@ func (s *Session) GetFPRegs() (FPRegs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGSET), - uintptr(s.pid), + uintptr(s.tid), uintptr(ntPrFPREG), uintptr(unsafe.Pointer(&iovec)), 0, 0, diff --git a/debug/ptrace_linux_riscv64.go b/debug/ptrace_linux_riscv64.go index 090046e..013f771 100644 --- a/debug/ptrace_linux_riscv64.go +++ b/debug/ptrace_linux_riscv64.go @@ -17,7 +17,7 @@ func (s *Session) GetRegs() (Regs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(®s)), 0, 0, @@ -33,7 +33,7 @@ func (s *Session) SetRegs(regs *Regs) error { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_SETREGS), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(regs)), 0, 0, @@ -65,7 +65,7 @@ func (s *Session) GetFPRegs() (FPRegs, error) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGSET), - uintptr(s.pid), + uintptr(s.tid), uintptr(ntPrFPREG), uintptr(unsafe.Pointer(&iovec)), 0, 0, diff --git a/debug/stopinfo_linux.go b/debug/stopinfo_linux.go index 26b6d97..fcf17ea 100644 --- a/debug/stopinfo_linux.go +++ b/debug/stopinfo_linux.go @@ -47,7 +47,7 @@ func (s *Session) StopInfo() (StopReason, uint64) { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETSIGINFO), - uintptr(s.pid), + uintptr(s.tid), 0, uintptr(unsafe.Pointer(&info)), 0, 0, diff --git a/debug/target_linux.go b/debug/target_linux.go index b4a29ba..50aa840 100644 --- a/debug/target_linux.go +++ b/debug/target_linux.go @@ -29,6 +29,18 @@ func mapRWX(code []byte) ([]byte, error) { return mem, nil } +// parkSelf stops the calling thread for the launch handshake. The stop is +// thread-directed on purpose: a process-directed SIGSTOP may be queued on +// any of the debuggee's untraced runtime threads, and delivering it there +// establishes a group-stop over the whole thread group, which leaves the +// traced leader in a plain stop the tracer cannot resume (PTRACE_CONT fails +// ESRCH) and the siblings stopped with nobody to continue them. Directed +// at the traced thread, the stop always surfaces as the signal-delivery-stop +// the handshake consumes. +func parkSelf() { + syscall.Tgkill(syscall.Getpid(), syscall.Gettid(), syscall.SIGSTOP) +} + // setupBuffers allocates buffers in the debuggee's memory. func setupBuffers(spec string, args []byte, tmpDir string) ([]byte, error) { type bufSpec struct { diff --git a/debug/target_linux_amd64.go b/debug/target_linux_amd64.go index da14e95..535e692 100644 --- a/debug/target_linux_amd64.go +++ b/debug/target_linux_amd64.go @@ -99,11 +99,14 @@ func runTarget(asmPath, funcName, argsFile, tmpDir string) error { if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 { return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno) } + // The tracer must address the thread that called PTRACE_TRACEME, which + // is not necessarily the process leader: report it before readiness. + os.WriteFile(tmpDir+"/tracetid", []byte(fmt.Sprintf("%d", syscall.Gettid())), 0o644) os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() fnAddr := codeBase + uintptr(fl.Offset) stackArgs := make([]byte, fl.Args) diff --git a/debug/target_linux_arm64.go b/debug/target_linux_arm64.go index 2be53e2..752f204 100644 --- a/debug/target_linux_arm64.go +++ b/debug/target_linux_arm64.go @@ -100,11 +100,14 @@ func runTarget(asmPath, funcName, argsFile, tmpDir string) error { if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 { return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno) } + // The tracer must address the thread that called PTRACE_TRACEME, which + // is not necessarily the process leader: report it before readiness. + os.WriteFile(tmpDir+"/tracetid", []byte(fmt.Sprintf("%d", syscall.Gettid())), 0o644) os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() fnAddr := codeBase + uintptr(fl.Offset) stackArgs := make([]byte, fl.Args) diff --git a/debug/target_linux_loong64.go b/debug/target_linux_loong64.go index a044ead..2a3506b 100644 --- a/debug/target_linux_loong64.go +++ b/debug/target_linux_loong64.go @@ -100,11 +100,14 @@ func runTarget(asmPath, funcName, argsFile, tmpDir string) error { if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 { return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno) } + // The tracer must address the thread that called PTRACE_TRACEME, which + // is not necessarily the process leader: report it before readiness. + os.WriteFile(tmpDir+"/tracetid", []byte(fmt.Sprintf("%d", syscall.Gettid())), 0o644) os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() fnAddr := codeBase + uintptr(fl.Offset) stackArgs := make([]byte, fl.Args) diff --git a/debug/target_linux_riscv64.go b/debug/target_linux_riscv64.go index 46cd728..027a43e 100644 --- a/debug/target_linux_riscv64.go +++ b/debug/target_linux_riscv64.go @@ -100,11 +100,14 @@ func runTarget(asmPath, funcName, argsFile, tmpDir string) error { if _, _, errno := syscall.Syscall(syscall.SYS_PTRACE, uintptr(syscall.PTRACE_TRACEME), 0, 0); errno != 0 { return fmt.Errorf("debug target: PTRACE_TRACEME: %v", errno) } + // The tracer must address the thread that called PTRACE_TRACEME, which + // is not necessarily the process leader: report it before readiness. + os.WriteFile(tmpDir+"/tracetid", []byte(fmt.Sprintf("%d", syscall.Gettid())), 0o644) os.WriteFile(tmpDir+"/ready", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() os.WriteFile(tmpDir+"/entry", []byte("ok"), 0o644) - syscall.Kill(syscall.Getpid(), syscall.SIGSTOP) + parkSelf() fnAddr := codeBase + uintptr(fl.Offset) stackArgs := make([]byte, fl.Args) diff --git a/debug/watchpoint_linux_amd64.go b/debug/watchpoint_linux_amd64.go index 384d9ee..f2a85c3 100644 --- a/debug/watchpoint_linux_amd64.go +++ b/debug/watchpoint_linux_amd64.go @@ -36,7 +36,7 @@ const ( // read, or the next hit on a different slot would still see this slot's bit // set and report this slot's address again. func archWatchpointAddr(s *Session, siAddr uint64) uint64 { - dr6, err := ptracePeekUser(s.pid, dr6Off) + dr6, err := ptracePeekUser(s.tid, dr6Off) if err != nil { return siAddr } @@ -46,10 +46,10 @@ func archWatchpointAddr(s *Session, siAddr uint64) uint64 { } // Best effort: the write-back only fails for a debuggee that died, in // which case no further watchpoint can fire anyway. - _ = ptracePokeUser(s.pid, dr6Off, dr6&^0xF) + _ = ptracePokeUser(s.tid, dr6Off, dr6&^0xF) for slot := range 4 { if status&(1<