// Copyright (c) 2026 Petr BalvĂ­n (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause //go:build linux && amd64 package debug import ( "encoding/binary" "fmt" "syscall" "unsafe" ) // GetRegs reads the general-purpose registers of the stopped debuggee. func (s *Session) GetRegs() (Regs, error) { var regs Regs _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGS), uintptr(s.pid), 0, uintptr(unsafe.Pointer(®s)), 0, 0, ) if errno != 0 { return regs, fmt.Errorf("debug: PTRACE_GETREGS: %w", errno) } return regs, nil } // SetRegs writes the general-purpose registers of the stopped debuggee. func (s *Session) SetRegs(regs *Regs) error { _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_SETREGS), uintptr(s.pid), 0, uintptr(unsafe.Pointer(regs)), 0, 0, ) if errno != 0 { return fmt.Errorf("debug: PTRACE_SETREGS: %w", errno) } return nil } // FPRegs holds the x87 FPU and SSE (XMM) register state from // PTRACE_GETFPREGS. The layout is the kernel's struct user_fpregs_struct // (sys/user.h), the FXSAVE image: 512 bytes with XMM0-15 at offset 160. // The i387 fcs/ds segment fields do not exist in the 64-bit layout. The // size matters: the copy fills all 512 bytes, so a short or misaligned // struct makes PTRACE_GETFPREGS overflow the caller's memory. type FPRegs struct { FCW uint16 FSW uint16 FTW uint16 FOP uint16 FIP uint64 FDP uint64 MXCSR uint32 MXCSRMask uint32 ST [8][16]byte // x87 stack (10 bytes per reg, padded to 16) XMM [16][16]byte // XMM0-15, struct offset 160 Reserved [96]byte // FXSAVE padding, to the full 512 bytes } // GetFPRegs retrieves the FPU/SSE register state of the stopped debuggee. func (s *Session) GetFPRegs() (FPRegs, error) { var fp FPRegs _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETFPREGS), uintptr(s.pid), 0, uintptr(unsafe.Pointer(&fp)), 0, 0, ) if errno != 0 { return fp, fmt.Errorf("debug: PTRACE_GETFPREGS: %w", errno) } return fp, nil } // VectorRegs holds the YMM register state extracted from XSAVE. type VectorRegs struct { YMM [16][32]byte // YMM0-15 (full 256-bit values) } // NT_X86_XSTATE (0x202), the xsave extended-state regset // (include/uapi/linux/elf.h). const ntX86XState = 0x202 // Layout of the buffer PTRACE_GETREGSET returns for NT_X86_XSTATE: the // 512-byte legacy fxsave image (x87 state in 0-159, XMM0-15 in 160-511), // then the 64-byte xsave header whose first 8 bytes are xstate_bv, then one // component per set feature bit, each 64-byte aligned. The YMM high halves // are the first extended component, at offset 576; that offset is fixed by // the ISA on AVX-capable x86-64. XFEATURE_MASK_YMM is bit 2 of xstate_bv // (arch/x86/include/asm/fpu/types.h); the high halves are zero when the bit // is clear. const ( xsaveXMMOffset = 160 xsaveXMMSize = 256 xsaveHeaderOffset = 512 xsaveBVOffset = xsaveHeaderOffset ymmOffset = xsaveHeaderOffset + 64 // 576 ymmSize = 256 // 16 registers, 16 bytes each xfeatureMaskYMM = 1 << 2 xstateMaxBuffer = 4096 // CPUID(0xD).xsave_size is far below this ) // GetVectorRegs retrieves the YMM registers via PTRACE_GETREGSET on // NT_X86_XSTATE. The low (XMM) halves always come from the legacy image; // the high halves are copied only when xstate_bv reports the YMM feature, // and read as zero otherwise. When the regset request fails the FP image // still provides correct XMM halves, so that is the fallback. func (s *Session) GetVectorRegs() (VectorRegs, error) { var v VectorRegs buf := make([]byte, xstateMaxBuffer) iovec := syscall.Iovec{ Base: &buf[0], Len: uint64(len(buf)), } _, _, errno := syscall.Syscall6( syscall.SYS_PTRACE, uintptr(syscall.PTRACE_GETREGSET), uintptr(s.pid), uintptr(ntX86XState), uintptr(unsafe.Pointer(&iovec)), 0, 0, ) if errno != 0 { fp, err := s.GetFPRegs() if err != nil { return v, err } for i := range 16 { copy(v.YMM[i][:16], fp.XMM[i][:]) } return v, nil } n := int(iovec.Len) for i := range 16 { copy(v.YMM[i][:16], buf[xsaveXMMOffset+16*i:xsaveXMMOffset+16*i+16]) } if n >= ymmOffset+ymmSize { if binary.LittleEndian.Uint64(buf[xsaveBVOffset:xsaveBVOffset+8])&xfeatureMaskYMM != 0 { for i := range 16 { copy(v.YMM[i][16:], buf[ymmOffset+16*i:ymmOffset+16*i+16]) } } } return v, nil }