Files
gasm-sdk/asm/elf_dwarf_test.go
T

373 lines
11 KiB
Go

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"bytes"
"encoding/binary"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/parser"
)
// ulebIter reads ULEB128 values, the .debug_abbrev and line-header
// encoding.
type ulebIter struct {
b []byte
i int
}
func (r *ulebIter) uleb(t *testing.T) uint64 {
t.Helper()
v, n := binary.Uvarint(r.b[r.i:])
if n <= 0 {
t.Fatalf("bad ULEB at %d", r.i)
}
r.i += n
return v
}
func (r *ulebIter) byteAt(t *testing.T) byte {
t.Helper()
if r.i >= len(r.b) {
t.Fatalf("read past end at %d", r.i)
}
c := r.b[r.i]
r.i++
return c
}
func (r *ulebIter) uint32At(t *testing.T) uint32 {
t.Helper()
v := binary.LittleEndian.Uint32(r.b[r.i:])
r.i += 4
return v
}
// sleb reads a signed LEB128, the DWARF encoding (sign-extended two's
// complement, not Go's zigzag varint).
func (r *ulebIter) sleb(t *testing.T) int64 {
t.Helper()
var v int64
var shift uint
for {
c := r.byteAt(t)
v |= int64(c&0x7f) << shift
shift += 7
if c&0x80 == 0 {
if c&0x40 != 0 {
v |= -1 << shift
}
return v
}
}
}
// dwarfAttr is one attribute/form pair of an abbreviation.
type dwarfAttr struct{ attr, form uint64 }
// dwarfAbbrev is one parsed abbreviation declaration.
type dwarfAbbrev struct {
code uint64
tag uint64
children bool
attrs []dwarfAttr
}
// parseAbbrevs walks a .debug_abbrev table: abbreviation code, tag,
// children flag, then attr/form ULEB pairs terminated by a double zero.
func parseAbbrevs(t *testing.T, b []byte) map[uint64]dwarfAbbrev {
t.Helper()
out := map[uint64]dwarfAbbrev{}
r := &ulebIter{b: b}
for {
code := r.uleb(t)
if code == 0 {
return out
}
ab := dwarfAbbrev{code: code, tag: r.uleb(t)}
ab.children = r.byteAt(t) == 1
for {
attr := r.uleb(t)
form := r.uleb(t)
if attr == 0 && form == 0 {
break
}
if attr == 0 || form == 0 {
t.Fatalf("abbrev %d: half-terminated attr/form pair (%d, %d)", code, attr, form)
}
ab.attrs = append(ab.attrs, dwarfAttr{attr, form})
}
out[code] = ab
}
}
func eqAttrs(t *testing.T, ab dwarfAbbrev, want []dwarfAttr) {
t.Helper()
if len(ab.attrs) != len(want) {
t.Fatalf("abbrev %d attrs = %v, want %v", ab.code, ab.attrs, want)
}
for i, w := range want {
if ab.attrs[i] != w {
t.Fatalf("abbrev %d attr %d = (%#x, %#x), want (%#x, %#x)", ab.code, i, ab.attrs[i].attr, ab.attrs[i].form, w.attr, w.form)
}
}
}
// TestDwarfAbbrevTable walks the abbreviation table as a consumer does and
// checks the attribute/form sets against the constants the toolchain uses
// (cmd/internal/dwarf/dwarf_defs.go). A wrong constant here renames an
// attribute (0x1b is comp_dir, not low_pc; 0x29 and 0x37 are bounds and
// count) and a wrong form desynchronises the DIE parse: 0x25 is strx1, one
// byte, where the writer emits four for a section offset.
func TestDwarfAbbrevTable(t *testing.T) {
abbrev := dwarfAbbrevTable()
if len(abbrev) == 0 {
t.Fatal("empty abbrev table")
}
// Must end with a zero byte (end of table).
if abbrev[len(abbrev)-1] != 0 {
t.Fatalf("abbrev table last byte = %d, want 0", abbrev[len(abbrev)-1])
}
abs := parseAbbrevs(t, abbrev)
if len(abs) != 2 {
t.Fatalf("abbreviations = %d, want 2", len(abs))
}
cu, ok := abs[1]
if !ok {
t.Fatal("missing abbreviation 1 (compile unit)")
}
if cu.tag != dwTagCompUnit || !cu.children {
t.Errorf("abbrev 1: tag %#x children %v, want compile unit with children", cu.tag, cu.children)
}
eqAttrs(t, cu, []dwarfAttr{
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtStmtList, dwFormSecOff},
{dwAtName, dwFormString},
})
sp, ok := abs[2]
if !ok {
t.Fatal("missing abbreviation 2 (subprogram)")
}
if sp.tag != dwTagSubprog || sp.children {
t.Errorf("abbrev 2: tag %#x children %v, want subprogram without children", sp.tag, sp.children)
}
eqAttrs(t, sp, []dwarfAttr{
{dwAtName, dwFormString},
{dwAtLowPC, dwFormAddr},
{dwAtHighPC, dwFormData8},
{dwAtFrameBase, dwFormExprloc},
{dwAtDeclFile, dwFormData1},
{dwAtDeclLine, dwFormData1},
{dwAtExternal, 0x0c}, // DW_FORM_flag
})
}
// TestDwarfLineHeaderV5 parses the .debug_line header under DWARF5 rules:
// the directory and file tables are format-descriptor lists, not the
// DWARF2-4 shape of null-terminated strings, and the file entry references
// the source name through .debug_line_str.
func TestDwarfLineHeaderV5(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
r := &ulebIter{b: ds.debugLine}
r.uint32At(t) // unit_length
if v := binary.LittleEndian.Uint16(ds.debugLine[4:]); v != 5 {
t.Fatalf("version = %d, want 5", v)
}
r.i = 6
r.byteAt(t) // address_size
r.byteAt(t) // segment_selector_size
r.uint32At(t) // header_length
r.byteAt(t) // minimum_instruction_length
r.byteAt(t) // maximum_ops_per_instruction
r.byteAt(t) // default_is_stmt
r.byteAt(t) // line_base
r.byteAt(t) // line_range
opcodeBase := r.byteAt(t)
for range int(opcodeBase) - 1 {
r.byteAt(t) // standard opcode lengths
}
// Directory table (DWARF5 §6.2.4).
if n := r.byteAt(t); n != 1 {
t.Fatalf("directory_entry_format_count = %d, want 1", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("directory content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("directory form = %#x, want DW_FORM_line_strp", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("directories_count = %d, want 1", n)
}
if off := r.uint32At(t); off != 0 {
t.Errorf("compilation directory line_strp = %d, want 0 (the empty string)", off)
}
// File table (DWARF5 §6.2.5).
if n := r.byteAt(t); n != 2 {
t.Fatalf("file_name_entry_format_count = %d, want 2", n)
}
if lnct := r.uleb(t); lnct != dwLnctPath {
t.Errorf("file content type = %#x, want DW_LNCT_path", lnct)
}
if form := r.uleb(t); form != dwFormLineStrp {
t.Errorf("file path form = %#x, want DW_FORM_line_strp", form)
}
if lnct := r.uleb(t); lnct != dwLnctDirIndex {
t.Errorf("file content type = %#x, want DW_LNCT_directory_index", lnct)
}
if form := r.uleb(t); form != dwFormUdata {
t.Errorf("file dir-index form = %#x, want DW_FORM_udata", form)
}
if n := r.uleb(t); n != 1 {
t.Fatalf("file_names_count = %d, want 1", n)
}
strOff := r.uint32At(t)
if dirIdx := r.uleb(t); dirIdx != 0 {
t.Errorf("file directory index = %d, want 0", dirIdx)
}
// The file entry's line_strp must resolve to the source name.
end := int(strOff) + len("test_amd64.s")
if int(strOff) >= len(ds.debugLineStr) || !bytes.Equal(ds.debugLineStr[strOff:end], []byte("test_amd64.s")) {
t.Errorf("file entry line_strp %d does not name the source: %q", strOff, ds.debugLineStr)
}
// The fixed header fields: address_size 8 and a header_length that
// points just past the file table (the patch site is offset 8 in the
// v5 header, and the field counts from its own end).
if ds.debugLine[6] != 8 || ds.debugLine[7] != 0 {
t.Errorf("address_size/segment_selector = %d/%d, want 8/0", ds.debugLine[6], ds.debugLine[7])
}
if hl := binary.LittleEndian.Uint32(ds.debugLine[8:]); hl != uint32(r.i-12) {
t.Errorf("header_length = %d, want %d (the byte after the file table is %d)", hl, r.i-12, r.i)
}
}
// TestDwarfFrameCIEArch checks the shared CIE carries each architecture's
// stack-pointer and return-address registers: the values the Go linker
// writes (cmd/link/internal/<arch>/l.go dwarfRegSP/dwarfRegLR).
func TestDwarfFrameCIEArch(t *testing.T) {
for _, tc := range []struct {
name string
cfi cfiArch
}{
{"amd64", cfiAMD64},
{"arm64", cfiARM64},
{"riscv64", cfiRISCV64},
{"loong64", cfiLOONG64},
} {
frame := dwarfBuildFrameSection(&Image{}, tc.cfi, &dwarfSections{})
r := &ulebIter{b: frame}
r.uint32At(t) // length
if cid := r.uint32At(t); cid != 0xFFFFFFFF {
t.Errorf("%s: CIE id = %#x, want 0xffffffff", tc.name, cid)
}
if v := r.byteAt(t); v != 3 {
t.Errorf("%s: CIE version = %d, want 3", tc.name, v)
}
if aug := r.byteAt(t); aug != 0 {
t.Errorf("%s: CIE augmentation = %d, want 0", tc.name, aug)
}
if ca := r.uleb(t); ca != 1 {
t.Errorf("%s: code alignment = %d, want 1", tc.name, ca)
}
if da := r.sleb(t); da != -8 {
t.Errorf("%s: data alignment = %d, want -8 (signed LEB128, not zigzag)", tc.name, da)
}
if ra := r.uleb(t); ra != uint64(tc.cfi.raReg) {
t.Errorf("%s: return-address register = %d, want %d", tc.name, ra, tc.cfi.raReg)
}
if op := r.byteAt(t); op != 0x0c {
t.Errorf("%s: expected DW_CFA_def_cfa, got opcode %#x", tc.name, op)
}
if cfa := r.uleb(t); cfa != uint64(tc.cfi.cfaReg) {
t.Errorf("%s: CFA register = %d, want %d", tc.name, cfa, tc.cfi.cfaReg)
}
if off := r.uleb(t); off != 0 {
t.Errorf("%s: CFA offset = %d, want 0", tc.name, off)
}
}
}
func TestEmitDWARF(t *testing.T) {
src := `#include "textflag.h"
TEXT ·add(SB), NOSPLIT, $0-24
MOVQ a+0(FP), AX
MOVQ b+8(FP), BX
ADDQ BX, AX
MOVQ AX, ret+16(FP)
RET
`
f, errs := parser.Parse("test_amd64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFile(f)
if err != nil {
t.Fatalf("assemble: %v", err)
}
ds := emitDWARF(img, "test_amd64.s", cfiAMD64)
// .debug_abbrev must not be empty and must start with abbrev code 1.
if len(ds.debugAbbrev) == 0 {
t.Fatal("empty .debug_abbrev")
}
if ds.debugAbbrev[0] != 1 {
t.Fatalf(".debug_abbrev first byte = %d, want 1", ds.debugAbbrev[0])
}
// .debug_info must have a compile unit header (DWARF5 version 5).
if len(ds.debugInfo) < 12 {
t.Fatalf(".debug_info too short: %d bytes", len(ds.debugInfo))
}
// Version field at offset 4 (after unit_length).
if ds.debugInfo[4] != 5 || ds.debugInfo[5] != 0 {
t.Fatalf(".debug_info version = %d, want 5", uint16(ds.debugInfo[4])|uint16(ds.debugInfo[5])<<8)
}
// .debug_line must have a header.
if len(ds.debugLine) < 20 {
t.Fatalf(".debug_line too short: %d bytes", len(ds.debugLine))
}
// Version at offset 4.
if ds.debugLine[4] != 5 || ds.debugLine[5] != 0 {
t.Fatalf(".debug_line version = %d, want 5", uint16(ds.debugLine[4])|uint16(ds.debugLine[5])<<8)
}
// .debug_line_str must contain the source file name.
if len(ds.debugLineStr) == 0 {
t.Fatal("empty .debug_line_str")
}
// Relocations must reference the function.
if len(ds.lineRelocs) == 0 {
t.Fatal("no .debug_line relocations")
}
if len(ds.infoRelocs) == 0 {
t.Fatal("no .debug_info relocations")
}
}