From 31ee8e7941d388e97c809898d8b54d3d7cf112d8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Thu, 13 Aug 2026 11:24:44 +0200 Subject: [PATCH] feat(asm): add LoongArch encoder with ELF and GOOBJ emission Assisted-by: DeepSeek V4 Pro --- CHANGELOG.md | 18 + README.md | 15 +- asm/elfloong64.go | 223 +++++ asm/elfloong64_test.go | 200 +++++ asm/goobj.go | 211 +++-- asm/goobj_dwarf.go | 188 +++++ asm/goobj_dwarf_test.go | 214 +++++ asm/goobj_test.go | 112 ++- asm/goobjloong64.go | 91 ++ asm/goobjriscv.go | 8 +- asm/l64_goobj_test.go | 326 ++++++++ asm/link.go | 85 +- asm/loong64_assemble.go | 1385 ++++++++++++++++++++++++++++++- asm/loong64_encode.go | 595 +++++++++++++ asm/loong64_encode_test.go | 293 +++++++ asm/loong64_frame.go | 129 +++ asm/loong64_more_test.go | 367 ++++++++ cmd/gasm/main.go | 129 ++- docs/ARCHITECTURE.md | 30 +- parser/parser.go | 6 +- testdata/verify/basic_loong64.s | 54 ++ testdata/verify/fp_loong64.s | 143 ++++ verify/groundtruth.go | 7 + verify/l64_groundtruth_test.go | 97 +++ 24 files changed, 4777 insertions(+), 149 deletions(-) create mode 100644 asm/elfloong64.go create mode 100644 asm/elfloong64_test.go create mode 100644 asm/goobj_dwarf.go create mode 100644 asm/goobj_dwarf_test.go create mode 100644 asm/goobjloong64.go create mode 100644 asm/l64_goobj_test.go create mode 100644 asm/loong64_encode.go create mode 100644 asm/loong64_encode_test.go create mode 100644 asm/loong64_frame.go create mode 100644 asm/loong64_more_test.go create mode 100644 testdata/verify/basic_loong64.s create mode 100644 testdata/verify/fp_loong64.s create mode 100644 verify/l64_groundtruth_test.go diff --git a/CHANGELOG.md b/CHANGELOG.md index 5f0ef85..1a24e6d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,24 @@ and this project adheres to [Conventional Commits](https://www.conventionalcommi Unreleased changes on the `development` branch. +### Added + +- **LoongArch encoder (Phase 5).** `gasm asm` can now assemble `_loong64.s` + files: the full LoongArch64 instruction set with the dual-form arithmetic + mnemonics, the 16/21-bit branch families, the MOV pseudo-instruction and + its immediate-constant expansions, FP/SP frame mapping, SB/global symbol + references (pcalau12i pairs) and ELF64 emission + (`gasm asm --format elf`). Ground-truth verification against + `GOARCH=loong64 go tool asm` matches byte-for-byte; GOOBJ emission + (`gasm asm --format goobj`) is proven end-to-end by linking the object + into a cross-compiled `go build`. +- **GOOBJ DWARF symbols.** The GOOBJ emitters now write the per-function + DWARF symbols the linker requires (the subprogram DIE and the `.debug_line` + program, byte-identical to `cmd/asm`'s), and the pc-value table deltas are + in the architecture's MinLC units as the runtime expects — the amd64 link + test now genuinely substitutes the gasm object, and the amd64/loong64 + end-to-end GOOBJ link tests pass. + ### Fixed - **Debugger watchpoint slots.** `gasm debug`'s `watch` command always used diff --git a/README.md b/README.md index 5f862ad..b424c29 100644 --- a/README.md +++ b/README.md @@ -33,7 +33,9 @@ gasm profile show basic-block structure of functions > profiling) in v0.25.0; Phase 4 (interactive debugger — ptrace-based, > breakpoints, watchpoints, stepping, vector register display, named buffer > allocation) in v0.27.0; RISC-V encoder (RV64IMAFDC + RVC, ELF emission, -> ground-truth, GOOBJ) in v0.28.0–v0.29.0. See [Roadmap](#roadmap). +> ground-truth, GOOBJ) in v0.28.0–v0.29.0; LoongArch encoder (the full +> instruction set with the MOV expansions, ELF and GOOBJ emission, and +> ground-truth verification) after v0.29.0. See [Roadmap](#roadmap). ## Architecture support @@ -269,8 +271,13 @@ portable Go implementation every kernel is derived from. - **RISC-V encoding — done.** RV64IMAFDC instruction set, RVC compression, MOV pseudo-instruction, SB/global symbols (AUIPC pairs), ELF64 and GOOBJ emission, and ground-truth verification against `go tool asm`. -- **Remaining:** arm64 and loong64 encoding, plus the same encode-and-verify - treatment for each (instruction tables already generated from the toolchain). +- **LoongArch encoding — done.** The LoongArch64 instruction set with the + MOV pseudo-instruction and its immediate-constant expansions, FP/SP frame + handling, SB/global symbol references (pcalau12i pairs), ELF64 and GOOBJ + emission, and ground-truth verification against `go tool asm` — the emitted + GOOBJ links into a real `go build` for `GOARCH=loong64`. +- **Remaining:** arm64 encoding, plus the same encode-and-verify treatment + (instruction tables already generated from the toolchain). ## Principles @@ -303,7 +310,7 @@ portable Go implementation every kernel is derived from. | `arch` | amd64, arm64, riscv64 and loong64 register files and instruction tables. | | `lint` | Conservative static checks. | | `format` | A canonical formatter — `gofmt` for assembly. | -| `asm` | The standalone assembler: amd64 and RISC-V encoders, linker, object-file emitters (ELF, GOOBJ). | +| `asm` | The standalone assembler: amd64, RISC-V and LoongArch encoders, linker, object-file emitters (ELF, GOOBJ). | | `verify` | JIT execution substrate for dynamic analysis, combined ABI+fuzz differential testing (Phase 3). | | `debug` | Interactive ptrace debugger with GPR/YMM register display and named buffer allocation (Phase 4). | | `lsp` | Language Server Protocol server. | diff --git a/asm/elfloong64.go b/asm/elfloong64.go new file mode 100644 index 0000000..e321fe9 --- /dev/null +++ b/asm/elfloong64.go @@ -0,0 +1,223 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "encoding/binary" + "fmt" +) + +// LoongArch ELF64 relocatable object emission. + +const ( + emLOONGARCH = 258 // EM_LOONGARCH + + // LoongArch relocation types (the ELF psABI). + rLarchPCALAHI20 = 71 // R_LARCH_PCALA_HI20 (pcalau12i) + rLarchPCALALO12 = 72 // R_LARCH_PCALA_LO12 (addi.d/ld/st) +) + +// ELFLOONG64Object returns the image as an ELF64 relocatable object file for +// LoongArch (EM_LOONGARCH, 64-bit, little-endian). The structure mirrors the +// amd64 and RISC-V ELF emitters: .text, .data, .symtab, .strtab and an +// optional .rela.text. +func (img *Image) ELFLOONG64Object() ([]byte, error) { + le := binary.LittleEndian + + const ( + secText = 1 + secData = 2 + ) + + // Build symbol table. + var locals, globals []elfSym + for _, fn := range img.Funcs { + s := elfSym{ + name: objectName(fn.Pkg, fn.Name), + info: sttFunc, + shndx: secText, + value: uint64(fn.Offset), + size: uint64(fn.Size), + } + if fn.Static { + locals = append(locals, s) + } else { + s.info |= stbGlobal << stInfoShift + globals = append(globals, s) + } + } + for _, d := range img.DataSyms { + s := elfSym{ + name: objectName(d.Pkg, d.Name), + info: sttObject, + shndx: secData, + value: uint64(d.Offset), + size: uint64(d.Size), + } + if d.Static { + locals = append(locals, s) + } else { + s.info |= stbGlobal << stInfoShift + globals = append(globals, s) + } + } + for _, name := range img.Externals { + globals = append(globals, elfSym{name: name, info: stbGlobal << stInfoShift}) + } + syms := []elfSym{ + {}, + {name: ".text", info: sttSection, shndx: secText}, + {name: ".data", info: sttSection, shndx: secData}, + } + syms = append(syms, locals...) + shInfo := len(syms) + syms = append(syms, globals...) + symIdx := map[string]int{} + for i, s := range syms { + symIdx[s.name] = i + } + + // Build relocations. Each SB reference is a pcalau12i pair: + // pcalau12i rd, 0 → R_LARCH_PCALA_HI20 + // addi.d/ld/st → R_LARCH_PCALA_LO12 + type elfRela struct { + off uint64 + typ uint32 + sym int + addend int64 + } + var relas []elfRela + for _, fn := range img.Funcs { + for _, r := range fn.Relocs { + idx, ok := symIdx[r.Name] + if !ok { + return nil, fmt.Errorf("relocation references unknown symbol %q", r.Name) + } + typ := uint32(rLarchPCALAHI20) + if r.Kind == RelLoong64AddrLo { + typ = rLarchPCALALO12 + } + relas = append(relas, elfRela{ + off: uint64(fn.Offset + r.Off), + typ: typ, + sym: idx, + addend: r.Addend - int64(r.After-r.Off), + }) + } + } + + // String tables. + stNames := newElfStrtab() + for _, s := range syms { + stNames.add(s.name) + } + stSections := newElfStrtab() + for _, n := range []string{".text", ".data", ".symtab", ".strtab", ".rela.text", ".shstrtab"} { + stSections.add(n) + } + + hasRela := len(relas) > 0 + nSections := 6 + if hasRela { + nSections = 7 + } + secSymtab, secStrtab := 3, 4 + secShstr := nSections - 1 + + // Layout. + var out []byte + out = append(out, make([]byte, 64)...) + + align := func(n int) { + for len(out)%n != 0 { + out = append(out, 0) + } + } + + align(16) + textOff := len(out) + out = append(out, img.Code...) + + align(16) + dataOff := len(out) + out = append(out, img.Data...) + + align(8) + symtabOff := len(out) + for _, s := range syms { + var b [24]byte + le.PutUint32(b[0:], uint32(stNames.at(s.name))) + b[4] = s.info + b[5] = 0 + le.PutUint16(b[6:], s.shndx) + le.PutUint64(b[8:], s.value) + le.PutUint64(b[16:], s.size) + out = append(out, b[:]...) + } + + strtabOff := len(out) + out = append(out, stNames.bytes()...) + + var relaOff int + if hasRela { + align(8) + relaOff = len(out) + for _, r := range relas { + var b [24]byte + le.PutUint64(b[0:], r.off) + le.PutUint64(b[8:], uint64(r.sym)<<32|uint64(r.typ)) + le.PutUint64(b[16:], uint64(r.addend)) + out = append(out, b[:]...) + } + } + + shstrOff := len(out) + out = append(out, stSections.bytes()...) + + align(8) + shoff := len(out) + + putSh := func(name string, typ int, flags uint64, off, size int, link, info int, alignV, entsize uint64) { + var b [64]byte + le.PutUint32(b[0:], uint32(stSections.at(name))) + le.PutUint32(b[4:], uint32(typ)) + le.PutUint64(b[8:], flags) + le.PutUint64(b[16:], 0) + le.PutUint64(b[24:], uint64(off)) + le.PutUint64(b[32:], uint64(size)) + le.PutUint32(b[40:], uint32(link)) + le.PutUint32(b[44:], uint32(info)) + le.PutUint64(b[48:], alignV) + le.PutUint64(b[56:], entsize) + out = append(out, b[:]...) + } + putSh("", shtNull, 0, 0, 0, 0, 0, 0, 0) + putSh(".text", shtProgbits, shfAlloc|shfExecInstr, textOff, len(img.Code), 0, 0, 16, 0) + putSh(".data", shtProgbits, shfAlloc|shfWrite, dataOff, len(img.Data), 0, 0, 16, 0) + putSh(".symtab", shtSymtab, 0, symtabOff, 24*len(syms), secStrtab, shInfo, 8, 24) + putSh(".strtab", shtStrtab, 0, strtabOff, len(stNames.bytes()), 0, 0, 1, 0) + if hasRela { + putSh(".rela.text", shtRela, 0, relaOff, 24*len(relas), secSymtab, secText, 8, 24) + } + putSh(".shstrtab", shtStrtab, 0, shstrOff, len(stSections.bytes()), 0, 0, 1, 0) + + // ELF header. + hdr := out[:64] + copy(hdr[0:], []byte{0x7f, 'E', 'L', 'F', elfClass64, elfDataLSB, elfVersion, 0}) + le.PutUint16(hdr[16:], etREL) + le.PutUint16(hdr[18:], emLOONGARCH) + le.PutUint32(hdr[20:], elfVersion) + le.PutUint64(hdr[24:], 0) + le.PutUint64(hdr[32:], 0) + le.PutUint64(hdr[40:], uint64(shoff)) + le.PutUint32(hdr[48:], 0) + le.PutUint16(hdr[52:], 64) + le.PutUint16(hdr[54:], 0) + le.PutUint16(hdr[56:], 0) + le.PutUint16(hdr[58:], 64) + le.PutUint16(hdr[60:], uint16(nSections)) + le.PutUint16(hdr[62:], uint16(secShstr)) + + return out, nil +} diff --git a/asm/elfloong64_test.go b/asm/elfloong64_test.go new file mode 100644 index 0000000..2a2bd0f --- /dev/null +++ b/asm/elfloong64_test.go @@ -0,0 +1,200 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "debug/elf" + "encoding/binary" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestELFLOONG64Object checks the structure of the emitted LoongArch ELF64 +// relocatable object: sections, the symbol table (bindings, types, values, +// sizes) and the .rela.text relocation pair for the static-symbol load, +// parsed back with debug/elf. +func TestELFLOONG64Object(t *testing.T) { + f, errs := parser.Parse("k_loong64.s", ` +#include "textflag.h" + +TEXT ·add(SB), NOSPLIT, $0-24 + MOVV a+0(FP), R4 + MOVV b+8(FP), R5 + ADDV R5, R4, R4 + MOVV R4, ret+16(FP) + RET + +TEXT ·getanswer(SB), NOSPLIT, $0-8 + MOVV answer<>(SB), R4 + MOVV R4, ret+0(FP) + RET + +GLOBL answer<>(SB), RODATA, $8 +DATA answer<>+0(SB)/8, $42 +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + obj, err := img.ELFLOONG64Object() + if err != nil { + t.Fatalf("ELFLOONG64Object: %v", err) + } + ef, err := elf.NewFile(bytes.NewReader(obj)) + if err != nil { + t.Fatalf("parse emitted object: %v", err) + } + defer ef.Close() + + if ef.Type != elf.ET_REL || ef.Machine != elf.EM_LOONGARCH { + t.Errorf("type/machine = %v/%v, want ET_REL/EM_LOONGARCH", ef.Type, ef.Machine) + } + + text := ef.Section(".text") + data := ef.Section(".data") + if text == nil || data == nil { + t.Fatal("missing .text or .data section") + } + if text.Flags&elf.SHF_EXECINSTR == 0 || text.Flags&elf.SHF_ALLOC == 0 { + t.Errorf(".text flags = %v", text.Flags) + } + if data.Flags&elf.SHF_WRITE == 0 { + t.Errorf(".data flags = %v", data.Flags) + } + textData, err := text.Data() + if err != nil { + t.Fatal(err) + } + if !bytes.Equal(textData, img.Code) { + t.Errorf(".text contents differ from the image code") + } + dataData, err := data.Data() + if err != nil { + t.Fatal(err) + } + + syms, err := ef.Symbols() + if err != nil { + t.Fatalf("symbols: %v", err) + } + byName := map[string]elf.Symbol{} + for _, s := range syms { + byName[s.Name] = s + } + wantSym := func(name string, bind elf.SymBind, typ elf.SymType, section elf.SectionIndex, size uint64) { + t.Helper() + s, ok := byName[name] + if !ok { + t.Errorf("symbol %q not found", name) + return + } + if elf.ST_BIND(s.Info) != bind || elf.ST_TYPE(s.Info) != typ { + t.Errorf("%s: bind/type = %v/%v, want %v/%v", name, elf.ST_BIND(s.Info), elf.ST_TYPE(s.Info), bind, typ) + } + if s.Section != section { + t.Errorf("%s: section = %v, want %v", name, s.Section, section) + } + if s.Size != size { + t.Errorf("%s: size = %d, want %d", name, s.Size, size) + } + } + if ef.Sections[1].Name != ".text" || ef.Sections[2].Name != ".data" { + t.Fatalf("section layout = %s, %s; want .text, .data", ef.Sections[1].Name, ef.Sections[2].Name) + } + textIdx := elf.SectionIndex(1) + dataIdx := elf.SectionIndex(2) + wantSym("add", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 20) + wantSym("getanswer", elf.STB_GLOBAL, elf.STT_FUNC, textIdx, 16) + wantSym("answer", elf.STB_LOCAL, elf.STT_OBJECT, dataIdx, 8) + + // The data section carries 16-byte alignment padding; the answer + // symbol sits at its padded offset. + ans := byName["answer"] + if ans.Value+8 > uint64(len(dataData)) { + t.Fatalf("answer value %d outside .data (%d bytes)", ans.Value, len(dataData)) + } + if got := dataData[ans.Value : ans.Value+8]; !bytes.Equal(got, []byte{42, 0, 0, 0, 0, 0, 0, 0}) { + t.Errorf("answer data = % x, want $42", got) + } + + // Relocations: the static-symbol load is a pcalau12i+ld.d pair, so one + // R_LARCH_PCALA_HI20 and one R_LARCH_PCALA_LO12, both against the local + // data symbol. debug/elf does not surface rela entries, so read the + // section directly. + relaSec := ef.Section(".rela.text") + if relaSec == nil { + t.Fatal("missing .rela.text") + } + raw, err := relaSec.Data() + if err != nil { + t.Fatal(err) + } + if len(raw)%24 != 0 || len(raw)/24 != 2 { + t.Fatalf(".rela.text has %d bytes, want two 24-byte entries", len(raw)) + } + le := binary.LittleEndian + for i := 0; i < 2; i++ { + e := raw[i*24 : (i+1)*24] + off := le.Uint64(e[0:]) + info := le.Uint64(e[8:]) + typ := info & 0xffffffff + sym := int(info >> 32) + if i == 0 && (typ != uint64(elf.R_LARCH_PCALA_HI20) || off != 20) { + t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_HI20 at 20", i, typ, off) + } + if i == 1 && (typ != uint64(elf.R_LARCH_PCALA_LO12) || off != 24) { + t.Errorf("reloc %d: type %d off %d, want R_LARCH_PCALA_LO12 at 24", i, typ, off) + } + if sym != 3 { // NULL, .text, .data, then the first local: answer + t.Errorf("reloc %d: symbol index %d, want 3 (answer)", i, sym) + } + } +} + +// TestELFLOONG64ObjectNoRelocations checks a file with no static-symbol +// references emits a valid object without a .rela.text section. +func TestELFLOONG64ObjectNoRelocations(t *testing.T) { + f, errs := parser.Parse("n_loong64.s", ` +#include "textflag.h" +TEXT ·nop(SB), NOSPLIT, $0 + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + obj, err := img.ELFLOONG64Object() + if err != nil { + t.Fatalf("ELFLOONG64Object: %v", err) + } + ef, err := elf.NewFile(bytes.NewReader(obj)) + if err != nil { + t.Fatalf("parse emitted object: %v", err) + } + defer ef.Close() + if ef.Section(".rela.text") != nil { + t.Error("unexpected .rela.text section") + } + syms, err := ef.Symbols() + if err != nil { + t.Fatal(err) + } + found := false + for _, s := range syms { + if s.Name == "nop" && elf.ST_TYPE(s.Info) == elf.STT_FUNC { + found = true + } + } + if !found { + t.Error("function symbol nop not found") + } +} diff --git a/asm/goobj.go b/asm/goobj.go index c387e66..a5c37fe 100644 --- a/asm/goobj.go +++ b/asm/goobj.go @@ -22,9 +22,15 @@ import ( // // The object carries what the linker requires of an assembly object: the // functions (non-package symbols, as cmd/asm emits them), the GLOBL data, -// one FuncInfo per function, and the pc-value tables (pcsp, pcfile, -// pcline, pcinline). DWARF and the implicit funcdata symbols are omitted; -// the linker fills their defaults. +// one FuncInfo per function, the per-function DWARF symbols (the +// .debug_line program and the subprogram DIE, which the linker's DWARF +// pass reads verbatim), and the pc-value tables (pcsp, pcfile, pcline, +// pcinline). The implicit funcdata symbols are omitted; the linker fills +// their defaults. +// +// emitGOObject is architecture-agnostic; the per-architecture GOObject* +// methods supply the toolchain preamble, the MinLC (pc-value delta unit) +// and the relocation-type mapping for code relocations. // GOOBJ block indices (cmd/internal/goobj). const ( @@ -51,9 +57,11 @@ const ( // Symbol kinds used by assembly objects (cmd/internal/objabi). const ( - kindSTEXT = 1 - kindSRODATA = 3 - kindSDATA = 7 + kindSTEXT = 1 + kindSRODATA = 3 + kindSDATA = 7 + kindSDWARFFCN = 14 + kindSDWARFLINES = 20 ) // Symbol flags (cmd/internal/goobj). @@ -66,11 +74,13 @@ const ( // Aux entry types (cmd/internal/goobj). const ( - auxFuncInfo = 1 - auxPcsp = 7 - auxPcfile = 8 - auxPcline = 9 - auxPcinline = 10 + auxFuncInfo = 1 + auxDwarfInfo = 3 + auxDwarfLines = 6 + auxPcsp = 7 + auxPcfile = 8 + auxPcline = 9 + auxPcinline = 10 ) // FuncInfo flags (internal/abi). @@ -80,7 +90,11 @@ const ( ) // Relocation types (cmd/internal/objabi). -const relocPCRel = 14 +const ( + relocPCRel = 14 // R_PCREL + relocAddr = 1 // R_ADDR + relocDWTXTADDRU4 = 106 // R_DWTXTADDR_U4 +) // Special package indices for symbol references. const ( @@ -110,6 +124,13 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte { return binary.LittleEndian.AppendUint32(b, s.align) } +// dwarfRelocSet attaches emitter-generated relocations (the DWARF +// lines/info symbols' address references) to a definition index. +type dwarfRelocSet struct { + si int + relocs []goobjReloc +} + // GOObject returns the image as a GOOBJ object file for the given package // path (the linker qualifies the exported symbols with it, the way cmd/asm // does with its -p flag). srcPath names the source file recorded in the @@ -117,51 +138,26 @@ func (s goSym) append(b []byte, strOff map[string]uint32) []byte { // captured from the installed go tool asm, so the output links with the // toolchain it was produced on — exactly like a real assembly object. func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { - if pkgPath == "" { - return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)") - } pre, err := toolchainObjectPreamble() if err != nil { return nil, err } + // amd64: MinLC 1, R_PCREL for the code relocations. + return img.emitGOObject(pkgPath, srcPath, pre, 1, func(Reloc) uint16 { return relocPCRel }) +} - // The symbol tables. Package definitions: the GLOBL symbols, then one - // anonymous FuncInfo symbol per function. Non-package definitions: the - // pc-value tables and the functions themselves, as cmd/asm lays them - // out. defIdx maps a GLOBL's bare name to its definition index for the - // relocations; fnNpIdx maps a function to its non-package index. - var defs []goSym - var defData [][]byte - defIdx := map[string]int{} - for _, d := range img.DataSyms { - name := d.Name - if !d.Static { - name = pkgPath + "." + name - } - typ := uint8(kindSDATA) - if d.Rodata { - typ = kindSRODATA - } - flag := uint8(0) - if d.Dupok { - flag = symFlagDupok - } - abi := uint16(0) - if d.Static { - abi = symABIStatic - } - defIdx[d.Name] = len(defs) - defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)}) - defData = append(defData, img.Data[d.Offset:d.Offset+d.Size]) - } - fnFiIdx := make([]int, len(img.Funcs)) - for i := range img.Funcs { - data := marshalFuncInfo(img.Funcs[i]) - fnFiIdx[i] = len(defs) - defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))}) - defData = append(defData, data) +// emitGOObject assembles the GOOBJ payload for any architecture. pre is +// the toolchain's object preamble; minLC is the architecture's minimum +// instruction length, the unit of the pc-value table deltas; relocType +// maps a code relocation to its objabi relocation type. +func (img *Image) emitGOObject(pkgPath, srcPath string, pre []byte, minLC int, relocType func(Reloc) uint16) ([]byte, error) { + if pkgPath == "" { + return nil, fmt.Errorf("GOOBJ emission requires a package path (-p)") } + // The non-package definitions first — the DWARF symbols reference the + // functions by these indices: per function the four pc-value tables + // and the function itself, as cmd/asm lays them out. type npSym struct { sym goSym data []byte @@ -175,10 +171,10 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { data []byte dst *int }{ - {pcspTable(fn), &pcIdx[i].sp}, - {pcValueFlat(0, fn.Size), &pcIdx[i].file}, - {pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line}, - {pcValueFlat(-1, fn.Size), &pcIdx[i].inl}, + {pcspTable(fn, minLC), &pcIdx[i].sp}, + {pcValueFlat(0, fn.Size, minLC), &pcIdx[i].file}, + {pcValueFlat(int32(fn.Line), fn.Size, minLC), &pcIdx[i].line}, + {pcValueFlat(-1, fn.Size, minLC), &pcIdx[i].inl}, } for _, t := range tables { *t.dst = len(nps) @@ -213,6 +209,66 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { }) } + // The package definitions: the GLOBL symbols, then, per function, the + // FuncInfo and the two DWARF symbols (the .debug_line program and the + // subprogram DIE). defIdx maps a GLOBL's bare name to its definition + // index for the code relocations. + var defs []goSym + var defData [][]byte + defIdx := map[string]int{} + for _, d := range img.DataSyms { + name := d.Name + if !d.Static { + name = pkgPath + "." + name + } + typ := uint8(kindSDATA) + if d.Rodata { + typ = kindSRODATA + } + flag := uint8(0) + if d.Dupok { + flag = symFlagDupok + } + abi := uint16(0) + if d.Static { + abi = symABIStatic + } + defIdx[d.Name] = len(defs) + defs = append(defs, goSym{name: name, abi: abi, typ: typ, flag: flag, flag2: symFlag2Link, size: uint32(d.Size)}) + defData = append(defData, img.Data[d.Offset:d.Offset+d.Size]) + } + fnFiIdx := make([]int, len(img.Funcs)) + fnLinesIdx := make([]int, len(img.Funcs)) + fnDIEIdx := make([]int, len(img.Funcs)) + var dwarfRelocs []dwarfRelocSet + for i, fn := range img.Funcs { + data := marshalFuncInfo(fn) + fnFiIdx[i] = len(defs) + defs = append(defs, goSym{typ: kindSDATA, size: uint32(len(data))}) + defData = append(defData, data) + + name := fn.Name + if !fn.Static { + name = pkgPath + "." + name + } + + // The DWARF symbols: the .debug_line state-machine program and the + // subprogram DIE, both referencing the function by its non-package + // index (package definitions, like cmd/asm's). + lines, lrel := goobjDwarfLines(fn, fnNpIdx[i]) + fnLinesIdx[i] = len(defs) + defs = append(defs, goSym{typ: kindSDWARFLINES, size: uint32(len(lines))}) + defData = append(defData, lines) + die, drel := goobjDwarfInfo(fn, name, fnNpIdx[i]) + fnDIEIdx[i] = len(defs) + defs = append(defs, goSym{typ: kindSDWARFFCN, size: uint32(len(die))}) + defData = append(defData, die) + dwarfRelocs = append(dwarfRelocs, + dwarfRelocSet{si: fnLinesIdx[i], relocs: lrel}, + dwarfRelocSet{si: fnDIEIdx[i], relocs: drel}, + ) + } + // Resolve external symbol references (cross-package). Build the // package index table and determine each external symbol's SymIdx // by reading the target package's export data. @@ -251,7 +307,7 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { var rec [23]byte binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off))) rec[4] = 4 // field width - binary.LittleEndian.PutUint16(rec[5:], relocPCRel) + binary.LittleEndian.PutUint16(rec[5:], relocType(r)) binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend)) binary.LittleEndian.PutUint32(rec[15:], uint32(pIdx)) binary.LittleEndian.PutUint32(rec[19:], uint32(sIdx)) @@ -265,16 +321,29 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { var rec [23]byte binary.LittleEndian.PutUint32(rec[0:], uint32(int32(r.Off))) rec[4] = 4 // field width - binary.LittleEndian.PutUint16(rec[5:], relocPCRel) + binary.LittleEndian.PutUint16(rec[5:], relocType(r)) binary.LittleEndian.PutUint64(rec[7:], uint64(r.Addend)) binary.LittleEndian.PutUint32(rec[15:], pkgIdxSelf) binary.LittleEndian.PutUint32(rec[19:], uint32(di)) symRelocs[si] = append(symRelocs[si], rec[:]...) } } + // The DWARF symbols' own relocations (the function address references). + for _, ds := range dwarfRelocs { + for _, r := range ds.relocs { + var rec [23]byte + binary.LittleEndian.PutUint32(rec[0:], uint32(r.off)) + rec[4] = r.siz + binary.LittleEndian.PutUint16(rec[5:], r.typ) + binary.LittleEndian.PutUint64(rec[7:], uint64(r.add)) + binary.LittleEndian.PutUint32(rec[15:], r.pkg) + binary.LittleEndian.PutUint32(rec[19:], r.sym) + symRelocs[ds.si] = append(symRelocs[ds.si], rec[:]...) + } + } - // Aux entries per function: FuncInfo, then the four pc tables. - // References into the non-package table use pkgIdxNone. + // Aux entries per function: FuncInfo, the DWARF symbols, then the four + // pc tables. References into the non-package table use pkgIdxNone. symAux := make([][]byte, nsyms) for i := range img.Funcs { si := len(defs) + fnNpIdx[i] @@ -286,10 +355,14 @@ func (img *Image) GOObject(pkgPath, srcPath string) ([]byte, error) { symAux[si] = append(symAux[si], rec[:]...) } aux(auxFuncInfo, pkgIdxSelf, uint32(fnFiIdx[i])) - aux(auxPcsp, pkgIdxNone, uint32(len(defs)+pcIdx[i].sp)) - aux(auxPcfile, pkgIdxNone, uint32(len(defs)+pcIdx[i].file)) - aux(auxPcline, pkgIdxNone, uint32(len(defs)+pcIdx[i].line)) - aux(auxPcinline, pkgIdxNone, uint32(len(defs)+pcIdx[i].inl)) + aux(auxDwarfInfo, pkgIdxSelf, uint32(fnDIEIdx[i])) + aux(auxDwarfLines, pkgIdxSelf, uint32(fnLinesIdx[i])) + // The pc-table references are 0-based within the non-package + // definitions; the loader adds the package-definition count itself. + aux(auxPcsp, pkgIdxNone, uint32(pcIdx[i].sp)) + aux(auxPcfile, pkgIdxNone, uint32(pcIdx[i].file)) + aux(auxPcline, pkgIdxNone, uint32(pcIdx[i].line)) + aux(auxPcinline, pkgIdxNone, uint32(pcIdx[i].inl)) } // The string table. Absolute offsets: it starts right after the @@ -417,19 +490,21 @@ func marshalFuncInfo(fn FuncLayout) []byte { } // pcValueFlat encodes a pc-value table holding v over the whole function. -func pcValueFlat(v int32, size int) []byte { +// The pc deltas are in MinLC units (the runtime scales them by the +// architecture's minimum instruction length). +func pcValueFlat(v int32, size, minLC int) []byte { // The table is delta-encoded from an implicit value of -1: a varint // value delta, an unsigned pc delta to the end, and a zero terminator. out := binary.AppendVarint(nil, int64(v)+1) - out = binary.AppendUvarint(out, uint64(size)) + out = binary.AppendUvarint(out, uint64(size/minLC)) return append(out, 0) } // pcspTable encodes the stack-adjustment table: the SP delta in effect at // every pc, from the function's prologue and epilogue boundaries. -func pcspTable(fn FuncLayout) []byte { +func pcspTable(fn FuncLayout, minLC int) []byte { if len(fn.Spadj) == 0 { - return pcValueFlat(0, fn.Size) + return pcValueFlat(0, fn.Size, minLC) } pts := make([]SpadjStep, 0, len(fn.Spadj)+1) pts = append(pts, SpadjStep{PC: 0, Value: 0}) @@ -437,11 +512,11 @@ func pcspTable(fn FuncLayout) []byte { out := binary.AppendVarint(nil, int64(pts[0].Value)+1) cur, old := pts[0].PC, pts[0].Value for _, p := range pts[1:] { - out = binary.AppendUvarint(out, uint64(p.PC-cur)) + out = binary.AppendUvarint(out, uint64((p.PC-cur)/minLC)) out = binary.AppendVarint(out, int64(p.Value-old)) cur, old = p.PC, p.Value } - out = binary.AppendUvarint(out, uint64(fn.Size-cur)) + out = binary.AppendUvarint(out, uint64((fn.Size-cur)/minLC)) return append(out, 0) } diff --git a/asm/goobj_dwarf.go b/asm/goobj_dwarf.go new file mode 100644 index 0000000..6d79ffa --- /dev/null +++ b/asm/goobj_dwarf.go @@ -0,0 +1,188 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "encoding/binary" +) + +// This file generates the per-function DWARF symbols the linker's DWARF +// pass requires of an assembly object, byte-identical to what cmd/asm +// emits: the .debug_line state-machine program (SDWARFLINES) and the +// subprogram DIE (SDWARFFCN). The linker copies the DIE and line-program +// bytes verbatim into .debug_info and .debug_line, fixing up their +// relocations, so the formats here must match cmd/internal/dwarf's +// DW_ABRV_FUNCTION and generateDebugLinesSymbol exactly. +// +// DWARF5 is assumed throughout (the toolchain's default on Linux and the +// other non-Darwin targets gasm supports). + +// Line-program parameters (cmd/internal/obj/dwarf.go). +const ( + dwLineBase = -4 + dwLineRange = 10 + dwOpcodeBase = 11 + dwPCRange = (255 - dwOpcodeBase) / dwLineRange +) + +// goobjReloc is one relocation attached to an emitter-generated symbol +// (the DWARF lines/info symbols), in goobj's on-disk encoding fields. +type goobjReloc struct { + off int32 + siz uint8 + typ uint16 + add int64 + pkg uint32 + sym uint32 +} + +// goobjDwarfLines builds the function's .debug_line state-machine program: +// an LNE_set_address extended opcode establishing the function's start +// address (carrying the R_ADDR relocation), one row per source line +// change across the function's instructions, an advance to the end of the +// function and an end-of-sequence opcode. The linker appends these bytes +// after the unit's line header, so they must start with the address and +// leave the state machine terminated. +func goobjDwarfLines(fn FuncLayout, fnNpIdx int) ([]byte, []goobjReloc) { + // Rows: the prologue, if any, then the body instructions (fn.Lines + // covers the body only). The first body offset > 0 means a prologue + // precedes it; the toolchain reports the prologue on the TEXT line. + pts := make([]LineEntry, 0, len(fn.Lines)+1) + if len(fn.Lines) == 0 || fn.Lines[0].Offset > 0 { + pts = append(pts, LineEntry{Offset: 0, Line: fn.Line}) + } + pts = append(pts, fn.Lines...) + + out := []byte{0, 9, 2, 0, 0, 0, 0, 0, 0, 0, 0} // LNE_set_address, address zeroed + relocs := []goobjReloc{{ + off: 3, siz: 8, typ: relocAddr, + pkg: pkgIdxNone, sym: uint32(fnNpIdx), + }} + + // The state machine starts at line 1, pc 0 (function-relative); the + // implicit initial pc is the function entry, so the first pc delta is + // against 0. + line := int64(1) + pc := uint64(0) + for _, p := range pts { + if p.Line == 0 || uint64(p.Offset) < pc { + continue + } + // Rows mark source-line changes only; the pc delta is measured from + // the previous row, not the previous instruction. + if int64(p.Line) == line { + continue + } + deltaPC := uint64(p.Offset) - pc + deltaLC := int64(p.Line) - line + out = dwPutPCLCDelta(out, deltaPC, deltaLC) + line, pc = int64(p.Line), uint64(p.Offset) + } + + // Cover the rest of the function and close the sequence. + if end := uint64(fn.Size) - pc; end > 0 { + out = append(out, 2) // DW_LNS_advance_pc + out = binary.AppendUvarint(out, end) + } + out = append(out, 0, 1, 1) // LNE_end_sequence + return out, relocs +} + +// dwPutPCLCDelta encodes one (pcDelta, lineDelta) step as the shortest +// special opcode plus any standard-opcode remainder, exactly like +// cmd/internal/obj's putpclcdelta. +func dwPutPCLCDelta(b []byte, deltaPC uint64, deltaLC int64) []byte { + opcode := dwSelectOpcode(deltaPC, deltaLC) + deltaPC -= uint64((opcode - dwOpcodeBase) / dwLineRange) + deltaLC -= (opcode-dwOpcodeBase)%dwLineRange + dwLineBase + + // The remainder: standard opcodes first, then the special opcode + // (which emits the row). + if deltaPC != 0 { + switch { + case deltaPC <= uint64(dwPCRange): + opcode -= dwLineRange * int64(uint64(dwPCRange)-deltaPC) + b = append(b, 8) // DW_LNS_const_add_pc + case (1<<14) <= deltaPC && deltaPC < (1<<16): + b = append(b, 9) // DW_LNS_fixed_advance_pc + b = binary.LittleEndian.AppendUint16(b, uint16(deltaPC)) + default: + b = append(b, 2) // DW_LNS_advance_pc + b = binary.AppendUvarint(b, deltaPC) + } + } + if deltaLC != 0 { + b = append(b, 3) // DW_LNS_advance_line + b = binary.AppendVarint(b, deltaLC) + } + return append(b, byte(opcode)) +} + +// dwSelectOpcode picks the special opcode for (deltaPC, deltaLC) per +// cmd/internal/obj's putpclcdelta selection logic. +func dwSelectOpcode(deltaPC uint64, deltaLC int64) int64 { + switch { + case deltaLC < dwLineBase: + if deltaPC >= uint64(dwPCRange) { + return dwOpcodeBase + dwLineRange*dwPCRange + } + return dwOpcodeBase + dwLineRange*int64(deltaPC) + case deltaLC < dwLineBase+dwLineRange: + if deltaPC >= uint64(dwPCRange) { + op := int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*dwPCRange + if op > 255 { + op -= dwLineRange + } + return op + } + return int64(dwOpcodeBase) + (deltaLC - dwLineBase) + dwLineRange*int64(deltaPC) + default: + if deltaPC <= uint64(dwPCRange) { + op := int64(dwOpcodeBase) + (dwLineRange - 1) + dwLineRange*int64(deltaPC) + if op > 255 { + op = 255 + } + return op + } + switch deltaPC - uint64(dwPCRange) { + case uint64(dwPCRange), (1 << 7) - 1, (1 << 16) - 1, (1 << 21) - 1, + (1 << 28) - 1, (1 << 35) - 1, (1 << 42) - 1, (1 << 49) - 1, + (1 << 56) - 1, (1 << 63) - 1: + return 255 + default: + // 250: the toolchain's "249" comment is stale. + return dwOpcodeBase + dwLineRange*dwPCRange - 1 + } + } +} + +// goobjDwarfInfo builds the function's DWARF5 subprogram DIE (abbrev +// DW_ABRV_FUNCTION): name, low_pc as a .debug_addr index (the +// R_DWTXTADDR_U4 relocation), high_pc as the size, the call-frame-CFA +// frame base, the decl file/line and the external flag. name is the +// symbol's object name (package-qualified unless static). +func goobjDwarfInfo(fn FuncLayout, name string, fnNpIdx int) ([]byte, []goobjReloc) { + out := []byte{3} // DW_ABRV_FUNCTION + out = append(out, name...) + out = append(out, 0) + + addrx := len(out) + out = append(out, 0, 0, 0, 0) // DW_AT_low_pc: addrx slot, zeroed + out = binary.AppendUvarint(out, uint64(fn.Size)) + out = append(out, 1, 0x9c) // DW_AT_frame_base: block1, DW_OP_call_frame_cfa + out = binary.LittleEndian.AppendUint32(out, 1) + out = binary.AppendUvarint(out, uint64(fn.Line)) + if fn.Static { + out = append(out, 0) + } else { + out = append(out, 1) // DW_AT_external + } + out = append(out, 0) // end of children + + relocs := []goobjReloc{{ + off: int32(addrx), siz: 4, typ: relocDWTXTADDRU4, + pkg: pkgIdxNone, sym: uint32(fnNpIdx), + }} + return out, relocs +} diff --git a/asm/goobj_dwarf_test.go b/asm/goobj_dwarf_test.go new file mode 100644 index 0000000..815297e --- /dev/null +++ b/asm/goobj_dwarf_test.go @@ -0,0 +1,214 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "encoding/binary" + "testing" +) + +// TestDWSelectOpcode checks the special-opcode selection against +// hand-computed values for the boundary cases: line deltas below, inside +// and above the line range, and pc deltas at and beyond PC_RANGE (24). +func TestDWSelectOpcode(t *testing.T) { + cases := []struct { + deltaPC uint64 + deltaLC int64 + want int64 + }{ + {0, 2, 17}, // the common single-instruction step + {0, -4, 11}, // deltaLC == LINE_BASE + {0, -5, 11}, // deltaLC below LINE_BASE: opcode adds nothing + {0, 6, 20}, // deltaLC == LINE_BASE+LINE_RANGE, remainder via advance_line + {4, 1, 56}, // the 4-byte loong64 instruction step + {23, 1, 246}, // deltaPC == PC_RANGE-1 + {24, 1, 246}, // deltaPC == PC_RANGE: wraps past 255 + {25, 1, 246}, // deltaPC past PC_RANGE (the const_add_pc remainder adjusts it later) + {100, 1, 246}, + {151, 10, 255}, // deltaPC-PC_RANGE == (1<<7)-1, large line delta + {100, 10, 250}, // deltaPC-PC_RANGE not on a switch boundary + {23, 10, 250}, // large line delta inside PC_RANGE + } + for _, c := range cases { + if got := dwSelectOpcode(c.deltaPC, c.deltaLC); got != c.want { + t.Errorf("dwSelectOpcode(%d, %d) = %d, want %d", c.deltaPC, c.deltaLC, got, c.want) + } + } +} + +// decodeDWLineProgram decodes a .debug_line state-machine program (as +// emitted by goobjDwarfLines) into (pc, line) rows. +func decodeDWLineProgram(t *testing.T, b []byte) (pcs []uint64, lines []int64) { + t.Helper() + pc, line := uint64(0), int64(1) + emit := func() { + if len(pcs) == 0 || pcs[len(pcs)-1] != pc || lines[len(lines)-1] != line { + pcs = append(pcs, pc) + lines = append(lines, line) + } + } + advancePC := func(delta uint64) { pc += delta } + advanceLine := func(delta int64) { line += delta } + for i := 0; i < len(b); { + op := b[i] + i++ + switch { + case op == 0: // extended opcode + ln, n := binary.Uvarint(b[i:]) + i += n + sub := b[i] + i++ + _ = ln + switch sub { + case 2: // DW_LNE_set_address: 8-byte address + pc = binary.LittleEndian.Uint64(b[i:]) + i += 8 + case 1: // DW_LNE_end_sequence + // terminates the sequence; no new row + } + case op == 2: // DW_LNS_advance_pc + v, n := binary.Uvarint(b[i:]) + i += n + advancePC(v) + case op == 3: // DW_LNS_advance_line + v, n := binary.Varint(b[i:]) + i += n + advanceLine(v) + case op == 8: // DW_LNS_const_add_pc + advancePC(uint64(dwPCRange)) + case op == 9: // DW_LNS_fixed_advance_pc + advancePC(uint64(binary.LittleEndian.Uint16(b[i:]))) + i += 2 + case op >= dwOpcodeBase: // special opcode + advancePC(uint64((int64(op) - dwOpcodeBase) / dwLineRange)) + advanceLine((int64(op)-dwOpcodeBase)%dwLineRange + dwLineBase) + emit() + } + } + return pcs, lines +} + +// TestGoobjDwarfLinesRows checks the emitted line program's rows for +// synthetic functions: a zero-frame function with one instruction per +// line, a framed function (the prologue row is prepended on the TEXT +// line), instructions sharing a line, and a function with a large pc gap +// (the const_add_pc remainder path). +func TestGoobjDwarfLinesRows(t *testing.T) { + cases := []struct { + name string + fn FuncLayout + want [][2]int64 // (pc, line) + }{ + { + "one instruction per line", + FuncLayout{Size: 20, Line: 2, Lines: []LineEntry{ + {0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}, + }}, + [][2]int64{{0, 3}, {4, 4}, {8, 5}, {12, 6}, {16, 7}}, + }, + { + "framed: prologue row on the TEXT line", + FuncLayout{Size: 24, Line: 2, Lines: []LineEntry{ + {12, 3}, {16, 4}, + }}, + [][2]int64{{0, 2}, {12, 3}, {16, 4}}, + }, + { + "instructions sharing a line fold into one row", + FuncLayout{Size: 16, Line: 2, Lines: []LineEntry{ + {0, 3}, {4, 3}, {8, 4}, {12, 4}, + }}, + [][2]int64{{0, 3}, {8, 4}}, + }, + { + "large gap crosses PC_RANGE", + FuncLayout{Size: 60, Line: 2, Lines: []LineEntry{ + {0, 3}, {40, 4}, + }}, + [][2]int64{{0, 3}, {40, 4}}, + }, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + prog, relocs := goobjDwarfLines(c.fn, 0) + if len(relocs) != 1 || relocs[0].off != 3 || relocs[0].siz != 8 || relocs[0].typ != relocAddr || relocs[0].sym != 0 { + t.Fatalf("relocs = %+v", relocs) + } + pcs, lines := decodeDWLineProgram(t, prog) + if len(pcs) != len(c.want) { + t.Fatalf("rows = %d (%v / %v), want %d", len(pcs), pcs, lines, len(c.want)) + } + for i, w := range c.want { + if pcs[i] != uint64(w[0]) || lines[i] != w[1] { + t.Errorf("row %d = (%d, %d), want (%d, %d)", i, pcs[i], lines[i], w[0], w[1]) + } + } + }) + } +} + +// TestGoobjDwarfInfo checks the subprogram DIE for an exported and a +// static function: the abbrev, name, high_pc, frame base, decl file/line, +// the external flag and the addrx relocation position. +func TestGoobjDwarfInfo(t *testing.T) { + fn := FuncLayout{Size: 20, Line: 2} + die, relocs := goobjDwarfInfo(fn, "pkg.f", 3) + want := []byte{ + 0x03, + 'p', 'k', 'g', '.', 'f', 0, + 0, 0, 0, 0, // addrx slot at offset 7 + 0x14, // high_pc: 20 + 0x01, 0x9c, // frame_base + 0x01, 0, 0, 0, // decl_file 1 + 0x02, // decl_line 2 + 0x01, // external + 0x00, // end of children + } + if !bytes.Equal(die, want) { + t.Errorf("DIE = %x, want %x", die, want) + } + if len(relocs) != 1 || relocs[0].off != 7 || relocs[0].siz != 4 || relocs[0].typ != relocDWTXTADDRU4 || relocs[0].sym != 3 { + t.Errorf("relocs = %+v", relocs) + } + + // A static function carries no external flag and no package prefix. + fn.Static = true + die, _ = goobjDwarfInfo(fn, "f", 1) + if die[len(die)-2] != 0 { + t.Errorf("static external flag = %d, want 0", die[len(die)-2]) + } +} + +// TestDwPutPCLCDeltaRemainders checks the standard-opcode remainders: +// const_add_pc and fixed_advance_pc after a special opcode. +func TestDwPutPCLCDeltaRemainders(t *testing.T) { + // deltaPC 25 past PC_RANGE: opcode 26 covers (1, 1), const_add_pc + // covers the remaining 23 pc and 0 line. + got := dwPutPCLCDelta(nil, 25, 1) + if !bytes.Equal(got, []byte{8, 26}) { + t.Errorf("25/1 = %x, want [8 1a]", got) + } + + // deltaPC 20000: opcode 246 covers 23, fixed_advance_pc covers the + // remaining 19977. + got = dwPutPCLCDelta(nil, 20000, 1) + if got[0] != 9 || binary.LittleEndian.Uint16(got[1:]) != 19977 || got[3] != 246 { + t.Errorf("20000/1 = %x, want fixed_advance_pc 19977 then 246", got) + } + + // Line remainder: deltaLC 10 leaves 5 past the opcode's reach, encoded + // as advance_line 5 (zigzag 0x0a) before opcode 250. + got = dwPutPCLCDelta(nil, 23, 10) + if !bytes.Equal(got, []byte{3, 0x0a, 250}) { + t.Errorf("23/10 = %x, want [03 0a fa]", got) + } + + // Negative line remainder: deltaLC -5 leaves advance_line -1 (zigzag + // 0x01) after opcode 11. + got = dwPutPCLCDelta(nil, 0, -5) + if !bytes.Equal(got, []byte{3, 1, 11}) { + t.Errorf("0/-5 = %x, want [03 01 0b]", got) + } +} diff --git a/asm/goobj_test.go b/asm/goobj_test.go index b895aa7..f047075 100644 --- a/asm/goobj_test.go +++ b/asm/goobj_test.go @@ -111,17 +111,26 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09 t.Errorf("flags = %#x, want ObjFlagFromAssembly (4)", flags) } - // Package defs: the static GLOBL, then one anonymous FuncInfo per - // function. + // Package defs: the static GLOBL, then per function the FuncInfo and the + // two DWARF symbols (debug_line program, subprogram DIE). defs := v.syms(blkSymdef) - if len(defs) != 3 { - t.Fatalf("symdefs = %d, want 3", len(defs)) + if len(defs) != 7 { + t.Fatalf("symdefs = %d, want 7", len(defs)) } if defs[0].name != "mask" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 16 || defs[0].flag2 != symFlag2Link { t.Errorf("mask symbol = %+v", defs[0]) } if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 { - t.Errorf("funcinfo symbol = %+v", defs[1]) + t.Errorf("addq funcinfo symbol = %+v", defs[1]) + } + if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 { + t.Errorf("addq lines symbol = %+v", defs[2]) + } + if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 { + t.Errorf("addq DIE symbol = %+v", defs[3]) + } + if defs[4].name != "" || defs[4].typ != kindSDATA || defs[4].size != 28 { + t.Errorf("loadmask funcinfo symbol = %+v", defs[4]) } // Non-package defs: four pc tables and the function, per function. @@ -142,45 +151,62 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09 // FuncInfo: args 24, FuncFlag Asm, one file, no inline tree. le := binary.LittleEndian data := v.blk(blkData) + didx := v.blk(blkDataIdx) fi := data[16:44] if le.Uint32(fi[0:]) != 24 || le.Uint32(fi[4:]) != 0 || fi[8] != 0 || fi[9] != funcFlagAsm || le.Uint32(fi[16:]) != 1 || le.Uint32(fi[20:]) != 0 || le.Uint32(fi[24:]) != 0 { t.Errorf("funcinfo bytes %x", fi) } - // pcsp: a flat zero over the whole function (zero-frame NOSPLIT). - if got := data[72:75]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) { + // The pc-value tables of addq (non-package indices 0–3, so global + // indices 7–10): pcsp a flat zero over the whole function, pcinline a + // flat -1, both with the pc delta in MinLC (1) units. + pcsp := data[le.Uint32(didx[4*7:]):] + if got := pcsp[:3]; !bytes.Equal(got, []byte{0x02, 19, 0x00}) { t.Errorf("pcsp = %x, want 021300", got) } - // pcinline: a flat -1. - if got := data[81:84]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) { + pcinl := data[le.Uint32(didx[4*10:]):] + if got := pcinl[:3]; !bytes.Equal(got, []byte{0x00, 19, 0x00}) { t.Errorf("pcinline = %x, want 001300", got) } - // The one relocation: R_PCREL, four bytes wide, against the GLOBL, - // with the field in the function code left zero. The loadmask code's - // offset comes from the data index (symbol 3 defs + 9 non-package). + // Relocations: the four DWARF address references (two per function, in + // definition order), then the loadmask code's R_PCREL against the + // GLOBL, with the field in the function code left zero. The loadmask + // code's offset comes from the data index (7 defs + 9 non-package). relocs := v.blk(blkReloc) - if len(relocs) != 23 { - t.Fatalf("relocs = %d bytes, want one 23-byte entry", len(relocs)) + if len(relocs) != 5*23 { + t.Fatalf("relocs = %d bytes, want 5 entries", len(relocs)) } - off := int32(le.Uint32(relocs[0:])) - if off != 4 || relocs[4] != 4 || le.Uint16(relocs[5:]) != relocPCRel || - le.Uint64(relocs[7:]) != 0 || le.Uint32(relocs[15:]) != pkgIdxSelf || le.Uint32(relocs[19:]) != 0 { - t.Errorf("reloc = %x", relocs) + // addq's DWARF references (defs 2 and 3) against the function, which + // is non-package index 4. + lr := relocs[:23] + if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr || + le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 { + t.Errorf("addq lines reloc = %x", lr) } - didx := v.blk(blkDataIdx) - lm := le.Uint32(didx[4*(3+9):]) + dr := relocs[23:46] + if dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4 || + le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 { + t.Errorf("addq DIE reloc = %x", dr) + } + cr := relocs[4*23:] + off := int32(le.Uint32(cr[0:])) + if off != 4 || cr[4] != 4 || le.Uint16(cr[5:]) != relocPCRel || + le.Uint64(cr[7:]) != 0 || le.Uint32(cr[15:]) != pkgIdxSelf || le.Uint32(cr[19:]) != 0 { + t.Errorf("loadmask reloc = %x", cr) + } + lm := le.Uint32(didx[4*16:]) code := data[lm : lm+18] if !bytes.Equal(code[4:8], []byte{0, 0, 0, 0}) { t.Errorf("relocated field = %x, want zeroed", code[4:8]) } - // Aux wiring: FuncInfo (package symbol), then the four pc tables - // (non-package symbols). + // Aux wiring: FuncInfo, the two DWARF symbols (package symbols), then + // the four pc tables (non-package symbols). auxs := v.blk(blkAux) - if len(auxs) != 2*5*9 { - t.Fatalf("aux = %d bytes, want 10 entries", len(auxs)) + if len(auxs) != 2*7*9 { + t.Fatalf("aux = %d bytes, want 14 entries", len(auxs)) } wantAux := []struct { typ uint8 @@ -188,15 +214,19 @@ DATA mask<>+8(SB)/8, $0x800f0e0d0c0b0a09 idx uint32 }{ {auxFuncInfo, pkgIdxSelf, 1}, - {auxPcsp, pkgIdxNone, uint32(len(defs) + 0)}, - {auxPcfile, pkgIdxNone, uint32(len(defs) + 1)}, - {auxPcline, pkgIdxNone, uint32(len(defs) + 2)}, - {auxPcinline, pkgIdxNone, uint32(len(defs) + 3)}, - {auxFuncInfo, pkgIdxSelf, 2}, - {auxPcsp, pkgIdxNone, uint32(len(defs) + 5)}, - {auxPcfile, pkgIdxNone, uint32(len(defs) + 6)}, - {auxPcline, pkgIdxNone, uint32(len(defs) + 7)}, - {auxPcinline, pkgIdxNone, uint32(len(defs) + 8)}, + {auxDwarfInfo, pkgIdxSelf, 3}, + {auxDwarfLines, pkgIdxSelf, 2}, + {auxPcsp, pkgIdxNone, 0}, + {auxPcfile, pkgIdxNone, 1}, + {auxPcline, pkgIdxNone, 2}, + {auxPcinline, pkgIdxNone, 3}, + {auxFuncInfo, pkgIdxSelf, 4}, + {auxDwarfInfo, pkgIdxSelf, 6}, + {auxDwarfLines, pkgIdxSelf, 5}, + {auxPcsp, pkgIdxNone, 5}, + {auxPcfile, pkgIdxNone, 6}, + {auxPcline, pkgIdxNone, 7}, + {auxPcinline, pkgIdxNone, 8}, } for i, w := range wantAux { e := auxs[i*9:] @@ -253,7 +283,7 @@ TEXT ·framed(SB), NOSPLIT, $8-0 t.Fatalf("AssembleFile: %v", err) } fn := img.Funcs[0] - pcs, vals := decodePCValues(pcspTable(fn)) + pcs, vals := decodePCValues(pcspTable(fn, 1)) // Prologue: PUSHQ BP (1 byte, +8), MOVQ SP, BP (3 bytes, no change), // SUBQ $8, SP (4 bytes, +16 in total); the RET's epilogue unwinds // ADDQ $8, SP (+8) then POPQ BP (0). @@ -401,13 +431,12 @@ func main() { if err != nil { t.Fatalf("GOObject: %v", err) } - if err := os.WriteFile(asmObj, obj, 0o644); err != nil { - t.Fatal(err) - } // Rebuild the package archive with our object in place of the // toolchain's (go tool pack has no replace-in-place that dedupes, so - // extract, substitute and repack). + // extract, substitute and repack). The archive member holding the + // assembler's output is named after the asm object file, e.g. + // main_amd64.o. extract := exec.Command(goBin, "tool", "pack", "x", pkgArch) membersDir := filepath.Join(dir, "members") if err := os.MkdirAll(membersDir, 0o755); err != nil { @@ -417,6 +446,13 @@ func main() { if out, err := extract.CombinedOutput(); err != nil { t.Fatalf("pack x: %v\n%s", err, out) } + member := filepath.Join(membersDir, filepath.Base(asmObj)) + if err := os.Chmod(member, 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(member, obj, 0o644); err != nil { + t.Fatal(err) + } listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch) listOut, err := listCmd.CombinedOutput() if err != nil { diff --git a/asm/goobjloong64.go b/asm/goobjloong64.go new file mode 100644 index 0000000..14578c8 --- /dev/null +++ b/asm/goobjloong64.go @@ -0,0 +1,91 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "fmt" + "os" + "os/exec" + "path/filepath" + "sync" +) + +// GOObjectLOONG64 emits a GOOBJ object file for LoongArch. The layout is +// the shared one in goobj.go — the toolchain preamble, the go120ld header +// with its block offsets, the string table, the symbol definitions and the +// reloc/aux/data index arrays — with the loong64 preamble, the MinLC of 4 +// for the pc-value deltas, and R_LOONG64_ADDR_HI/LO relocation types for +// the pcalau12i+addi.d address pairs. +func (img *Image) GOObjectLOONG64(pkgPath, srcPath string) ([]byte, error) { + pre, err := toolchainObjectPreambleLOONG64() + if err != nil { + return nil, err + } + return img.emitGOObject(pkgPath, srcPath, pre, 4, func(r Reloc) uint16 { + // A pcalau12i+addi.d pair: the high part carries + // R_LOONG64_ADDR_HI, the low part R_LOONG64_ADDR_LO. + if r.Kind == RelLoong64AddrLo { + return relocLoong64AddrLo + } + return relocLoong64AddrHi + }) +} + +// Loong64 relocation types (cmd/internal/objabi). R_LOONG64_ADDR_HI +// resolves the high 20 bits of a PC-relative address into pcalau12i; +// R_LOONG64_ADDR_LO the low 12 bits into addi.d/ld/st. +const ( + relocLoong64AddrHi = 77 // R_LOONG64_ADDR_HI + relocLoong64AddrLo = 78 // R_LOONG64_ADDR_LO +) + +// toolchainObjectPreambleLOONG64 returns the "go object ...\n!\n" header +// the installed go tool asm writes for loong64, captured by assembling a +// one-instruction probe (see toolchainObjectPreamble). +var ( + preambleLOONG64Once sync.Once + preambleLOONG64 []byte + preambleLOONG64Err error +) + +func toolchainObjectPreambleLOONG64() ([]byte, error) { + preambleLOONG64Once.Do(func() { + goBin, err := exec.LookPath("go") + if err != nil { + preambleLOONG64Err = fmt.Errorf("GOOBJ emission needs the Go toolchain: %w", err) + return + } + dir, err := os.MkdirTemp("", "gasm-preamble-loong64") + if err != nil { + preambleLOONG64Err = err + return + } + defer os.RemoveAll(dir) + src := filepath.Join(dir, "probe_loong64.s") + if err := os.WriteFile(src, []byte("TEXT \u00b7x(SB), $0-0\n\tRET\n"), 0o644); err != nil { + preambleLOONG64Err = err + return + } + obj := filepath.Join(dir, "probe.o") + cmd := exec.Command(goBin, "tool", "asm", "-p", "probe", "-o", obj, src) + cmd.Env = append(os.Environ(), "GOARCH=loong64") + if out, err := cmd.CombinedOutput(); err != nil { + preambleLOONG64Err = fmt.Errorf("probing the assembler for the object header: %v\n%s", err, out) + return + } + data, err := os.ReadFile(obj) + if err != nil { + preambleLOONG64Err = err + return + } + i := bytes.Index(data, []byte("\n!\n")) + if i < 0 || !bytes.HasPrefix(data[i+3:], []byte(goobjMagic)) { + preambleLOONG64Err = fmt.Errorf("unrecognised assembler object layout") + return + } + preambleLOONG64 = data[:i+3] + }) + return preambleLOONG64, preambleLOONG64Err +} diff --git a/asm/goobjriscv.go b/asm/goobjriscv.go index 5515cbc..5463b4e 100644 --- a/asm/goobjriscv.go +++ b/asm/goobjriscv.go @@ -75,10 +75,10 @@ func (img *Image) GOObjectRISCV(pkgPath, srcPath string) ([]byte, error) { data []byte dst *int }{ - {pcspTable(fn), &pcIdx[i].sp}, - {pcValueFlat(0, fn.Size), &pcIdx[i].file}, - {pcValueFlat(int32(fn.Line), fn.Size), &pcIdx[i].line}, - {pcValueFlat(-1, fn.Size), &pcIdx[i].inl}, + {pcspTable(fn, 2), &pcIdx[i].sp}, + {pcValueFlat(0, fn.Size, 2), &pcIdx[i].file}, + {pcValueFlat(int32(fn.Line), fn.Size, 2), &pcIdx[i].line}, + {pcValueFlat(-1, fn.Size, 2), &pcIdx[i].inl}, } for _, t := range tables { *t.dst = len(nps) diff --git a/asm/l64_goobj_test.go b/asm/l64_goobj_test.go new file mode 100644 index 0000000..66aba09 --- /dev/null +++ b/asm/l64_goobj_test.go @@ -0,0 +1,326 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "encoding/binary" + "os" + "os/exec" + "path/filepath" + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestGOObjectLOONG64Structure checks the emitted loong64 object's blocks: +// the symbol tables, the function code bytes and the relocation wiring. +func TestGOObjectLOONG64Structure(t *testing.T) { + f, errs := parser.Parse("k_loong64.s", ` +#include "textflag.h" + +TEXT ·add(SB), NOSPLIT, $0-24 + MOVV a+0(FP), R4 + MOVV b+8(FP), R5 + ADDV R5, R4, R4 + MOVV R4, ret+16(FP) + RET + +GLOBL ·table<>(SB), RODATA, $8 +DATA ·table<>+0(SB)/8, $0x1122334455667788 +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + obj, err := img.GOObjectLOONG64("testpkg", "k_loong64.s") + if err != nil { + t.Fatalf("GOObjectLOONG64: %v", err) + } + v := openGoobj(t, obj) + + // Package defs: the static GLOBL, then the FuncInfo and the two DWARF + // symbols (debug_line program, subprogram DIE). + defs := v.syms(blkSymdef) + if len(defs) != 4 { + t.Fatalf("symdefs = %d, want 4", len(defs)) + } + if defs[0].name != "table" || defs[0].abi != 0xffff || defs[0].typ != kindSRODATA || defs[0].size != 8 { + t.Errorf("table symbol = %+v", defs[0]) + } + if defs[1].name != "" || defs[1].typ != kindSDATA || defs[1].size != 28 { + t.Errorf("funcinfo symbol = %+v", defs[1]) + } + if defs[2].name != "" || defs[2].typ != kindSDWARFLINES || defs[2].size == 0 { + t.Errorf("lines symbol = %+v", defs[2]) + } + if defs[3].name != "" || defs[3].typ != kindSDWARFFCN || defs[3].size == 0 { + t.Errorf("DIE symbol = %+v", defs[3]) + } + + // Non-package defs: four pc tables and the function. + nps := v.syms(blkNonpkgdef) + if len(nps) != 5 { + t.Fatalf("nonpkgdefs = %d, want 5", len(nps)) + } + fn := nps[4] + if fn.name != "testpkg.add" || fn.typ != kindSTEXT || fn.flag != symFlagNoSplit || fn.size != 20 { + t.Errorf("add symbol = %+v", fn) + } + + // The function code: 20 bytes, the ground-truth encoding. It sits + // after the GLOBL, FuncInfo, two DWARF symbols and four pc tables. + dataIdx := v.blk(blkDataIdx) + dataBlk := v.blk(blkData) + le := binary.LittleEndian + dOff := le.Uint32(dataIdx[8*4:]) + code := dataBlk[dOff : dOff+20] + want := []byte{ + 0x64, 0x20, 0xc0, 0x28, // ld.d r4, 8(r3) + 0x65, 0x40, 0xc0, 0x28, // ld.d r5, 16(r3) + 0x84, 0x94, 0x10, 0x00, // add.d r4, r4, r5 + 0x64, 0x60, 0xc0, 0x29, // st.d r4, 24(r3) + 0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0 + } + for i := range want { + if code[i] != want[i] { + t.Fatalf("code byte %d = %02x, want %02x", i, code[i], want[i]) + } + } + + // The debug_line program: LNE_set_address (the R_ADDR relocation + // carries the function address), then one row per line change — the + // TEXT is on line 4 (a leading blank line precedes the include), the + // instructions on lines 5–9 — an advance to the 20-byte end and an + // end-of-sequence. + linesOff := le.Uint32(dataIdx[4*2:]) + lines := dataBlk[linesOff : linesOff+21] + wantLines := []byte{ + 0x00, 0x09, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, // LNE_set_address + 0x13, // pc 0, line 5 + 0x38, // pc 4, line 6 + 0x38, // pc 8, line 7 + 0x38, // pc 12, line 8 + 0x38, // pc 16, line 9 + 0x02, 0x04, // advance_pc to 20 + 0x00, 0x01, 0x01, // end_sequence + } + for i := range wantLines { + if lines[i] != wantLines[i] { + t.Fatalf("lines byte %d = %02x, want %02x", i, lines[i], wantLines[i]) + } + } + + // The subprogram DIE: abbrev 3 (FUNCTION), the qualified name, the + // addrx low_pc slot (R_DWTXTADDR_U4), the size as high_pc, the + // call-frame-CFA frame base, decl file/line and the external flag. + dieOff := le.Uint32(dataIdx[4*3:]) + die := dataBlk[dieOff : dieOff+27] + wantDie := []byte{ + 0x03, + 't', 'e', 's', 't', 'p', 'k', 'g', '.', 'a', 'd', 'd', 0, + 0x00, 0x00, 0x00, 0x00, // low_pc: addrx slot + 0x14, // high_pc: 20 + 0x01, 0x9c, // frame_base: DW_OP_call_frame_cfa + 0x01, 0x00, 0x00, 0x00, // decl_file: 1 + 0x04, // decl_line: 4 + 0x01, // external + 0x00, // end of children + } + for i := range wantDie { + if die[i] != wantDie[i] { + t.Fatalf("DIE byte %d = %02x, want %02x", i, die[i], wantDie[i]) + } + } + + // The DWARF symbols carry the function-address references: R_ADDR for + // the line program's set_address, R_DWTXTADDR_U4 for the DIE's addrx + // slot, both against the function's non-package index. The reloc + // index counts relocations, not bytes. + relocIdx := v.blk(blkRelocIdx) + relocs := v.blk(blkReloc) + if le.Uint32(relocIdx[4*2:]) != 0 || le.Uint32(relocIdx[4*3:]) != 1 || le.Uint32(relocIdx[4*4:]) != 2 { + t.Fatalf("dwarf reloc index ranges: %d %d %d", le.Uint32(relocIdx[4*2:]), le.Uint32(relocIdx[4*3:]), le.Uint32(relocIdx[4*4:])) + } + lr := relocs[:23] + if int32(le.Uint32(lr[0:])) != 3 || lr[4] != 8 || le.Uint16(lr[5:]) != relocAddr || + le.Uint32(lr[15:]) != pkgIdxNone || le.Uint32(lr[19:]) != 4 { + t.Errorf("lines reloc = %x", lr) + } + dr := relocs[23:46] + if int32(le.Uint32(dr[0:])) != 13 || dr[4] != 4 || le.Uint16(dr[5:]) != relocDWTXTADDRU4 || + le.Uint32(dr[15:]) != pkgIdxNone || le.Uint32(dr[19:]) != 4 { + t.Errorf("die reloc = %x", dr) + } + + // The pc-value deltas are in MinLC (4) units: the flat pcsp covers + // the whole 20-byte function with a delta of 5. + pcspOff := le.Uint32(dataIdx[4*4:]) + if got := dataBlk[pcspOff : pcspOff+3]; !bytes.Equal(got, []byte{0x02, 0x05, 0x00}) { + t.Errorf("pcsp = %x, want 020500", got) + } +} + +// TestGOObjectLOONG64Link cross-compiles a Go program with the gasm-produced +// object substituted into the package archive, proving cmd/link accepts the +// emitted GOOBJ. The binary is not executed (no LoongArch host or qemu). +// Skipped when no Go toolchain is available. +func TestGOObjectLOONG64Link(t *testing.T) { + goBin, err := exec.LookPath("go") + if err != nil { + t.Skip("no Go toolchain available") + } + dir := t.TempDir() + asmSrc := `#include "textflag.h" +TEXT ·add(SB), NOSPLIT, $0-24 + MOVV a+0(FP), R4 + MOVV b+8(FP), R5 + ADDV R5, R4, R4 + MOVV R4, ret+16(FP) + RET +` + if err := os.WriteFile(filepath.Join(dir, "main_loong64.s"), []byte(asmSrc), 0o644); err != nil { + t.Fatal(err) + } + mainSrc := `package main + +func add(a, b int64) int64 + +func main() { + if add(20, 22) != 42 { + panic("bad add") + } +} +` + if err := os.WriteFile(filepath.Join(dir, "main.go"), []byte(mainSrc), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dir, "go.mod"), []byte("module l64link\n\ngo 1.21\n"), 0o644); err != nil { + t.Fatal(err) + } + + // Capture the cross build (GOARCH=loong64): the package archive and the + // link line. + build := exec.Command(goBin, "build", "-x", "-work", "-o", filepath.Join(dir, "prog"), ".") + build.Dir = dir + build.Env = append(os.Environ(), "GOARCH=loong64") + buildLog, err := build.CombinedOutput() + if err != nil { + t.Fatalf("baseline build: %v\n%s", err, buildLog) + } + var pkgArch, work, linkLine, asmObj string + for _, line := range strings.Split(string(buildLog), "\n") { + switch { + case strings.HasPrefix(line, "WORK="): + work = strings.TrimPrefix(line, "WORK=") + case strings.Contains(line, "/asm ") && strings.Contains(line, "main_loong64.s") && !strings.Contains(line, "-gensymabis"): + asmObj = fieldAfter(line, "-o") + case strings.Contains(line, "pack r") && strings.Contains(line, "_pkg_.a"): + pkgArch = strings.TrimSpace(strings.SplitN(line, "pack r", 2)[1]) + pkgArch = strings.Fields(strings.SplitN(pkgArch, "#", 2)[0])[0] + case strings.Contains(line, "/link ") && strings.Contains(line, "-importcfg"): + linkLine = line + } + } + if pkgArch == "" || linkLine == "" || asmObj == "" { + t.Skip("could not locate the archive, asm output or link line in the build log") + } + pkgArch = strings.ReplaceAll(pkgArch, "$WORK", work) + // The archive member holding the assembler's output is named after the + // asm object file (main_loong64.o), as cmd/go packs it with `pack r`. + asmMember := filepath.Base(strings.ReplaceAll(asmObj, "$WORK", work)) + + // Assemble the same source with gasm and swap the object in. + pf, perrs := parser.Parse(filepath.Join(dir, "main_loong64.s"), asmSrc) + if len(perrs) > 0 { + t.Fatalf("parse: %v", perrs) + } + pimg, err := AssembleFileLOONG64(pf) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + obj, err := pimg.GOObjectLOONG64("main", filepath.Join(dir, "main_loong64.s")) + if err != nil { + t.Fatalf("GOObjectLOONG64: %v", err) + } + + // Extract the archive, substitute the object member, repack. + membersDir := filepath.Join(dir, "members") + if err := os.MkdirAll(membersDir, 0o755); err != nil { + t.Fatal(err) + } + extract := exec.Command(goBin, "tool", "pack", "x", pkgArch) + extract.Dir = membersDir + extract.Env = append(os.Environ(), "GOARCH=loong64") + if out, err := extract.CombinedOutput(); err != nil { + t.Fatalf("pack x: %v\n%s", err, out) + } + // Substitute the gasm object for the assembler's archive member (pack + // extracts members read-only). + member := filepath.Join(membersDir, asmMember) + if err := os.Chmod(member, 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(member, obj, 0o644); err != nil { + t.Fatal(err) + } + listCmd := exec.Command(goBin, "tool", "pack", "t", pkgArch) + listCmd.Env = append(os.Environ(), "GOARCH=loong64") + listOut, err := listCmd.CombinedOutput() + if err != nil { + t.Fatalf("pack t: %v\n%s", err, listOut) + } + newArch := filepath.Join(dir, "pkg.a") + args := []string{"tool", "pack", "c", newArch} + seen := map[string]bool{} + for _, m := range strings.Fields(string(listOut)) { + if seen[m] { + continue + } + seen[m] = true + if err := os.Chmod(filepath.Join(membersDir, m), 0o644); err != nil { + t.Fatal(err) + } + args = append(args, filepath.Join(membersDir, m)) + } + pack := exec.Command(goBin, args...) + pack.Dir = membersDir + pack.Env = append(os.Environ(), "GOARCH=loong64") + if out, err := pack.CombinedOutput(); err != nil { + t.Fatalf("pack c: %v\n%s", err, out) + } + + // Re-link with our archive in place of the toolchain's. The link line + // carries a GOROOT assignment and $WORK placeholders; run it through the + // shell with the GOEXPERIMENT and GOARCH the toolchain expects (the + // linker compares the object header against its own, experiments + // included). + linkLine = strings.ReplaceAll(linkLine, "$WORK", work) + linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "_pkg_.a"), newArch) + linkLine = strings.ReplaceAll(linkLine, filepath.Join(work, "b001", "exe", "a.out"), filepath.Join(dir, "app2")) + link := exec.Command("sh", "-c", linkLine) + link.Dir = dir + goExp, _ := exec.Command(goBin, "env", "GOEXPERIMENT").Output() + link.Env = append(os.Environ(), "GOEXPERIMENT="+strings.TrimSpace(string(goExp)), "GOARCH=loong64") + if out, err := link.CombinedOutput(); err != nil { + t.Fatalf("link with gasm object: %v\n%s", err, out) + } + + // The binary is not executed: there is no LoongArch host or qemu here. + // The link itself and the symbol table prove cmd/link accepted the gasm + // object and laid out the function. + nm := exec.Command(goBin, "tool", "nm", filepath.Join(dir, "app2")) + nm.Env = append(os.Environ(), "GOARCH=loong64") + nmOut, err := nm.CombinedOutput() + if err != nil { + t.Fatalf("nm gasm-linked binary: %v\n%s", err, nmOut) + } + if !strings.Contains(string(nmOut), "main.add") { + t.Errorf("main.add not found in linked binary:\n%s", nmOut) + } +} diff --git a/asm/link.go b/asm/link.go index 7724dac..53484d9 100644 --- a/asm/link.go +++ b/asm/link.go @@ -89,11 +89,13 @@ func (fl *FuncLayout) LineAt(offset int) int { type RelocKind int const ( - RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64) - RelPCRelHI20 // R_RISCV_PCREL_HI20 (AUIPC) - RelPCRelLO12 // R_RISCV_PCREL_LO12_I (ADDI, LD) - RelPCRelLO12S // R_RISCV_PCREL_LO12_S (SD) - RelPCRelAbs // 32-bit absolute (R_RISCV_32) + RelPCRel32 RelocKind = iota // 32-bit PC-relative (amd64) + RelPCRelHI20 // R_RISCV_PCREL_HI20 (AUIPC) + RelPCRelLO12 // R_RISCV_PCREL_LO12_I (ADDI, LD) + RelPCRelLO12S // R_RISCV_PCREL_LO12_S (SD) + RelPCRelAbs // 32-bit absolute (R_RISCV_32) + RelLoong64AddrHi // R_LOONG64_ADDR_HI (pcalau12i) + RelLoong64AddrLo // R_LOONG64_ADDR_LO (addi.d/ld/st) ) type Reloc struct { @@ -289,7 +291,78 @@ func AssembleFileRISCV(f *ast.File) (*Image, error) { img.DataSyms = append(img.DataSyms, DataSymbol{ Name: d.name, Pkg: d.pkg, - Offset: pos, + Offset: len(img.Data) - len(d.buf), // relative to the data section + Size: d.size, + Static: d.static, + Rodata: d.rodata, + Dupok: d.dupok, + }) + } + + return img, nil +} + +// AssembleFileLOONG64 assembles every TEXT function of a parsed loong64 file +// and lays out its static symbols (GLOBL/DATA) in a data section behind the +// code. SB references in the code are encoded as pcalau12i pairs with zero +// immediates; the object-file emitters record R_LOONG64_ADDR_HI/LO +// relocations for the linker. +func AssembleFileLOONG64(f *ast.File) (*Image, error) { + dataSyms, err := collectData(f) + if err != nil { + return nil, err + } + + img := &Image{Symbols: map[string]int{}} + for _, d := range f.Decls { + t, ok := d.(*ast.Text) + if !ok { + continue + } + code, labels, relocs, lines, spadj, err := assembleLOONG64(t) + if err != nil { + return nil, fmt.Errorf("%s: %w", t.Name.Name, err) + } + fl := FuncLayout{ + Name: t.Name.Name, + Pkg: t.Name.Pkg, + Static: t.Name.Static, + Offset: len(img.Code), + Size: len(code), + Frame: frameSize(t), + Args: argsSize(t), + Line: t.Pos().Line, + Labels: labels, + Lines: lines, + Spadj: spadj, + Relocs: relocs, + } + for _, f := range t.Flags { + switch f { + case "NOSPLIT": + fl.NoSplit = true + case "SPWRITE": + fl.SPWrite = true + } + } + img.Funcs = append(img.Funcs, fl) + img.Code = append(img.Code, code...) + } + + // Lay out the data section behind the code, 16-aligned. + dataStart := len(img.Code) + for _, d := range dataSyms { + pos := dataStart + len(img.Data) + for pos%16 != 0 { + img.Data = append(img.Data, 0) + pos++ + } + img.Symbols[d.name] = pos + img.Data = append(img.Data, d.buf...) + img.DataSyms = append(img.DataSyms, DataSymbol{ + Name: d.name, + Pkg: d.pkg, + Offset: len(img.Data) - len(d.buf), // relative to the data section Size: d.size, Static: d.static, Rodata: d.rodata, diff --git a/asm/loong64_assemble.go b/asm/loong64_assemble.go index f5d6f8e..1132bb9 100644 --- a/asm/loong64_assemble.go +++ b/asm/loong64_assemble.go @@ -5,19 +5,1386 @@ package asm import ( "fmt" + "math/bits" + "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) -// assembleLOONG64 is a stub. The loong64 (LoongArch) instruction encoder is -// not yet implemented — the instruction tables, register files and operand-count -// metadata are in place (package arch), and the lexer, parser, formatter and -// linter already handle loong64 source files. -func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, error) { - return nil, nil, nil, fmt.Errorf("loong64 instruction encoding is not yet implemented") +// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into +// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and +// the immediate-arithmetic forms expand to 2–5 instructions when the +// immediate does not fit, so the layout is computed in two passes (sizes, +// then encoding with resolved branch targets). +// +// The emitted bytes match the Go toolchain's loong64 assembler, which is the +// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch +// encodings and the MOV immediate expansions all follow cmd/internal/obj/ +// loong64's asmout cases. +func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { + fi := loong64ComputeFrame(t) + prologue := loong64Prologue(fi) + chain := loong64JumpChain(t) + resolve := func(name string) string { + if r, ok := chain[name]; ok { + return r + } + return name + } + + var relocs []Reloc + var spadj []SpadjStep + + // The prologue (3 instructions when a frame is present) raises the SP + // delta by autosize; the boundary is reported at the third instruction's + // pc, exactly as the toolchain's pctospadj does. + if fi.autosize != 0 { + spadj = append(spadj, SpadjStep{PC: 8, Value: fi.autosize}) + } + + // Pass 1: label offsets from the instruction sizes. + offsets := map[string]int{} + pos := len(prologue) + for _, stmt := range t.Body { + switch s := stmt.(type) { + case *ast.Label: + offsets[s.Name.Text] = pos + case *ast.Instr: + pos += loong64InstrSize(s, fi) + } + } + + // Pass 2: encode. Relocation offsets are recorded function-relative. + out := append([]byte(nil), prologue...) + pc := len(prologue) + preCount := len(relocs) + var lines []LineEntry + for _, stmt := range t.Body { + in, ok := stmt.(*ast.Instr) + if !ok { + continue + } + code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve) + if err != nil { + return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) + } + for j := preCount; j < len(relocs); j++ { + relocs[j].Off += pc - len(prologue) + } + preCount = len(relocs) + lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) + // The RET's epilogue closes the frame: the SP delta returns to zero + // after the addi.d (one instruction for a leaf, two for a non-leaf + // with the LR restore). + if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { + epi := 4 + if !fi.leaf { + epi = 8 + } + spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0}) + } + out = append(out, code...) + pc += len(code) + } + return out, offsets, relocs, lines, spadj, nil } -// AssembleFileLOONG64 is a stub, returning the same error as assembleLOONG64. -func AssembleFileLOONG64(f *ast.File) (*Image, error) { - return nil, fmt.Errorf("loong64 instruction encoding is not yet implemented") +// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's +// branch-chasing pass: a label whose first instruction is an unconditional +// local jump redirects its own jumpers to the ultimate target. The Go +// toolchain chases these chains before it encodes branches, so matching its +// bytes requires the same redirection. +func loong64JumpChain(t *ast.Text) map[string]string { + leadsTo := map[string]string{} + for i, stmt := range t.Body { + l, ok := stmt.(*ast.Label) + if !ok { + continue + } + j := i + 1 + for j < len(t.Body) { + if _, isLabel := t.Body[j].(*ast.Label); !isLabel { + break + } + j++ + } + if j >= len(t.Body) { + continue + } + in, ok := t.Body[j].(*ast.Instr) + if !ok { + continue + } + mnem := strings.ToUpper(in.Mnemonic.Text) + if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 { + continue + } + if name, ok := l64LabelOK(in.Operands[0]); ok { + leadsTo[l.Name.Text] = name + } + } + chain := map[string]string{} + for name := range leadsTo { + visited := map[string]bool{name: true} + cur := name + for { + next, ok := leadsTo[cur] + if !ok || visited[next] { + break + } + visited[next] = true + cur = next + } + if cur != name { + chain[name] = cur + } + } + return chain +} + +// l64LabelOK returns the local label name of a jump operand. +func l64LabelOK(op *ast.Operand) (string, bool) { + if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && + op.Addr.Base == "" && op.Addr.Sym.Name != "" { + return op.Addr.Sym.Name, true + } + return "", false +} + +// loong64InstrSize returns the encoded size of an instruction: 4 bytes for +// most, more for the multi-instruction expansions. +func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int { + mnem := strings.ToUpper(instr.Mnemonic.Text) + ops := instr.Operands + + if mnem == "RET" { + return len(loong64Return(fi)) + } + switch mnem { + case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": + return loong64MovSize(mnem, ops, fi) + case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU": + if len(ops) >= 2 && isImmOperand(ops[0]) { + v := l64Imm64(ops[0]) + if v == 0 { + return 4 // folds into the 3R form (rk = R0) + } + switch mnem { + case "ADD", "ADDW", "ADDV", "ADDVU", "SGT", "SGTU": + // C_US12CON (−2048..0x7ff) encodes directly as addi/slti. + if v >= -2048 && v <= 0x7ff { + return 4 + } + // C_U12CON (0x800..0xfff) → ori r30, r0, v; op rd, rj, r30. + if v >= 0x800 && v <= 0xfff { + return 8 + } + default: // AND/OR/XOR + // C_UU12CON (0..0x7ff) encodes directly as andi/ori/xori. + if v >= 0 && v <= 0x7ff { + return 4 + } + // C_S12CON (−2048..−1) → addi.d r30, r0, v; op rd, rj, r30. + if v >= -2048 && v < 0 { + return 8 + } + } + // 0x800..0xfff for AND/OR/XOR and the 32/64-bit ranges go through + // the lu12i.w materialisation. + if v == int64(int32(v)) { + if v&0xfff == 0 && (v < 0x800 || v > 0xfff) { + return 8 // lu12i.w r30, v>>12; op rd, rj, r30 + } + return 12 // lu12i.w r30, v>>12; ori r30, r30, v; op rd, rj, r30 + } + return 4 * (len(l64DconMovWords(0, v)) + 1) // dcon materialisation + op + } + } + return 4 +} + +// encodeLOONG64Instr encodes a single LoongArch instruction. +func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) { + mnem := strings.ToUpper(instr.Mnemonic.Text) + ops := instr.Operands + + // Pseudo-instructions and the branches first. + switch mnem { + case "RET": + return loong64Return(fi), nil + case "NOP", "NOOP": + // andi r0, r0, 0 + return l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)), nil + case "UNDEF": + // break 0 + return l64wordLE(l64i15(l64InstrTable["BREAK"].op, 0)), nil + case "WORD": + if len(ops) != 1 { + return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) + } + return l64wordLE(uint32(immFromOperand(ops[0]))), nil + case "JMP", "B": + return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve) + case "JAL", "CALL", "BL": + return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve) + case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": + return encodeLOONG64Mov(instr, mnem, fi, relocs) + } + + // 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL. + if op, ok := l64branchTable[mnem]; ok { + return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve) + } + // Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ, + // BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0). + if op, ok := l64branch21Table[mnem]; ok { + return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve) + } + // B/BL aliases reached only via JMP/JAL above. + + // The dual-form arithmetic mnemonics: register (3R) or immediate (2RI12). + if de, ok := l64DualTable[mnem]; ok { + if len(ops) >= 2 && isImmOperand(ops[0]) { + if de.shift { + // INSTR $shamt, rd or INSTR $shamt, rj, rd. + if len(ops) != 2 && len(ops) != 3 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + shamt := int(immFromOperand(ops[0])) + rd := l64Reg(ops[len(ops)-1]) + rj := rd + if len(ops) == 3 { + rj = l64Reg(ops[1]) + } + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + // $0 folds into the register form (the toolchain matches the + // zero constant against the 3R optab entry first). + if shamt == 0 { + return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil + } + // The .d variants take a 6-bit amount, the .w variants 5 bits. + if isLoong64ShiftD(de.imm) { + shamt &= 0x3f + } else { + shamt &= 0x1f + } + return l64wordLE(l64irr(de.imm, shamt, rj, rd)), nil + } + return encodeLOONG64ImmArith(mnem, de, ops) + } + // Register form: 3R. + if len(ops) == 3 { + rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]) + if rk < 0 || rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64rrr(de.rrr, rk, rj, rd)), nil + } + if len(ops) == 2 { + rk, rd := l64Reg(ops[0]), l64Reg(ops[1]) + if rk < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64rrr(de.rrr, rk, rd, rd)), nil + } + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + + enc, ok := l64InstrTable[mnem] + if !ok { + return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem) + } + + switch enc.format { + case l64Frrr: + // INSTR rk, rj, rd (3 operands) or INSTR rk, rd (rj = rd). + switch len(ops) { + case 3: + rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]) + if rk < 0 || rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil + case 2: + rk, rd := l64Reg(ops[0]), l64Reg(ops[1]) + if rk < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64rrr(enc.op, rk, rd, rd)), nil + } + return nil, fmt.Errorf("%s expects 2 or 3 register operands, got %d", mnem, len(ops)) + + case l64Frr: + // INSTR rj, rd. + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rj, rd := l64Reg(ops[0]), l64Reg(ops[1]) + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64rr(enc.op, rj, rd)), nil + + case l64Firr: + // LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd. + if len(ops) < 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects an immediate operand", mnem) + } + imm := int(immFromOperand(ops[0])) + rd := l64Reg(ops[len(ops)-1]) + rj := rd + if len(ops) == 3 { + rj = l64Reg(ops[1]) + } + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64irr(enc.op, imm, rj, rd)), nil + + case l64Firr16: + // ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be + // a multiple of 65536 and is shifted right by 16. + if len(ops) < 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects an immediate operand", mnem) + } + v := int(immFromOperand(ops[0])) + if v&0xFFFF != 0 { + return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem) + } + rd := l64Reg(ops[len(ops)-1]) + rj := rd + if len(ops) == 3 { + rj = l64Reg(ops[1]) + } + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64irr16(enc.op, v>>16, rj, rd)), nil + + case l64Firr14: + // LL/SC/MOVWP: INSTR mem, rd (load) or INSTR rd, mem (store); the + // 14-bit offset is scaled by 4 (byte offset >> 2). + rd, rj, off, load, err := l64MemOperands(ops, fi) + if err != nil { + return nil, err + } + op := enc.op + if load && (mnem == "MOVWP" || mnem == "MOVVP") { + // ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field. + op -= 1 << 24 + } + return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil + + case l64Fir20: + // LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm. + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rd := l64Reg(ops[0]) + if rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64ir(enc.op, int(immFromOperand(ops[1])), rd)), nil + + case l64Frrrr: + // FMADD/FMSUB/FNMADD/FNMSUB: INSTR fa, fk, fj, fd (4 operands) or + // INSTR fa, fk, fd (fj = fd). + fa, fk, fj, fd, err := l64FmaOperands(ops) + if err != nil { + return nil, err + } + return l64wordLE(l64rrrr(enc.op, fa, fk, fj, fd)), nil + + case l64Firir: + // BSTRINS/BSTRPICK: INSTR $msb, rj, $lsb, rd (or $msb, rj, rd with + // lsb = 0). + if len(ops) != 4 && len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) + } + msb := int(immFromOperand(ops[0])) + lsb := 0 + rj := l64Reg(ops[1]) + rd := l64Reg(ops[len(ops)-1]) + if len(ops) == 4 { + lsb = int(immFromOperand(ops[2])) + } + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil + + case l64Firrr: + // ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in + // the second register position); the source amount is 1–4, encoded + // as sa-1. + if len(ops) != 4 { + return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) + } + sa := int(immFromOperand(ops[0])) - 1 + rj, rk, rd := l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3]) + if sa < 0 || sa > 3 { + return nil, fmt.Errorf("shift amount out of range [1, 4]") + } + if rk < 0 || rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64irrr(enc.op, sa, rk, rj, rd)), nil + + case l64Fi15: + // SYSCALL/BREAK/DBAR: no operands, or SYSCALL $code / BREAK $code. + code := 0 + if len(ops) == 1 { + code = int(immFromOperand(ops[0])) + } else if len(ops) > 1 { + return nil, fmt.Errorf("%s expects at most 1 operand, got %d", mnem, len(ops)) + } + return l64wordLE(l64i15(enc.op, code)), nil + + case l64Fam: + // AM* val, (addr), result. + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + rk := l64Reg(ops[0]) + rj, _ := l64Mem(ops[1]) + rd := l64Reg(ops[2]) + if rk < 0 || rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil + + case l64Frdtime: + // RDTIME* rd, rj (rd at bits [9:5], rj at bits [4:0]). + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rd, rj := l64Reg(ops[0]), l64Reg(ops[1]) + if rd < 0 || rj < 0 { + return nil, fmt.Errorf("invalid register operand") + } + return l64wordLE(l64rr(enc.op, rd, rj)), nil + + case l64Fpreld: + // PRELD off(rj), $hint. + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + rj, off := l64Mem(ops[0]) + hint := int(immFromOperand(ops[1])) + if rj < 0 { + return nil, fmt.Errorf("invalid operand in %s", mnem) + } + return l64wordLE(l64irr5i(enc.op, int(off), rj, hint)), nil + } + return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops)) +} + +// encodeLOONG64Branch encodes a label or indirect jump/call: +// +// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0 +// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0 +func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) { + if len(instr.Operands) != 1 { + return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands)) + } + op := instr.Operands[0] + if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil { + // Indirect: (rj) → jirl. + rj := loong64RegNum(op.Addr.Base) + if rj < 0 { + return nil, fmt.Errorf("invalid register operand") + } + rd := 0 + if link { + rd = 1 // link register + } + return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil + } + // Direct: label → b/bl. + target := resolve(l64Label(op)) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) + } + v := (targetOff - pc) >> 2 + if v < -1<<25 || v >= 1<<25 { + return nil, fmt.Errorf("branch to %q too far (26-bit range)", target) + } + opc := l64jumpTable[mnem] + return l64wordLE(l64bbl(opc, v)), nil +} + +// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU): +// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain +// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand. +func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { + if len(ops) != 2 && len(ops) != 3 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + target := resolve(l64Label(ops[len(ops)-1])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) + } + v := (targetOff - pc) >> 2 + if len(ops) == 2 { + // Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/ + // BGEZ-family aliases encoded with rj in the rj field. + rj := l64Reg(ops[0]) + if rj < 0 { + return nil, fmt.Errorf("invalid register operand") + } + if (v<<11)>>11 != v { + return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) + } + zop := l64branch21Table["BEQZ"] + if mnem == "BNE" { + zop = l64branch21Table["BNEZ"] + } + if mnem == "BLT" || mnem == "BLTZ" || mnem == "BGTZ" { + zop = l64branch21Table["BLTZ"] + } + if mnem == "BGE" || mnem == "BGEZ" || mnem == "BLEZ" { + zop = l64branch21Table["BGEZ"] + } + return l64wordLE(l64ir21(zop, v, rj)), nil + } + // Two registers: BEQ rj, rd, label. When one is R0 the toolchain + // re-encodes as the 21-bit BEQZ/BNEZ form. + rj, rd := l64Reg(ops[0]), l64Reg(ops[1]) + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + if rj == 0 { + rj, rd = rd, 0 + } + if rd == 0 { + if (v<<11)>>11 != v { + return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) + } + zop := l64branch21Table["BEQZ"] + if mnem == "BNE" { + zop = l64branch21Table["BNEZ"] + } + return l64wordLE(l64ir21(zop, v, rj)), nil + } + if (v<<16)>>16 != v { + return nil, fmt.Errorf("branch to %q too far (16-bit range)", target) + } + return l64wordLE(l64irr16(op, v, rj, rd)), nil +} + +// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and +// BFPT/BFPF use the 21-bit offset form (register in the rj field), while +// BGTZ/BLEZ — which the toolchain encodes with the register in the rd field +// and a 16-bit offset — are handled separately. +func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + target := resolve(l64Label(ops[1])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) + } + v := (targetOff - pc) >> 2 + rj := 0 // BFPT/BFPF default to FCC0 + if mnem != "BFPT" && mnem != "BFPF" { + rj = l64Reg(ops[0]) + if rj < 0 { + return nil, fmt.Errorf("invalid register operand") + } + } + if mnem == "BGTZ" || mnem == "BLEZ" { + // The toolchain swaps the register into the rd field and keeps the + // 16-bit offset form. + if (v<<16)>>16 != v { + return nil, fmt.Errorf("branch to %q too far (16-bit range)", target) + } + return l64wordLE(l64irr16(op, v, 0, rj)), nil + } + if (v<<11)>>11 != v { + return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) + } + return l64wordLE(l64ir21(op, v, rj)), nil +} + +// encodeLOONG64ImmArith encodes an immediate arithmetic/logic instruction, +// expanding the immediate exactly as the toolchain's aclass classifies it: +// +// ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes) +// 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8) +// AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4) +// −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8) +// 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12) +// 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20) +func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 && len(ops) != 3 { + return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) + } + v := l64Imm64(ops[0]) + rd := l64Reg(ops[len(ops)-1]) + rj := rd + if len(ops) == 3 { + rj = l64Reg(ops[1]) + } + if rj < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + + // The two immediate families classify differently. + additive := mnem == "ADD" || mnem == "ADDW" || mnem == "ADDV" || mnem == "ADDVU" || mnem == "SGT" || mnem == "SGTU" + if additive { + if v == 0 { + // $0 folds into the 3R form (rk = R0), matching the toolchain's + // optab matching of the zero constant against the register form. + return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil + } + if v >= -2048 && v <= 0x7ff { + return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil + } + if v >= 0x800 && v <= 0xfff { + return l64WordsLE( + l64irr(0x00e<<22, int(v), 0, 30), // ori r30, r0, v + l64rrr(de.rrr, 30, rj, rd), + ), nil + } + } else { + if v == 0 { + return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil + } + if v >= 0 && v <= 0x7ff { + return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil + } + if v >= -2048 && v < 0 { + return l64WordsLE( + l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v + l64rrr(de.rrr, 30, rj, rd), + ), nil + } + } + + // 32/64-bit constants are materialised in R30 (the assembler temp), + // using the same dcon classification as the toolchain's case 24/60/70/ + // 71/72 sequences. + const ( + lu12iw = 0x0a << 25 + ori = 0x00e << 22 + lu32id = 0x0b << 25 + lu52id = 0x00c << 22 + ) + if v == int64(int32(v)) { + if v&0xfff == 0 && (v < 0x800 || v > 0xfff) { + return l64WordsLE( + l64ir(lu12iw, int(int32(v)>>12), 30), + l64rrr(de.rrr, 30, rj, rd), + ), nil + } + return l64WordsLE( + l64ir(lu12iw, int(int32(v)>>12), 30), + l64irr(ori, int(v), 30, 30), + l64rrr(de.rrr, 30, rj, rd), + ), nil + } + words := l64DconMovWords(30, v) + words = append(words, l64rrr(de.rrr, 30, rj, rd)) + return l64WordsLE(words...), nil +} + +// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of +// the 6-bit (.d) variants — the toolchain distinguishes them by the bit +// position of the opcode field (bits [25:16]). +func isLoong64ShiftD(op uint32) bool { + return op&0x03ff0000 != 0 && op>>25 == 0 +} + +// l64FmaOperands extracts the four fused-multiply-add operands: +// INSTR fa, fk, fj, fd, or INSTR fa, fk, fd with fj = fd. +func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) { + switch len(ops) { + case 4: + fa, fk, fj, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3]) + case 3: + fa, fk, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]) + fj = fd + default: + return 0, 0, 0, 0, fmt.Errorf("expected 3 or 4 operands, got %d", len(ops)) + } + if fa < 0 || fk < 0 || fj < 0 || fd < 0 { + return 0, 0, 0, 0, fmt.Errorf("invalid register operand") + } + return fa, fk, fj, fd, nil +} + +// l64MemOperands extracts (rd, rj, off, load) from a load/store instruction: +// INSTR mem, rd is a load, INSTR rd, mem a store. +func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) { + if len(ops) != 2 { + return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops)) + } + if isMemOperand(ops[0]) { + rd = l64Reg(ops[1]) + rj, off = l64MemWithFrame(ops[0], fi) + load = true + } else if isMemOperand(ops[1]) { + rd = l64Reg(ops[0]) + rj, off = l64MemWithFrame(ops[1], fi) + } else { + return 0, 0, 0, false, fmt.Errorf("expected a memory operand") + } + if rd < 0 || rj < 0 { + return 0, 0, 0, false, fmt.Errorf("invalid operand") + } + return rd, rj, off, load, nil +} + +// ---- the MOV pseudo-instruction ---- + +// encodeLOONG64Mov encodes the MOV family — the load/store/immediate +// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width +// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the +// access width). The forms, mirroring the toolchain: +// +// MOVx $imm, rd load immediate (addi/lu12i+ori/lu32i/lu52i) +// MOVx mem, rd load from memory +// MOVx rd, mem store to memory +// MOVx rs, rd register move (incl. the FP-bank specials) +// MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d) +// MOVx sym(SB), rd load from a static symbol (pcalau12i+ld) +// MOVx rd, sym(SB) store to a static symbol (pcalau12i+st) +func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) { + ops := instr.Operands + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + if mnem == "MOV" { + mnem = "MOVV" + } + src, dst := ops[0], ops[1] + + // Immediate → register. + if isImmOperand(src) && !isMemOperand(src) { + if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { + rd := l64Reg(dst) + if rd < 0 { + return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem) + } + return encodeLOONG64SBAddr(src.Imm.Sym, rd, mnem, relocs), nil + } + rd := l64Reg(dst) + if rd < 0 { + return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) + } + // MOVF/MOVD $imm, Fd → materialise in R30, then movgr2fr.{w,d}. + if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP { + return encodeLOONG64ImmToFp(rd, l64Imm64(src), mnem), nil + } + return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil + } + + // Static symbol load/store via pcalau12i. + if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) { + rd := l64Reg(dst) + if rd < 0 { + return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem) + } + return encodeLOONG64SBLoad(src.Addr.Sym, rd, mnem, relocs), nil + } + if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) { + rs := l64Reg(src) + if rs < 0 { + return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem) + } + return encodeLOONG64SBStore(dst.Addr.Sym, rs, mnem, relocs), nil + } + + // Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk). + if src.Addr.Index != "" && !isMemOperand(dst) { + rd := l64Reg(dst) + rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index) + if rd < 0 || rj < 0 || rk < 0 { + return nil, fmt.Errorf("%s (rj)(rk): invalid register operand", mnem) + } + op, ok := l64IndexedTable[mnem] + if !ok { + return nil, fmt.Errorf("%s: no register-indexed form", mnem) + } + return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil + } + if dst.Addr.Index != "" && !isMemOperand(src) { + rs := l64Reg(src) + rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index) + if rs < 0 || rj < 0 || rk < 0 { + return nil, fmt.Errorf("%s rd, (rj)(rk): invalid register operand", mnem) + } + op, ok := l64IndexedTable[mnem] + if !ok { + return nil, fmt.Errorf("%s: no register-indexed form", mnem) + } + return l64wordLE(l64rrr(op.st, rk, rj, rs)), nil + } + + // Memory load/store with a 12-bit (or larger, via expansion) offset. + if isMemOperand(src) && !isMemOperand(dst) { + rd := l64Reg(dst) + if rd < 0 { + return nil, fmt.Errorf("%s: invalid destination register", mnem) + } + return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi, relocs) + } + if !isMemOperand(src) && isMemOperand(dst) { + rs := l64Reg(src) + if rs < 0 { + return nil, fmt.Errorf("%s: invalid source register", mnem) + } + return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi, relocs) + } + + // Register → register. + return encodeLOONG64RegMove(mnem, src, dst) +} + +// loong64MovSize returns the encoded size of a MOV instruction. +func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int { + if mnem == "MOV" { + mnem = "MOVV" + } + if len(ops) != 2 { + return 4 + } + src, dst := ops[0], ops[1] + switch { + case isImmOperand(src): + if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { + return 8 // pcalau12i + addi.d + } + if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP { + return 8 // addi/ori r30 + movgr2fr + } + v := l64Imm64(src) + if v == 0 { + return 4 + } + if v > 0 && v <= 0xfff { + return 4 // ori rd, r0, v + } + if v >= -2048 && v < 0 { + return 4 // addi.d rd, r0, v + } + if v == int64(int32(v)) { + if v&0xfff == 0 { + return 4 // lu12i.w + } + return 8 // lu12i.w + ori + } + return 4 * len(l64DconMovWords(0, v)) + case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": + return 8 // pcalau12i + ld + case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": + return 8 // pcalau12i + st + case src.Addr.Index != "" || dst.Addr.Index != "": + return 4 // ldx/stx + case isMemOperand(src) || isMemOperand(dst): + // A 12-bit offset fits in one instruction; larger offsets expand + // to lu12i.w + add.d + the access. + mem := src + if !isMemOperand(src) { + mem = dst + } + if l64MemOffset(mem, fi) >= -2048 && l64MemOffset(mem, fi) < 2048 { + return 4 + } + return 12 + default: + return 4 // register move + } +} + +// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it to +// an F register (the toolchain's case 34: movgr2fr.w/movgr2fr.d). +func encodeLOONG64ImmToFp(fd int, v int64, mnem string) []byte { + // ori for positive constants, addi.d for zero/negative. + op := uint32(0x00b << 22) + if v > 0 { + op = 0x00e << 22 + } + mov := uint32(0x452a << 10) // movgr2fr.d + if mnem == "MOVF" { + mov = 0x4529 << 10 // movgr2fr.w + } + return l64WordsLE( + l64irr(op, int(v), 0, 30), + l64rr(mov, 30, fd), + ) +} + +// ---- 64-bit immediate classification ---- + +// The dcon classes classify a 64-bit constant by which of the four +// materialisation instructions (lu12i.w, ori, lu32i.d, lu52i.d) can be +// dropped, mirroring the toolchain's dconClass: a field is ALL1/ALL0 when +// it is all ones/zeros (fillable by sign/zero extension) or ST1/ST0 when it +// starts with a 1/0 but is mixed. +const ( + l64All1 = iota + l64All0 + l64St1 + l64St0 + + l64Dcon12_0 + l64Dcon12_20S + l64Dcon20S_20 + l64Dcon12_12S + l64Dcon20S_12S + l64Dcon20S_0 + l64Dcon12_12U + l64Dcon20S_12U + l64Dcon32_12S + l64Dcon32_0 + l64Dcon32_20 + l64Dcon12_32S + l64Dcon20S_32 + l64Dcon32_12U + l64Dcon +) + +// l64BitField classifies the bit field of v at [suf+len-1 : suf]. +func l64BitField(v int64, suf, ln int8) int { + var mask1, mask2 uint64 + if ln == 12 { + if suf == 0 { + mask1, mask2 = 0xfff, 0x800 + } else { + mask1, mask2 = 0xfff0000000000000, 0x8000000000000000 + } + } else { + if suf == 12 { + mask1, mask2 = 0xfffff000, 0x80000000 + } else { + mask1, mask2 = 0xfffff00000000, 0x8000000000000 + } + } + u := uint64(v) + switch { + case u&mask1 == mask1: + return l64All1 + case u&mask1 == 0: + return l64All0 + case u&mask2 == mask2: + return l64St1 + } + return l64St0 +} + +// l64DconClass returns the materialisation class of a 64-bit constant, +// transcribed from cmd/internal/obj/loong64's dconClass. +func l64DconClass(v int64) int { + tzb := bits.TrailingZeros64(uint64(v)) + hi12 := l64BitField(v, 52, 12) + hi20 := l64BitField(v, 32, 20) + lo20 := l64BitField(v, 12, 20) + lo12 := l64BitField(v, 0, 12) + if tzb >= 52 { + return l64Dcon12_0 + } + if tzb >= 32 { + if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) { + return l64Dcon20S_0 + } + return l64Dcon32_0 + } + if tzb >= 12 { + if lo20 == l64St1 || lo20 == l64All1 { + if hi20 == l64All1 { + return l64Dcon12_20S + } + if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { + return l64Dcon20S_20 + } + return l64Dcon32_20 + } + if hi20 == l64All0 { + return l64Dcon12_20S + } + if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) { + return l64Dcon20S_20 + } + return l64Dcon32_20 + } + if lo12 == l64St1 || lo12 == l64All1 { + if lo20 == l64All1 { + if hi20 == l64All1 { + return l64Dcon12_12S + } + if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { + return l64Dcon20S_12S + } + return l64Dcon32_12S + } + if lo20 == l64St1 { + if hi20 == l64All1 { + return l64Dcon12_32S + } + if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { + return l64Dcon20S_32 + } + return l64Dcon + } + if lo20 == l64All0 { + if hi20 == l64All0 { + return l64Dcon12_12U + } + if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { + return l64Dcon20S_12U + } + return l64Dcon32_12U + } + if hi20 == l64All0 { + return l64Dcon12_32S + } + if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { + return l64Dcon20S_32 + } + return l64Dcon + } + if lo20 == l64All0 { + if hi20 == l64All0 { + return l64Dcon12_12U + } + if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { + return l64Dcon20S_12U + } + return l64Dcon32_12U + } + if lo20 == l64St1 || lo20 == l64All1 { + if hi20 == l64All1 { + return l64Dcon12_32S + } + if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { + return l64Dcon20S_32 + } + return l64Dcon + } + if hi20 == l64All0 { + return l64Dcon12_32S + } + if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { + return l64Dcon20S_32 + } + return l64Dcon +} + +// l64DconMovWords returns the materialisation words for a 64-bit constant +// into rd, per the toolchain's case 67/68/69/59 sequences. +func l64DconMovWords(rd int, v int64) []uint32 { + const ( + lu12iw = 0x0a << 25 + lu32id = 0x0b << 25 + lu52id = 0x00c << 22 + addiw = 0x00a << 22 + addid = 0x00b << 22 + ori = 0x00e << 22 + ) + switch l64DconClass(v) { + case l64Dcon12_0: + return []uint32{l64irr(lu52id, int(v>>52), 0, rd)} + case l64Dcon12_20S: + return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)} + case l64Dcon20S_20: + return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)} + case l64Dcon12_12S: + return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)} + case l64Dcon20S_12S, l64Dcon20S_0: + return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)} + case l64Dcon12_12U: + return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)} + case l64Dcon20S_12U: + return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)} + case l64Dcon32_12S, l64Dcon32_0: + return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} + case l64Dcon32_20: + return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} + case l64Dcon12_32S: + return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)} + case l64Dcon20S_32: + return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)} + case l64Dcon32_12U: + return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} + default: + return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} + } +} + +// encodeLOONG64LoadImm loads an immediate into a register, matching the +// toolchain's MOVV/MOVW case 3/19/25/59 expansion: +// +// $0: or rd, r0, r0 (MOVW: sll.w rd, r0, r0) +// 1..0xfff: ori rd, r0, imm +// −2048..−1: addi.d rd, r0, imm +// 32-bit (low 12 zero): lu12i.w rd, imm>>12 +// 32-bit: lu12i.w rd, imm>>12; ori rd, rd, imm +// 64-bit: lu12i.w rd, imm>>12; ori rd, rd, imm; +// lu32i.d rd, imm>>32; lu52i.d rd, rd, imm>>52 +func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte { + if v == 0 { + // The zero constant matches the register-form optab entry: MOVV → + // or rd, r0, r0, MOVW → sll.w rd, r0, r0. + op := l64movRegTable["MOVV"].op + if mnem == "MOVW" { + op = l64movRegTable["MOVW"].op + } + return l64wordLE(l64rrr(op, 0, 0, rd)) + } + if v > 0 && v <= 0xfff { + return l64wordLE(l64irr(l64DualTable["OR"].imm, int(v), 0, rd)) + } + if v >= -2048 && v < 0 { + // Both MOVV and MOVW use addi.d for negative constants. + return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(v), 0, rd)) + } + if v == int64(int32(v)) { + if v&0xfff == 0 { + return l64wordLE(l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd)) + } + return l64WordsLE( + l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd), + l64irr(l64DualTable["OR"].imm, int(v), rd, rd), + ) + } + // 64-bit constants use the shortest materialisation the bit pattern + // admits (dcon classification). + return l64WordsLE(l64DconMovWords(rd, v)...) +} + +// encodeLOONG64MemOp encodes a memory load (load = true) or store with a +// 12-bit offset, or the 3-instruction expansion for larger offsets: +// lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30). +func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) { + rj, off := l64MemWithFrame(mem, fi) + if rj < 0 { + return nil, fmt.Errorf("invalid memory operand") + } + ls, ok := l64loadStoreTable[mnem] + if !ok { + return nil, fmt.Errorf("unsupported MOV width %q", mnem) + } + op := ls.st + if load { + op = ls.ld + } + if off >= -2048 && off < 2048 { + return l64wordLE(l64irr(op, int(off), rj, reg)), nil + } + // Large offset: materialise the base in R30 (the assembler temp). + return l64WordsLE( + l64ir(l64InstrTable["LU12IW"].op, int((off+0x800)>>12), 30), + l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30), + l64irr(op, int(off), 30, reg), + ), nil +} + +// encodeLOONG64RegMove encodes a register-to-register move: the width +// extensions (ext.w.b, ext.w.h, sll.w, or, andi, bstrpick.d) between GPRs, +// fmov between F registers, and the special moves across the GPR/FP/FCC/FCSR +// banks. +func encodeLOONG64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { + rs, rd := l64Reg(src), l64Reg(dst) + if rs < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand") + } + sc, dc := loong64RegClass(operandRegName(src)), loong64RegClass(operandRegName(dst)) + + // FP-bank specials (MOVV/MOVW between GPR/FCC/FCSR and F registers). + if key, ok := l64FpMoveKey(mnem, sc, dc); ok { + op, ok := l64FpMovTable[key] + if !ok { + return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst)) + } + return l64wordLE(l64rr(op, rs, rd)), nil + } + + // GPR → GPR. + if sc == l64ClsGR && dc == l64ClsGR { + switch mnem { + case "MOVHU": + // bstrpick.d rd, rj, $15, $0 + return l64wordLE(l64irir(0x3<<22, 15, rs, 0, rd)), nil + case "MOVWU": + // bstrpick.d rd, rj, $31, $0 + return l64wordLE(l64irir(0x3<<22, 31, rs, 0, rd)), nil + } + if e, ok := l64movRegTable[mnem]; ok { + if e.rr { + return l64wordLE(l64rr(e.op, rs, rd)), nil + } + if e.imm != 0 { + return l64wordLE(l64irr(e.op, e.imm, rs, rd)), nil + } + // 3R with rk = r0: or rd, rj, r0 / sll.w rd, rj, r0. + return l64wordLE(l64rrr(e.op, 0, rs, rd)), nil + } + } + + // F → F. + if sc == l64ClsFP && dc == l64ClsFP { + if op, ok := l64movFpRegTable[mnem]; ok { + return l64wordLE(l64rr(op, rs, rd)), nil + } + } + return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst)) +} + +// l64FpMoveKey builds the l64FpMovTable key for a cross-bank move, reporting +// whether the move is a cross-bank special at all. +func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) { + bank := func(c l64RegClass) string { + switch c { + case l64ClsFP: + return "F" + case l64ClsFCC: + return "FCC" + case l64ClsFCSR: + return "FCSR" + default: + return "R" + } + } + if sc == dc { + return "", false + } + if mnem != "MOVV" && mnem != "MOVW" { + return "", false + } + key := mnem + "." + bank(sc) + "." + bank(dc) + _, ok := l64FpMovTable[key] + return key, ok +} + +// ---- static symbol references (pcalau12i + offset) ---- + +// encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the +// R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address. +func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte { + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset}, + ) + } + return l64WordsLE( + l64ir(l64InstrTable["PCALAU12I"].op, 0, rd), + l64irr(l64DualTable["ADDV"].imm, 0, rd, rd), + ) +} + +// encodeLOONG64SBLoad emits pcalau12i r30, 0; ld rd, 0(r30) with the +// R_LOONG64_ADDR_HI/LO pair, loading from a static symbol. +func encodeLOONG64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte { + ls := l64loadStoreTable[mnem] + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset}, + ) + } + return l64WordsLE( + l64ir(l64InstrTable["PCALAU12I"].op, 0, 30), + l64irr(ls.ld, 0, 30, rd), + ) +} + +// encodeLOONG64SBStore emits pcalau12i r30, 0; st rd, 0(r30) with the +// R_LOONG64_ADDR_HI/LO pair, storing to a static symbol. +func encodeLOONG64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) []byte { + ls := l64loadStoreTable[mnem] + if relocs != nil { + *relocs = append(*relocs, + Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, + Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset}, + ) + } + return l64WordsLE( + l64ir(l64InstrTable["PCALAU12I"].op, 0, 30), + l64irr(ls.st, 0, 30, rs), + ) +} + +// ---- operand helpers ---- + +// l64IndexedTable holds the register-indexed load/store (ldx/stx) opcodes. +var l64IndexedTable = map[string]struct{ ld, st uint32 }{ + "MOVB": {0x07000 << 15, 0x07020 << 15}, + "MOVH": {0x07008 << 15, 0x07028 << 15}, + "MOVW": {0x07010 << 15, 0x07030 << 15}, + "MOVV": {0x07018 << 15, 0x07038 << 15}, + "MOVBU": {0x07040 << 15, 0x07020 << 15}, + "MOVHU": {0x07048 << 15, 0x07028 << 15}, + "MOVWU": {0x07050 << 15, 0x07030 << 15}, + "MOVF": {0x07060 << 15, 0x07070 << 15}, + "MOVD": {0x07068 << 15, 0x07078 << 15}, +} + +// operandRegName returns the register name of an operand, or "". +func operandRegName(op *ast.Operand) string { + if op.Addr.Base != "" { + return op.Addr.Base + } + if op.Addr.Sym != nil && op.Addr.Sym.Name != "" { + return op.Addr.Sym.Name + } + return "" +} + +// l64Reg returns the register number of an operand, or -1. +func l64Reg(op *ast.Operand) int { + return loong64RegNum(operandRegName(op)) +} + +// l64Imm returns the immediate value of an operand. +func l64Imm(op *ast.Operand) int32 { + return immFromOperand(op) +} + +// l64Imm64 returns the full 64-bit immediate value of an operand. +func l64Imm64(op *ast.Operand) int64 { + if op.Imm.HasVal { + v := op.Imm.Val + if op.Imm.Neg { + v = -v + } + return v + } + return 0 +} + +// l64Mem returns the base register and byte offset of a memory operand. +func l64Mem(op *ast.Operand) (rj int, off int32) { + rj = loong64RegNum(op.Addr.Base) + off = int32(op.Addr.Offset) + return +} + +// l64MemWithFrame resolves a memory operand, translating FP/SP pseudo- +// registers via the frame mapping. +func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) { + if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { + return loong64ResolvePseudo(op.Addr.Sym, fi) + } + return l64Mem(op) +} + +// l64MemOffset returns the resolved byte offset of a memory operand. +func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 { + _, off := l64MemWithFrame(op, fi) + return off +} + +// l64Label returns the label name of an operand. +func l64Label(op *ast.Operand) string { + if op.Addr.Sym != nil { + return op.Addr.Sym.Name + } + return op.Raw } diff --git a/asm/loong64_encode.go b/asm/loong64_encode.go new file mode 100644 index 0000000..66a7961 --- /dev/null +++ b/asm/loong64_encode.go @@ -0,0 +1,595 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +// loong64 (LoongArch) instruction encoding. +// +// The encoder is data-driven: each mnemonic maps to an instruction format and +// an opcode constant, and the format selects the bit layout. The opcode +// constants and formats are transcribed from the Go toolchain's own loong64 +// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm` +// exactly — the ground-truth oracle for the verify suite. +// +// All LoongArch instructions are 32 bits, little-endian. The formats used +// here (per the LoongArch Volume I specification): +// +// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0] +// 2R opcode[31:15] | rj[4:0] | rd[4:0] +// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0] +// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0] +// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0] +// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0] +// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z) +// B/BL opcode[31:26] | offs[25:0] +// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0] +// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0] +// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0] +// +// The opcode constants are pre-positioned (they include the zero bit ranges +// of the immediate and register fields), mirroring the toolchain's OP_* +// helpers, so each l64* function only ORs its fields in. + +// loong64RegNum returns the 5-bit register number for a LoongArch register +// name: R0–R31 (integer), F0–F31 (floating point), FCC0–FCC7 (condition +// flags), FCSR0–FCSR31 (control/status) and the ABI aliases the runtime's +// assembly uses. Returns -1 for an unrecognised name. +func loong64RegNum(name string) int { + switch name { + case "R0", "ZERO": + return 0 + case "R1", "RA", "LINK": + return 1 + case "R2", "TP": + return 2 + case "R3", "SP": + return 3 + case "R4", "A0": + return 4 + case "R5", "A1": + return 5 + case "R6", "A2": + return 6 + case "R7", "A3": + return 7 + case "R8", "A4": + return 8 + case "R9", "A5": + return 9 + case "R10", "A6": + return 10 + case "R11", "A7": + return 11 + case "R12", "T0": + return 12 + case "R13", "T1": + return 13 + case "R14", "T2": + return 14 + case "R15", "T3": + return 15 + case "R16", "T4": + return 16 + case "R17", "T5": + return 17 + case "R18", "T6": + return 18 + case "R19", "T7": + return 19 + case "R20", "T8": + return 20 + case "R21": + return 21 + case "R22", "G", "g", "FP": + return 22 + case "R23", "S0": + return 23 + case "R24", "S1": + return 24 + case "R25", "S2": + return 25 + case "R26", "S3": + return 26 + case "R27", "S4": + return 27 + case "R28", "S5": + return 28 + case "R29", "S6", "CTXT": + return 29 + case "R30", "S7", "TMP": + return 30 + case "R31", "S8": + return 31 + } + // F0–F31, FCC0–FCC7, FCSR0–FCSR31. + if len(name) >= 4 && name[:4] == "FCSR" { + return loong64RegSpecial(name[4:], "FCSR", 31) + } + if len(name) >= 3 && name[:3] == "FCC" { + return loong64RegSpecial(name[3:], "FCC", 7) + } + if len(name) < 2 { + return -1 + } + prefix, digits := name[:1], name[1:] + if digits[0] < '0' || digits[0] > '9' { + return -1 + } + n := 0 + for i := 0; i < len(digits); i++ { + if digits[i] < '0' || digits[i] > '9' { + return -1 + } + n = n*10 + int(digits[i]-'0') + } + if prefix == "F" && n <= 31 { + return n + } + return -1 +} + +// loong64RegSpecial parses a numbered FCC/FCSR register. +func loong64RegSpecial(digits, prefix string, max int) int { + if digits == "" { + return -1 + } + n := 0 + for i := 0; i < len(digits); i++ { + if digits[i] < '0' || digits[i] > '9' { + return -1 + } + n = n*10 + int(digits[i]-'0') + } + if n <= max { + return n + } + return -1 +} + +// ---- format helpers ---- + +// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd. +func l64rrr(op uint32, rk, rj, rd int) uint32 { + return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) +} + +// l64rr encodes a 2R instruction: op | rj<<5 | rd. +func l64rr(op uint32, rj, rd int) uint32 { + return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) +} + +// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd. +func l64irr(op uint32, imm, rj, rd int) uint32 { + return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) +} + +// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd. +func l64irr14(op uint32, imm, rj, rd int) uint32 { + return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) +} + +// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd. +func l64irr16(op uint32, imm, rj, rd int) uint32 { + return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) +} + +// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd. +func l64ir(op uint32, imm, rd int) uint32 { + return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f) +} + +// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the +// 4-byte-aligned word distance (the toolchain stores the shifted value). +func l64bbl(op uint32, offs int) uint32 { + return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF +} + +// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF): +// op | si21[15:0]<<10 | rj<<5 | si21[20:16]. +func l64ir21(op uint32, offs, rj int) uint32 { + v := uint32(offs) + return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F +} + +// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4. +func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 { + return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f) +} + +// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd. +// The msb/lsb fields are 6 bits wide (0–63) and are validated by the caller. +func l64irir(op uint32, msb, rj, lsb, rd int) uint32 { + return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f) +} + +// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd. +func l64irrr(op uint32, sa, rk, rj, rd int) uint32 { + return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) +} + +// l64i15 encodes a no-operand system instruction with a 15-bit code field +// (SYSCALL, BREAK, DBAR): op | code[14:0]. +func l64i15(op uint32, code int) uint32 { + return op | uint32(code)&0x7FFF +} + +// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint. +func l64irr5i(op uint32, offs, rj, hint int) uint32 { + return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f) +} + +// l64wordLE encodes a uint32 as 4 little-endian bytes. +func l64wordLE(w uint32) []byte { + return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} +} + +// l64WordsLE concatenates one or more instruction words as little-endian bytes. +func l64WordsLE(ws ...uint32) []byte { + var out []byte + for _, w := range ws { + out = append(out, l64wordLE(w)...) + } + return out +} + +// ---- instruction formats ---- + +type l64Format uint8 + +const ( + l64Frrr l64Format = iota // 3R (integer and FP arithmetic) + l64Frr // 2R + l64Firr // 2RI12 (arithmetic with 12-bit immediate) + l64Firr14 // 2RI14 (ldptr/stptr) + l64Firr16 // 2RI16 (addu16i.d) + l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i) + l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub) + l64Firir // bstrins/bstrpick + l64Firrr // alsl + l64Fi15 // syscall/break/dbar + l64Fam // atomic (3R with the AM field order) + l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0]) + l64Fshift // 2RI12 with a 5/6-bit shift immediate + l64Fpreld // preld (2RI12 + 5-bit hint) +) + +// l64Enc is one instruction's encoding: its bit layout (format) and the +// opcode constant, positioned at its exact bit range. +type l64Enc struct { + format l64Format + op uint32 +} + +// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form +// and the 2RI12 immediate form (which is a shift for the shift mnemonics). +type l64DualEnc struct { + rrr uint32 // 3R register form + imm uint32 // 2RI12 immediate form + shift bool // the immediate form is a 5/6-bit shift amount +} + +// l64DualTable maps the dual-form arithmetic/logic mnemonics to both +// encodings; the assembler picks by operand kind. +var l64DualTable = map[string]l64DualEnc{} + +// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them) +// to their encoding. SIMD (LSX/LASX: V*/XV*) instructions are not covered +// yet; the base integer, memory and floating-point ISA is complete. +var l64InstrTable = map[string]l64Enc{} + +func init() { + // 3R — integer. + rrr := map[string]uint32{ + "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, + "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, + "SGT": 0x24 << 15, "SGTU": 0x25 << 15, + "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15, + "NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15, + "ORN": 0x2c << 15, "ANDN": 0x2d << 15, + "SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15, + "SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15, + "ROTR": 0x36 << 15, "ROTRV": 0x37 << 15, + "MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15, + "MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15, + "MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15, + "DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15, + "DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15, + "DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15, + "CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15, + "CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15, + } + // 3R — floating point. + rrr["MULF"] = 0x209 << 15 + rrr["MULD"] = 0x20a << 15 + rrr["DIVF"] = 0x20d << 15 + rrr["DIVD"] = 0x20e << 15 + rrr["SUBF"] = 0x205 << 15 + rrr["SUBD"] = 0x206 << 15 + rrr["ADDF"] = 0x201 << 15 + rrr["ADDD"] = 0x202 << 15 + rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15 + rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15 + rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15 + rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15 + rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15 + rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15 + rrr["FMINF"] = 0x215 << 15 + rrr["FMIND"] = 0x216 << 15 + rrr["FMAXF"] = 0x211 << 15 + rrr["FMAXD"] = 0x212 << 15 + rrr["FMAXAF"] = 0x219 << 15 + rrr["FMAXAD"] = 0x21a << 15 + rrr["FMINAF"] = 0x21d << 15 + rrr["FMINAD"] = 0x21e << 15 + rrr["FSCALEBF"] = 0x221 << 15 + rrr["FSCALEBD"] = 0x222 << 15 + rrr["FCOPYSGF"] = 0x225 << 15 + rrr["FCOPYSGD"] = 0x226 << 15 + for m, op := range rrr { + l64InstrTable[m] = l64Enc{format: l64Frrr, op: op} + } + + // 2R. + rr := map[string]uint32{ + "CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10, + "CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10, + "REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10, + "REVH2W": 0x10 << 10, "REVHV": 0x11 << 10, + "BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10, + "EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10, + "TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10, + "MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10, + "MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10, + "FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10, + "MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10, + "ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10, + "MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10, + "NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10, + "SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10, + "FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10, + "FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10, + "FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10, + "FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10, + "FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10, + "FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10, + "FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10, + "FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10, + "FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10, + "FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10, + } + for m, op := range rr { + l64InstrTable[m] = l64Enc{format: l64Frr, op: op} + } + // RDTIME is a 2R instruction with rd and rj in swapped positions. + l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10} + l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10} + l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10} + + // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), + // selected by the operand kind; the shift mnemonics pair the 3R form + // with a 5/6-bit shift immediate. + for m, e := range map[string]l64DualEnc{ + "ADD": {rrr: 0x20 << 15, imm: 0x00a << 22}, + "ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22}, + "ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22}, + "ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22}, + "AND": {rrr: 0x29 << 15, imm: 0x00d << 22}, + "OR": {rrr: 0x2a << 15, imm: 0x00e << 22}, + "XOR": {rrr: 0x2b << 15, imm: 0x00f << 22}, + "SGT": {rrr: 0x24 << 15, imm: 0x008 << 22}, + "SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22}, + "SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true}, + "SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true}, + "SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true}, + "ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true}, + "SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true}, + "SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true}, + "SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true}, + "ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true}, + } { + l64DualTable[m] = e + } + + // 2RI12 — pure immediate arithmetic (LU52ID has no register form). + l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22} + // ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16. + l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26} + + // 2RI14 — LL/SC are aliased by the Go assembler to the pointer loads and + // stores (ldptr/stptr), with the offset scaled by 4. + l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w + l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d + l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w + l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w + l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d + l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w) + l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w) + l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d) + + // 2RI20. + l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25} + l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25} + l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25} + l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25} + // LUI is the Plan 9 spelling of lu12i.w. + l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} + + // 4R — fused multiply-add. + rrrr := map[string]uint32{ + "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, + "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, + "FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20, + "FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20, + } + for m, op := range rrrr { + l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} + } + + // IRIR — bit-field insert/extract. + irir := map[string]uint32{ + "BSTRINSW": 0x3<<21 | 0x0<<15, + "BSTRINSV": 0x2 << 22, + "BSTRPICKW": 0x3<<21 | 0x1<<15, + "BSTRPICKV": 0x3 << 22, + } + for m, op := range irir { + l64InstrTable[m] = l64Enc{format: l64Firir, op: op} + } + + // 3RI2 — ALSL. + irrr := map[string]uint32{ + "ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17, + } + for m, op := range irrr { + l64InstrTable[m] = l64Enc{format: l64Firrr, op: op} + } + + // 0-operand system instructions. + l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15} + l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15} + l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15} + + // PRELD. + l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} + + // Atomics — 3R with the AM field order (rk=value, rj=address, rd=result). + am := map[string]uint32{ + "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, + "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, + "AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15, + "AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15, + "AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15, + "AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15, + "AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15, + "AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15, + "AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15, + "AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15, + "AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15, + "AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15, + "AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15, + "AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15, + "AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15, + "AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15, + } + for m, op := range am { + l64InstrTable[m] = l64Enc{format: l64Fam, op: op} + } +} + +// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the +// register move between the integer and floating-point register banks — the +// MOVW/MOVV specials the Go assembler accepts. +var l64FpMovTable = map[string]uint32{ + "MOVV.R.F": 0x452a << 10, // movgr2fr.d + "MOVV.R.FCC": 0x4536 << 10, // movgr2cf + "MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr + "MOVV.F.R": 0x452e << 10, // movfr2gr.d + "MOVV.F.FCC": 0x4534 << 10, // movfr2cf + "MOVV.FCC.R": 0x4537 << 10, // movcf2gr + "MOVV.FCC.F": 0x4535 << 10, // movcf2fr + "MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr + "MOVW.R.F": 0x4529 << 10, // movgr2fr.w + "MOVW.F.R": 0x452d << 10, // movfr2gr.s +} + +// l64branchTable holds the 16-bit branch and jump encodings (2RI16). +var l64branchTable = map[string]uint32{ + "BEQ": 0x16 << 26, + "BNE": 0x17 << 26, + "BLT": 0x18 << 26, + "BGE": 0x19 << 26, + "BLTU": 0x1a << 26, + "BGEU": 0x1b << 26, + "JIRL": 0x13 << 26, +} + +// l64branch21Table holds the single-register branches with 21-bit offsets: +// the negative opcode constants the toolchain uses for the short forms. +var l64branch21Table = map[string]uint32{ + "BEQZ": 0x10 << 26, // beq r0, rj → beqz + "BNEZ": 0x11 << 26, // bne r0, rj → bnez + "BLTZ": 0x18 << 26, // blt rj, r0 → bltz + "BGEZ": 0x19 << 26, // bge rj, r0 → bgez + "BGTZ": 0x18 << 26, // blt r0, rj → bgtz + "BLEZ": 0x19 << 26, // bge r0, rj → blez + "BFPT": 0x12<<26 | 0x1<<8, + "BFPF": 0x12<<26 | 0x0<<8, +} + +// l64jumpTable maps the jump pseudo-instructions and their aliases to the +// B/BL opcode constants. +var l64jumpTable = map[string]uint32{ + "JMP": 0x14 << 26, // b + "B": 0x14 << 26, // b + "JAL": 0x15 << 26, // bl + "CALL": 0x15 << 26, // bl + "BL": 0x15 << 26, // bl +} + +// l64loadStoreTable maps the MOV width mnemonics to their load and store +// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as +// the toolchain derives it. +var l64loadStoreTable = map[string]struct{ ld, st uint32 }{ + "MOVB": {0x0a0 << 22, 0x0a4 << 22}, + "MOVH": {0x0a1 << 22, 0x0a5 << 22}, + "MOVW": {0x0a2 << 22, 0x0a6 << 22}, + "MOVV": {0x0a3 << 22, 0x0a7 << 22}, + "MOVBU": {0x0a8 << 22, 0x0a4 << 22}, + "MOVHU": {0x0a9 << 22, 0x0a5 << 22}, + "MOVWU": {0x0aa << 22, 0x0a6 << 22}, + "MOVF": {0x0ac << 22, 0x0ad << 22}, + "MOVD": {0x0ae << 22, 0x0af << 22}, +} + +// l64movRegTable maps a register-to-register MOV mnemonic to its expansion, +// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h, +// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d +// and are handled separately in the assembler. +type l64MovRegEnc struct { + rr bool // 2R format (ext.w.b/ext.w.h) + op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode + imm int // 2RI12 immediate for MOVBU's andi +} + +var l64movRegTable = map[string]l64MovRegEnc{ + "MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj + "MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj + "MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0 + "MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0 + "MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff +} + +// l64movFpRegTable maps a floating-point register move mnemonic to its 2R +// opcode (fmov.s / fmov.d), used when both operands are F registers. +var l64movFpRegTable = map[string]uint32{ + "MOVF": 0x4525 << 10, + "MOVD": 0x4526 << 10, +} + +// l64RegClass discriminates integer (R), floating-point (F) and condition +// (FCC) registers for the MOV pseudo-instruction's register-move encoding. +type l64RegClass int + +const ( + l64ClsNone l64RegClass = iota + l64ClsGR + l64ClsFP + l64ClsFCC + l64ClsFCSR +) + +// loong64RegClass reports the register class of a register operand name. +func loong64RegClass(name string) l64RegClass { + switch { + case name == "": + return l64ClsNone + case len(name) >= 3 && name[:3] == "FCC": + return l64ClsFCC + case len(name) >= 4 && name[:4] == "FCSR": + return l64ClsFCSR + case name[0] == 'F': + return l64ClsFP + default: + return l64ClsGR + } +} diff --git a/asm/loong64_encode_test.go b/asm/loong64_encode_test.go new file mode 100644 index 0000000..43668c6 --- /dev/null +++ b/asm/loong64_encode_test.go @@ -0,0 +1,293 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "encoding/binary" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/ast" + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// firstTextLOONG64 parses assembly source and returns the first TEXT body. +func firstTextLOONG64(t *testing.T, src string) *ast.Text { + t.Helper() + f, errs := parser.Parse("f_loong64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + for _, d := range f.Decls { + if fn, ok := d.(*ast.Text); ok { + return fn + } + } + t.Fatal("no TEXT found") + return nil +} + +// assembleLOONG64Helper assembles one TEXT function and returns its bytes. +func assembleLOONG64Helper(t *testing.T, fn *ast.Text) []byte { + t.Helper() + code, _, _, _, _, err := assembleLOONG64(fn) + if err != nil { + t.Fatalf("assemble: %v", err) + } + return code +} + +// wantWords checks that code matches the expected little-endian words. +func wantWords(t *testing.T, code []byte, want ...uint32) { + t.Helper() + got := make([]uint32, 0, len(code)/4) + for i := 0; i+4 <= len(code); i += 4 { + got = append(got, binary.LittleEndian.Uint32(code[i:])) + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d\ncode: % x", len(got), len(want), code) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +func TestLOONG64_add(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·add(SB), NOSPLIT, $0-24 + MOVV a+0(FP), R4 + MOVV b+8(FP), R5 + ADDV R5, R4, R4 + MOVV R4, ret+16(FP) + RET +`) + code := assembleLOONG64Helper(t, fn) + // 5 instructions: two ld.d, add.d, st.d, jirl r0, r1, 0. + wantWords(t, code, + 0x28C02064, // ld.d r4, 8(r3) + 0x28C04065, // ld.d r5, 16(r3) + 0x00109484, // add.d r4, r4, r5 + 0x29C06064, // st.d r4, 24(r3) + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_arithmetic(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·arith(SB), NOSPLIT, $0 + ADDV R4, R5, R6 + SUBV R7, R8, R9 + MULV R10, R11, R12 + DIVV R13, R14, R15 + AND R16, R17, R18 + OR R18, R19, R20 + XOR R20, R21, R2 + SLLV R2, R23, R24 + SRLV R24, R25, R26 + SRAV R26, R27, R28 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x001090A6, // add.d r6, r5, r4 + 0x00119D09, // sub.d r9, r8, r7 + 0x001DA96C, // mul.d r12, r11, r10 + 0x002235CF, // div.d r15, r14, r13 + 0x0014C232, // and r18, r17, r16 + 0x00154A74, // or r20, r19, r18 + 0x0015D2A2, // xor r2, r21, r20 + 0x00188AF8, // sll.d r24, r23, r2 + 0x0019633A, // srl.d r26, r25, r24 + 0x0019EB7C, // sra.d r28, r27, r26 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_immediates(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·imm(SB), NOSPLIT, $0 + ADDV $42, R4, R5 + ADDV $-8, R6 + AND $0xff, R7, R8 + OR $1, R9, R10 + SGT $100, R13, R14 + SLLV $4, R15, R16 + MOVV $0x12345, R17 + MOVV $0, R18 + MOVW $0, R19 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x02C0A885, // addi.d r5, r4, 42 + 0x02FFE0C6, // addi.d r6, r6, -8 + 0x0343FCE8, // andi r8, r7, 0xff + 0x0380052A, // ori r10, r9, 1 + 0x020191AE, // slti r14, r13, 100 + 0x004111F0, // slli.d r16, r15, 4 + 0x14000251, // lu12i.w r17, 0x12 + 0x038D1631, // ori r17, r17, 0x345 + 0x00150012, // or r18, r0, r0 + 0x00170013, // sll.w r19, r0, r0 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_loadStore(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·mem(SB), NOSPLIT, $0 + MOVV (R4), R5 + MOVV R5, (R6) + MOVW 8(R7), R8 + MOVB R9, -4(R10) + MOVV (R11)(R12), R13 + MOVV R14, (R15)(R16) + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x28C00085, // ld.d r5, 0(r4) + 0x29C000C5, // st.d r5, 0(r6) + 0x288020E8, // ld.w r8, 8(r7) + 0x293FF149, // st.b r9, -4(r10) + 0x380C316D, // ldx.d r13, r11, r12 + 0x381C41EE, // stx.d r14, r15, r16 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_branches(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·br(SB), NOSPLIT, $0 + BEQ R4, R5, done + BNE R6, R7, skip + BLT R8, R9, done + BGE R10, R11, done + BLTU R12, R13, done + BGEU R14, R15, done +skip: + JMP done +done: + RET +`) + code := assembleLOONG64Helper(t, fn) + // skip is at 0x18 (6 words), done at 0x1c. + wantWords(t, code, + 0x58001C85, // beq r5, r4, +7 + 0x5C0018C7, // bne r7, r6, +6 + 0x60001509, // blt r9, r8, +5 + 0x6400114B, // bge r11, r10, +4 + 0x68000D8D, // bltu r13, r12, +3 + 0x6C0009CF, // bgeu r15, r14, +2 + 0x50000400, // b done (+1, chain-folded through skip) + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_frame(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·f(SB), NOSPLIT, $32-8 + MOVV R4, R5 + MOVV arg+0(FP), R6 + MOVV R7, local-8(SP) + MOVV local-8(SP), R8 + MOVV R9, ret+0(FP) + RET +`) + code := assembleLOONG64Helper(t, fn) + // autosize = align8(32+8) = 40; prologue stores LR at -40(SP), + // opens the frame, stores LR again at 0(SP). The function is a leaf + // (no calls), so the epilogue skips the LR restore. FP args are at + // autosize+8; SP locals at autosize+offset. + wantWords(t, code, + 0x29FF6061, // st.d r1, -40(r3) + 0x02FF6063, // addi.d r3, r3, -40 + 0x29C00061, // st.d r1, 0(r3) + 0x00150085, // or r5, r4, r0 + 0x28C0C066, // ld.d r6, 48(r3) arg+0(FP) → 0+40+8 + 0x29C08067, // st.d r7, 32(r3) local-8(SP) → 40-8 + 0x28C08068, // ld.d r8, 32(r3) + 0x29C0C069, // st.d r9, 48(r3) ret+0(FP) → 0+40+8 + 0x02C0A063, // addi.d r3, r3, 40 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_jumpChain(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·jc(SB), NOSPLIT, $0 + JMP a +a: + JMP b +b: + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x50000800, // b +2 (a, chain-folded to b) + 0x50000400, // b +1 (b) + 0x4C000020, // jirl r0, r1, 0 + ) +} + +func TestLOONG64_dconClasses(t *testing.T) { + cases := []struct { + v int64 + word int // expected word count + }{ + {0x123456789, 3}, // lu12i.w + ori + lu32i.d + {-1, 2}, // addi.d + lu52i.d (the MOV path handles -1 earlier) + {0x1000000000000, 2}, // addi.w + lu32i.d + {0x123456789abcdef0, 4}, // full sequence + {0xFFFFFFFFF, 2}, // lu12i.w + ori + {0x1234567800000000, 3}, // addi.w + lu32i.d + lu52i.d + } + for _, c := range cases { + if n := len(l64DconMovWords(0, c.v)); n != c.word { + t.Errorf("0x%x: %d words, want %d", c.v, n, c.word) + } + } +} + +func TestLOONG64_regNames(t *testing.T) { + cases := map[string]int{ + "R0": 0, "R31": 31, "F0": 0, "F31": 31, "FCC0": 0, "FCC7": 7, + "FCSR0": 0, "FCSR3": 3, "ZERO": 0, "RA": 1, "SP": 3, "g": 22, "G": 22, + "R32": -1, "FCC8": -1, "X0": -1, "R": -1, "TMP": 30, "CTXT": 29, + } + for name, want := range cases { + if got := loong64RegNum(name); got != want { + t.Errorf("loong64RegNum(%q) = %d, want %d", name, got, want) + } + } +} + +func TestLOONG64_bytesEqualGroundTruth(t *testing.T) { + // A spot-check that assembleLOONG64 emits the same bytes the Go + // toolchain does for a small kernel (the full comparison lives in + // verify's TestGroundTruthLOONG64). + src := `#include "textflag.h" +TEXT ·k(SB), NOSPLIT, $0-0 + ADDV R4, R5, R6 + MOVV $0x100000, R7 + BEQ R6, R7, done + JMP done +done: + RET +` + fn := firstTextLOONG64(t, src) + code := assembleLOONG64Helper(t, fn) + want := []byte{ + 0xa6, 0x90, 0x10, 0x00, // add.d r6, r5, r4 + 0x07, 0x20, 0x00, 0x14, // lu12i.w r7, 0x100 + 0xc7, 0x08, 0x00, 0x58, // beq r7, r6, +2 (done) + 0x00, 0x04, 0x00, 0x50, // b +1 (done) + 0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0 + } + if !bytes.Equal(code, want) { + t.Errorf("code = % x\nwant % x", code, want) + } +} diff --git a/asm/loong64_frame.go b/asm/loong64_frame.go new file mode 100644 index 0000000..2d06d63 --- /dev/null +++ b/asm/loong64_frame.go @@ -0,0 +1,129 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "strings" + + "sourcedock.dev/petrbalvin/gasm-devkit/ast" +) + +// Loong64 frame mapping, matching the Go toolchain's loong64 backend. +// +// Go's loong64 functions have no frame pointer: FP and SP are synthetic +// registers resolved against the hardware stack pointer (R3) and the frame +// size. The return address lives in R1 (the link register). +// +// The autosize is the real stack adjustment: the declared local frame plus +// the 8 bytes for the saved link register, rounded up to a multiple of 8 +// (the toolchain aligns frames with `if autosize&4 != 0 { autosize += 4 }`). +// A leaf function (no calls) with a zero frame gets no prologue at all. +// +// Prologue (autosize > 0), byte-identical to the toolchain: +// +// MOVV R1, -autosize(R3) // save LR below the new SP (traceback-safe) +// ADDV $-autosize, R3 // open the frame +// MOVV R1, 0(R3) // save LR again at SP (signal-safety) +// +// Epilogue: MOVV 0(R3), R1; ADDV $autosize, R3 (non-leaf only for the LR +// restore); the RET's jirl r0, r1, 0 follows. + +// loong64FrameInfo holds the frame layout derived from a TEXT directive. +type loong64FrameInfo struct { + autosize int // the real SP adjustment (locals + saved LR, aligned) + frame int // the declared $framesize + args int // the declared -argsize + noSplit bool // the NOSPLIT flag + leaf bool // no call instructions in the body +} + +// loong64ComputeFrame derives the frame layout for a TEXT function. +func loong64ComputeFrame(t *ast.Text) loong64FrameInfo { + fi := loong64FrameInfo{ + frame: frameSize(t), + args: argsSize(t), + } + for _, f := range t.Flags { + if f == "NOSPLIT" { + fi.noSplit = true + } + } + fi.leaf = loong64IsLeaf(t) + if fi.frame != 0 { + fi.autosize = fi.frame + 8 // space for the saved LR + if fi.autosize&4 != 0 { + fi.autosize += 4 + } + } else if !fi.leaf { + // A zero-frame non-leaf function still opens an 8-byte frame for LR. + fi.autosize = 8 + } + return fi +} + +// loong64IsLeaf reports whether a function contains no call instructions +// (JAL/BL/CALL), matching the toolchain's LEAF mark, which drives the frame +// and the epilogue shape. +func loong64IsLeaf(t *ast.Text) bool { + for _, stmt := range t.Body { + in, ok := stmt.(*ast.Instr) + if !ok { + continue + } + switch strings.ToUpper(in.Mnemonic.Text) { + case "JAL", "CALL", "BL": + return false + } + } + return true +} + +// loong64Prologue returns the prologue bytes for a loong64 function. +func loong64Prologue(fi loong64FrameInfo) []byte { + if fi.autosize == 0 { + return nil + } + addiD := l64DualTable["ADDV"].imm + return l64WordsLE( + l64irr(l64loadStoreTable["MOVV"].st, -fi.autosize, 3, 1), // MOVV R1, -autosize(R3) + l64irr(addiD, -fi.autosize, 3, 3), // ADDV $-autosize, R3 + l64irr(l64loadStoreTable["MOVV"].st, 0, 3, 1), // MOVV R1, 0(R3) + ) +} + +// loong64Return returns the bytes for a RET: the epilogue (restore LR and +// deallocate the frame when present) followed by jirl r0, r1, 0. +func loong64Return(fi loong64FrameInfo) []byte { + var ws []uint32 + if fi.autosize != 0 { + if !fi.leaf { + // MOVV 0(R3), R1 — restore the link register. + ws = append(ws, l64irr(l64loadStoreTable["MOVV"].ld, 0, 3, 1)) + } + // ADDV $autosize, R3 — close the frame. + ws = append(ws, l64irr(l64DualTable["ADDV"].imm, fi.autosize, 3, 3)) + } + // jirl r0, r1, 0 — return. + ws = append(ws, l64irr16(l64branchTable["JIRL"], 0, 1, 0)) + return l64WordsLE(ws...) +} + +// loong64ResolvePseudo translates a pseudo-register memory reference into a +// hardware base register and offset. x+N(FP) → (N + autosize + 8)(SP); +// x-N(SP) → (autosize - N)(SP). Returns base = -1 for an unresolvable +// reference (SB: static data, handled by the relocation path). +func loong64ResolvePseudo(sym *ast.Symbol, fi loong64FrameInfo) (base int, off int32) { + if sym == nil { + return -1, 0 + } + switch sym.Pseudo { + case "FP": + return 3, int32(sym.Offset) + int32(fi.autosize) + 8 + case "SP": + return 3, int32(fi.autosize) + int32(sym.Offset) + case "SB": + return -1, int32(sym.Offset) + } + return -1, 0 +} diff --git a/asm/loong64_more_test.go b/asm/loong64_more_test.go new file mode 100644 index 0000000..12cd9ab --- /dev/null +++ b/asm/loong64_more_test.go @@ -0,0 +1,367 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "bytes" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestLOONG64_sys exercises the no-operand system instructions and the +// bare-data pseudo-instructions. The words match `go tool asm` +// (GOARCH=loong64) for the same source. +func TestLOONG64_sys(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·sys(SB), NOSPLIT, $0 + NOOP + UNDEF + WORD $0x12345678 + SYSCALL $0x10 + BREAK $0x20 + DBAR $1 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x03400000, // andi r0, r0, 0 (NOOP) + 0x002A0000, // break 0 (UNDEF) + 0x12345678, // WORD + 0x002B0010, // syscall 0x10 + 0x002A0020, // break 0x20 + 0x38720001, // dbar 1 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +// TestLOONG64_branches21 exercises the single-register branch forms: the +// 21-bit BEQZ/BNEZ/BLTZ/BGEZ and the rd-field BGTZ/BLEZ. +func TestLOONG64_branches21(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·b21(SB), NOSPLIT, $0 + BEQZ R4, done + BNEZ R5, done + BLTZ R6, done + BGEZ R7, done + BGTZ R8, done + BLEZ R9, done +done: + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x40001880, // beqz r4, +6 + 0x440014A0, // bnez r5, +5 + 0x600010C0, // bltz r6, +4 + 0x64000CE0, // bgez r7, +3 + 0x60000808, // bgtz r8, +2 (register in the rd field) + 0x64000409, // blez r9, +1 + 0x4C000020, // jirl r0, r1, 0 + ) +} + +// TestLOONG64_fma exercises the four fused multiply-add forms (4 and 3 +// operand spellings). +func TestLOONG64_fma(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·fma(SB), NOSPLIT, $0 + FMADDD F0, F1, F2, F3 + FMSUBD F4, F5, F6 + FNMADDD F7, F8, F9, F10 + FNMSUBD F11, F12, F13 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x08200443, // fmadd.d f3, f2, f1, f0 + 0x086214C6, // fmsub.d f6, f5, f5, f4 + 0x08A3A12A, // fnmadd.d f10, f9, f8, f7 + 0x08E5B1AD, // fnmsub.d f13, f12, f12, f11 + 0x4C000020, + ) +} + +// TestLOONG64_bitops exercises BSTRINS/BSTRPICK (the 6-bit msb/lsb fields) +// and ALSL (the sa−1 shift field). +func TestLOONG64_bitops(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·bits(SB), NOSPLIT, $0 + BSTRINSW $3, R4, $0, R5 + BSTRINSV $3, R4, $1, R6 + BSTRPICKW $3, R4, $0, R5 + BSTRPICKV $6, R7, $0, R8 + ALSLW $1, R4, R5, R6 + ALSLW $4, R7, R8, R9 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x00630085, // bstrins.w r5, r4, $3, $0 + 0x00830486, // bstrins.d r6, r4, $3, $1 + 0x00638085, // bstrpick.w r5, r4, $3, $0 + 0x00C600E8, // bstrpick.d r8, r7, $6, $0 + 0x00041486, // alsl.w r6, r5, r4, $1 (sa-1) + 0x0005A0E9, // alsl.w r9, r8, r7, $4 + 0x4C000020, + ) +} + +// TestLOONG64_ptr exercises the 14-bit-offset memory forms (LL/SC/MOVWP/ +// MOVVP with the offset scaled by 4) and PRELD. +func TestLOONG64_ptr(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·ptr(SB), NOSPLIT, $0 + LLW 8(R14), R15 + SCW R16, -4(R17) + MOVWP 16(R18), R19 + MOVVP R20, 24(R21) + PRELD 32(R22), $0 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x200009CF, // ll.w r15, 8(r14) + 0x21FFFE30, // sc.w r16, -4(r17) + 0x24001253, // ldptr.w r19, 16(r18) + 0x27001AB4, // stptr.d r20, 24(r21) + 0x2AC082C0, // preld 32(r22), 0 + 0x4C000020, + ) +} + +// TestLOONG64_atomics exercises the AM* read-modify-write forms and +// RDTIME, plus the MOVV FP→GP move. +func TestLOONG64_atomics(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·atoms(SB), NOSPLIT, $0 + AMADDW R4, (R5), R6 + RDTIMED R7, R8 + MOVV F1, R2 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x386110A6, // amadd.w r6, r5, r4 + 0x000068E8, // rdtime.d r8, r7 + 0x0114B822, // movfr2gr.d r2, f1 + 0x4C000020, + ) +} + +// TestLOONG64_lu52 exercises the LU52I.D immediate form (a gasm extension +// the toolchain reaches only through its MOVV expansion). +func TestLOONG64_lu52(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·lu52(SB), NOSPLIT, $0 + LU52ID $0x345, R10 + LU52ID $0x123, R11, R12 + ADDV16 $0x10000, R13 + RET +`) + code := assembleLOONG64Helper(t, fn) + wantWords(t, code, + 0x030D154A, // lu52i.d r10, r10, 0x345 + 0x03048D6C, // lu52i.d r12, r11, 0x123 + 0x100005AD, // addu16i.d r13, r13, 0x10000>>16 + 0x4C000020, + ) +} + +// TestLOONG64_sbRefs checks the static-symbol reference forms through the +// full file assembly: each pcalau12i+addi.d/ld/st pair carries the +// R_LOONG64_ADDR_HI/LO relocation pair, and the immediate fields are left +// zero for the linker. +func TestLOONG64_sbRefs(t *testing.T) { + f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h" +TEXT ·sb(SB), NOSPLIT, $0 + MOVV $·table(SB), R4 + MOVV ·table+8(SB), R5 + MOVV R6, ·table(SB) + RET + +GLOBL ·table(SB), RODATA, $8 +DATA ·table+0(SB)/8, $42 +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + fn := img.Funcs[0] + if fn.Size != 28 { + t.Fatalf("function size = %d, want 28", fn.Size) + } + var hi, lo int + // The three references: $·table (0), ·table+8 (8), ·table (0). + wantAdd := []int64{0, 0, 8, 8, 0, 0} + for i, r := range fn.Relocs { + wantKind := RelLoong64AddrHi + wantOff := (i / 2) * 8 + if i%2 == 1 { + wantKind = RelLoong64AddrLo + wantOff += 4 + } + if r.Kind != wantKind || r.Off != wantOff || r.Name != "table" || r.Addend != wantAdd[i] { + t.Errorf("reloc %d = {kind %v off %d name %q addend %d}", i, r.Kind, r.Off, r.Name, r.Addend) + } + if r.Kind == RelLoong64AddrHi { + hi++ + } else { + lo++ + } + } + if hi != 3 || lo != 3 { + t.Errorf("relocs = %d hi + %d lo, want 3 + 3", hi, lo) + } + // The image carries the zero-immediate pair encodings (the linker + // fills the immediate fields from the relocations). + code := img.Code[fn.Offset : fn.Offset+fn.Size] + wantWords(t, code, + 0x1A000004, // pcalau12i r4, 0 + 0x02C00084, // addi.d r4, r4, 0 + 0x1A00001E, // pcalau12i r30, 0 + 0x28C003C5, // ld.d r5, 0(r30) + 0x1A00001E, // pcalau12i r30, 0 + 0x29C003C6, // st.d r6, 0(r30) + 0x4C000020, // jirl r0, r1, 0 + ) +} + +// TestLOONG64_errors checks the encoder's error paths: undefined labels, +// invalid register operands and operand-count mismatches. +func TestLOONG64_errors(t *testing.T) { + cases := []string{ + `TEXT ·e(SB), NOSPLIT, $0 + JMP nowhere + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + BEQZ X0, done +done: + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + ADDV R4 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + FMADDD F0, F1 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + AMADDW R4, R5 + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + WORD + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + PRELD 32(R4) + RET +`, + `TEXT ·e(SB), NOSPLIT, $0 + ALSLW $5, R4, R5, R6 + RET +`, + } + for i, src := range cases { + fn := firstTextLOONG64(t, src) + if _, _, _, _, _, err := assembleLOONG64(fn); err == nil { + t.Errorf("case %d: expected an error, got none", i) + } + } +} + +// TestLOONG64_pcsp checks the stack-adjustment table of a framed function: +// the prologue raises the SP delta by autosize (in effect from the third +// instruction) and the RET's epilogue restores it to zero, with the pc deltas +// in MinLC (4) units — byte-identical to `go tool asm`. +func TestLOONG64_pcsp(t *testing.T) { + cases := []struct { + name string + src string + want []byte + }{ + { + "leaf", + `#include "textflag.h" +TEXT ·leaf(SB), NOSPLIT, $8-0 + MOVV R4, R5 + RET +`, + []byte{0x02, 0x02, 0x20, 0x03, 0x1f, 0x01, 0x00}, + }, + { + "nonleaf", + `#include "textflag.h" +TEXT ·nonleaf(SB), NOSPLIT, $8-0 + MOVV R4, R5 + JAL (R12) + RET +`, + []byte{0x02, 0x02, 0x20, 0x05, 0x1f, 0x01, 0x00}, + }, + } + for _, c := range cases { + t.Run(c.name, func(t *testing.T) { + f, errs := parser.Parse("pcsp_loong64.s", c.src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + if got := pcspTable(img.Funcs[0], 4); !bytes.Equal(got, c.want) { + t.Errorf("pcsp = % x, want % x", got, c.want) + } + }) + } +} + +// TestLOONG64_sbRefsUndefined checks that a reference to a symbol no GLOBL +// defines assembles into a relocation and is rejected at object emission. +func TestLOONG64_sbRefsUndefined(t *testing.T) { + f, errs := parser.Parse("sb_loong64.s", `#include "textflag.h" +TEXT ·sb(SB), NOSPLIT, $0 + MOVV missing(SB), R4 + RET +`) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + if len(img.Funcs[0].Relocs) != 2 { + t.Fatalf("relocs = %d, want the HI/LO pair", len(img.Funcs[0].Relocs)) + } + if _, err := img.GOObjectLOONG64("p", "sb_loong64.s"); err == nil { + t.Error("expected an unknown-symbol error at emission") + } +} + +// TestLOONG64_movImmToFp checks the immediate-to-FP move forms. +func TestLOONG64_movImmToFp(t *testing.T) { + fn := firstTextLOONG64(t, `#include "textflag.h" +TEXT ·fpmov(SB), NOSPLIT, $0 + MOVV $0x1, F0 + MOVW $0x2, F4 + RET +`) + code := assembleLOONG64Helper(t, fn) + want := []byte{ + 0x00, 0x04, 0x80, 0x03, // ori f0, r0, 1 + 0x04, 0x08, 0x80, 0x03, // ori f4, r0, 2 + 0x20, 0x00, 0x00, 0x4c, // jirl r0, r1, 0 + } + if !bytes.Equal(code, want) { + t.Errorf("code = % x\nwant % x", code, want) + } +} diff --git a/cmd/gasm/main.go b/cmd/gasm/main.go index 737f0d3..c0bc1eb 100644 --- a/cmd/gasm/main.go +++ b/cmd/gasm/main.go @@ -398,8 +398,8 @@ func cmdAsm(args []string) int { fs := newCommand("asm", "gasm asm [--format raw|elf|goobj] [-p pkg] [-o out] ", ` Assemble FILE without the Go toolchain: every TEXT function is encoded to machine code and printed as a hex dump. Supported architectures: amd64 - (including VEX/AVX2 and EVEX/AVX-512) and riscv64 (RV64IMAFDC + RVC); - arm64 and loong64 encoding is not yet implemented. + (including VEX/AVX2 and EVEX/AVX-512), riscv64 (RV64IMAFDC + RVC) and + loong64 (LoongArch base ISA); arm64 encoding is not yet implemented. With -o the output is written to a file instead. The --format flag selects what is written: raw (the default) concatenates the functions and the data @@ -493,16 +493,22 @@ requires -p, the package path, and the installed Go toolchain). } obj, kind = img.Bytes(), "raw image" case "elf": - if targetArch == arch.RISCV { + switch targetArch { + case arch.RISCV: obj, err = img.ELFRISCVObject() - } else { + case arch.LOONG64: + obj, err = img.ELFLOONG64Object() + default: obj, err = img.ELFObject() } kind = "ELF object" case "goobj": - if targetArch == arch.RISCV { + switch targetArch { + case arch.RISCV: obj, err = img.GOObjectRISCV(*pkg, path) - } else { + case arch.LOONG64: + obj, err = img.GOObjectLOONG64(*pkg, path) + default: obj, err = img.GOObject(*pkg, path) } kind = "Go object" @@ -822,6 +828,99 @@ func cmdVerifyRISCV(path string, groundTruth, profile bool) int { return 0 } +// cmdVerifyLOONG64 verifies a loong64 source file against `go tool asm` +// (GOARCH=loong64) — the ground-truth oracle — since gasm cannot JIT-load +// LoongArch code on an amd64 host. Relocation sites are masked before the +// byte comparison, as the toolchain leaves them zero for the linker. +func cmdVerifyLOONG64(path string, groundTruth, profile bool) int { + src, err := readSource(path) + if err != nil { + fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err) + return 1 + } + f, errs := parser.Parse(path, src) + for _, e := range errs { + fmt.Fprintf(os.Stderr, "%s: %v\n", path, e) + } + if len(errs) > 0 { + return 1 + } + img, err := asm.AssembleFileLOONG64(f) + if err != nil { + fmt.Fprintf(os.Stderr, "gasm verify: %v\n", err) + return 1 + } + + if groundTruth { + gt, err := verify.GroundTruthLOONG64(path) + if err != nil { + fmt.Fprintf(os.Stderr, "gasm verify: ground truth: %v\n", err) + return 1 + } + matched, total := 0, 0 + for _, fn := range img.Funcs { + gasmCode := img.Code[fn.Offset : fn.Offset+fn.Size] + goCode, ok := gt[fn.Name] + if !ok { + fmt.Printf(" %s: SKIP (not in go tool asm output)\n", fn.Name) + continue + } + total++ + gasmCmp := make([]byte, len(gasmCode)) + goCmp := make([]byte, len(goCode)) + copy(gasmCmp, gasmCode) + copy(goCmp, goCode) + for _, r := range fn.Relocs { + for j := r.Off; j < r.Off+4 && j < len(gasmCmp); j++ { + gasmCmp[j] = 0 + } + for j := r.Off; j < r.Off+4 && j < len(goCmp); j++ { + goCmp[j] = 0 + } + } + if bytes.Equal(gasmCmp, goCmp) { + matched++ + if len(fn.Relocs) > 0 { + fmt.Printf(" %s: MATCH (%d bytes, %d relocs masked)\n", fn.Name, fn.Size, len(fn.Relocs)) + } else { + fmt.Printf(" %s: MATCH (%d bytes)\n", fn.Name, fn.Size) + } + } else { + fmt.Printf(" %s: MISMATCH (%d vs %d bytes)\n", fn.Name, fn.Size, len(goCode)) + for i := 0; i < len(gasmCode) || i < len(goCode); i += 16 { + var gb, gs string + for j := i; j < i+16 && j < len(gasmCode); j++ { + gb += fmt.Sprintf(" %02x", gasmCode[j]) + } + for j := i; j < i+16 && j < len(goCode); j++ { + gs += fmt.Sprintf(" %02x", goCode[j]) + } + fmt.Printf(" %04x: gasm:%s\n", i, gb) + fmt.Printf(" %04x: gt: %s\n", i, gs) + } + } + } + fmt.Printf("%s: %d/%d matched\n", path, matched, total) + if matched < total { + return 1 + } + return 0 + } + + if profile { + for _, fn := range img.Funcs { + fmt.Printf("%s: %d bytes, labels: %v\n", fn.Name, fn.Size, fn.Labels) + } + return 0 + } + + fmt.Printf("%s: %d functions assembled\n", path, len(img.Funcs)) + for _, fn := range img.Funcs { + fmt.Printf(" %s: %d bytes\n", fn.Name, fn.Size) + } + return 0 +} + func cmdVerify(args []string) int { fs := newCommand("verify", "gasm verify [-smoke] [-abi] [-fuzz] [-ground-truth] [-profile] [-call] ", ` Assemble FILE (amd64), map it into executable memory and report the available @@ -867,14 +966,18 @@ decoders) that crash on random input but should succeed on valid data. } path := fs.Arg(0) targetArch := arch.FromFilename(path) - if targetArch != arch.AMD64 && targetArch != arch.RISCV { - fmt.Fprintln(os.Stderr, "gasm verify: only amd64 and riscv64 are supported") - return 1 - } - - // RISC-V: ground-truth only (no JIT on non-RISC-V hosts). - if targetArch == arch.RISCV { + switch targetArch { + case arch.AMD64: + // JIT-based verification below. + case arch.RISCV: + // RISC-V: ground-truth only (no JIT on non-RISC-V hosts). return cmdVerifyRISCV(path, *groundTruth, *profile) + case arch.LOONG64: + // LoongArch: ground-truth only (no JIT on non-LoongArch hosts). + return cmdVerifyLOONG64(path, *groundTruth, *profile) + default: + fmt.Fprintln(os.Stderr, "gasm verify: only amd64, riscv64 and loong64 are supported") + return 1 } k, err := verify.Load(path) diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 8546310..54ac29c 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -201,6 +201,17 @@ R_RISCV_PCREL_HI20/LO12 relocations). The encoder compresses eligible instructions to 16-bit RVC forms and is validated byte-for-byte against `GOARCH=riscv64 go tool asm`. +A **LoongArch encoder** (Phase 5, LoongArch64) encodes the integer and +floating-point instruction sets with the dual-form arithmetic mnemonics (3R +vs 2RI12), the 16/21-bit branch families, the MOV pseudo-instruction and its +constant materialisation (the dcon classification driving lu12i.w/ori/lu32i.d/ +lu52i.d expansions), the FP/SP frame mapping (autosize = align8(frame+8), +prologue storing the link register before and after the SP decrement) and +SB/global symbol references (pcalau12i pairs with R_LOONG64_ADDR_HI/LO +relocations). Like the RISC-V encoder it is validated byte-for-byte against +`GOARCH=loong64 go tool asm`, and its GOOBJ output is proven end-to-end by +substituting it into a cross-compiled `go build` and linking with `cmd/link`. + On top of the encoder, `Assemble` walks a parsed `TEXT` body, converts each operand to an encoder operand, and lays the instructions out so local labels resolve to relative jump offsets: jumps start in the short (rel8) form and @@ -288,12 +299,19 @@ boundaries, plus flat `pcfile`, `pcline` and `pcinline` tables — so a gasm-assembled object drops into a `go build` in place of the toolchain's. The object preamble (the version-and-experiment header the linker compares verbatim) is captured from the installed `go tool asm`, so the output is -always consistent with the toolchain that links it. RISC-V GOOBJ emission -uses the same format with the RISC-V architecture marker and RISC-V relocation -types. External cross-package -references and the implicit funcdata/DWARF symbols remain future work (the -linker fills the latter's defaults); the rest of Phase 2 is those, the -remaining EVEX forms and the other architectures. +always consistent with the toolchain that links it. LoongArch GOOBJ emission +uses the same shared emitter with the loong64 marker and R_LOONG64_ADDR_HI/LO +relocation types, and the emitted object links into a real `go build` for +`GOARCH=loong64`. Per function, the emitter also writes the two DWARF +symbols the linker's DWARF pass reads verbatim — the subprogram DIE +(`SDWARFFCN`) and the `.debug_line` state-machine program (`SDWARFLINES`), +both built the way `cmd/asm` builds them (the DIE carries the +R_DWTXTADDR_U4 address reference; the line program one row per source-line +change, in the same special-opcode encoding) — and the pc-value deltas are +in the architecture's MinLC units, as the runtime's `pcvalue` expects. +External cross-package references remain future work (the amd64 and RISC-V +paths resolve them; LoongArch does not yet); the rest of Phase 2 is those +and the remaining EVEX forms. ### `verify` diff --git a/parser/parser.go b/parser/parser.go index 1aca30f..c0dffc5 100644 --- a/parser/parser.go +++ b/parser/parser.go @@ -298,10 +298,14 @@ func parseSymbolPrefix(g []token.Token) (*ast.Symbol, int) { sym.Static = true i += 2 } - if i < len(g) && g[i].Kind == token.Plus { + if i < len(g) && (g[i].Kind == token.Plus || g[i].Kind == token.Minus) { + neg := g[i].Kind == token.Minus i++ if i < len(g) && g[i].Kind == token.Number { sym.Offset, sym.HasOff = parseInt(g[i].Text), true + if neg { + sym.Offset = -sym.Offset + } i++ } } diff --git a/testdata/verify/basic_loong64.s b/testdata/verify/basic_loong64.s new file mode 100644 index 0000000..a6bf0b8 --- /dev/null +++ b/testdata/verify/basic_loong64.s @@ -0,0 +1,54 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// add returns a + b. +TEXT ·add(SB), NOSPLIT, $0-24 + MOVV a+0(FP), R4 + MOVV b+8(FP), R5 + ADDV R5, R4, R4 + MOVV R4, ret+16(FP) + RET + +// arith exercises the 3R integer and FP set. +TEXT ·arith(SB), NOSPLIT, $0-0 + ADDV R4, R5, R6 + SUBV R7, R8, R9 + MULV R10, R11, R12 + DIVV R13, R14, R15 + AND R16, R17, R18 + OR R18, R19, R20 + XOR R20, R21, R2 + SLLV R2, R23, R24 + SRLV R24, R25, R26 + SRAV R26, R27, R28 + RET + +// imm exercises the immediate forms. +TEXT ·imm(SB), NOSPLIT, $0-0 + ADDV $42, R4, R5 + ADDV $-8, R6 + AND $0xff, R7, R8 + OR $1, R9, R10 + XOR $0, R11, R12 + SGT $100, R13, R14 + SLLV $4, R15, R16 + MOVV $0x12345, R17 + RET + +// branch exercises conditional and unconditional control flow. +TEXT ·branch(SB), NOSPLIT, $0-0 + BEQ R4, R5, done + BNE R6, R7, skip + BLT R8, R9, done + BGE R10, R11, done + BLTU R12, R13, done + BGEU R14, R15, done +skip: + JMP loop +loop: + JAL skip + RET +done: + RET diff --git a/testdata/verify/fp_loong64.s b/testdata/verify/fp_loong64.s new file mode 100644 index 0000000..51da362 --- /dev/null +++ b/testdata/verify/fp_loong64.s @@ -0,0 +1,143 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +#include "textflag.h" + +// fp exercises the floating-point set: 3R arithmetic, 2R unary, compares +// into FCC, fused multiply-add and the register moves. +TEXT ·fp(SB), NOSPLIT, $0-0 + ADDD F4, F5, F6 + SUBD F7, F8, F9 + MULD F9, F10, F11 + DIVD F11, F12, F13 + MULF F13, F14, F15 + ADDF F15, F16, F17 + SQRTD F17, F18 + SQRTF F18, F19 + ABSD F19, F20 + NEGD F20, F21 + MOVD F21, F22 + CMPEQD F22, F23, FCC0 + CMPGTF F23, F24, FCC1 + CMPGED F24, F25, FCC2 + FMADDD F0, F1, F2, F3 + FMSUBF F3, F4, F5, F6 + FNMADDD F6, F7, F8, F9 + FNMSUBF F9, F10, F11, F12 + FMAXD F12, F13, F14 + FMINF F14, F15, F16 + FMAXAD F16, F17, F18 + FMINAF F18, F19, F20 + FSCALEBF F20, F21, F22 + FCOPYSGD F22, F23, F24 + MOVV F25, R25 + MOVV R26, F27 + MOVW R28, F29 + MOVW F30, R31 + RET + +// mov forms: register moves, immediates (12/32/64-bit), memory with FP/SP +// pseudo-registers and the register-indexed forms. +TEXT ·mov(SB), NOSPLIT, $0-16 + MOVV R4, R5 + MOVW R6, R7 + MOVB R8, R9 + MOVBU R10, R11 + MOVHU R12, R13 + MOVWU R14, R15 + MOVV $42, R16 + MOVV $0x12345, R17 + MOVV $0x100000, R18 + MOVW $-100, R19 + MOVV $0x123456789, R20 + MOVV a+0(FP), R21 + MOVV R23, b+8(FP) + MOVW c+16(FP), R24 + MOVV (R24)(R25), R26 + MOVV R27, (R28)(R29) + RET + +// frame exercises the prologue/epilogue of a function with a real frame. +TEXT ·frame(SB), NOSPLIT, $32-8 + MOVV R4, R5 + MOVV arg+0(FP), R6 + MOVV R7, local-8(SP) + MOVV local-8(SP), R8 + MOVV R9, ret+0(FP) + RET + +// branches21 exercises the single-register and zero-register branch forms +// with 21-bit offsets. +TEXT ·branches21(SB), NOSPLIT, $0-0 + BEQ R0, R4, l1 + BEQ R5, R0, l2 + BNE R0, R6, l3 + BNE R7, R0, l4 + BLTZ R8, l5 + BGEZ R9, l6 + BLEZ R10, l7 + BGTZ R11, l8 + JMP l9 +l1: + JMP l10 +l2: + JMP l11 +l3: + JMP l12 +l4: + JMP l13 +l5: + JMP l14 +l6: + JMP l15 +l7: + JMP l16 +l8: + JMP l16 +l9: + MOVV R1, R2 +l10: + LL (R12), R13 + LLV (R14), R15 + SC R16, (R17) + SCV R18, (R19) + RDTIMED R20, R21 + SYSCALL + DBAR + RET +l11: + JAL (R30) + RET +l12: + BSTRINSV $7, R4, $0, R5 + BSTRPICKV $63, R6, $32, R7 + ALSLV $2, R8, R9, R10 + ADDV16 $65536, R11, R12 + RET +l13: + MOVV $0xffffffffffffffff, R13 + RET +l14: + CPUCFG R14, R14 + RET +l15: + NOR R15, R16, R17 + ORN R18, R19, R20 + ANDN R21, R24, R25 + RET +l16: + MOVB R26, (R27) + MOVB (R28), R29 + RET + +// sbdata loads and stores a static symbol with relocations (the relocation +// fields are masked before comparison). +GLOBL ·table(SB), RODATA, $16 +DATA ·table+0(SB)/8, $0x1122334455667788 +DATA ·table+8(SB)/8, $0x8877665544332211 + +TEXT ·sbdata(SB), NOSPLIT, $0-0 + MOVV $·table(SB), R4 + MOVV ·table(SB), R5 + MOVV R6, ·table+8(SB) + RET diff --git a/verify/groundtruth.go b/verify/groundtruth.go index 91c939f..309643e 100644 --- a/verify/groundtruth.go +++ b/verify/groundtruth.go @@ -32,6 +32,12 @@ func GroundTruthRISCV(path string) (map[string][]byte, error) { return groundTruthArch(path, "riscv64") } +// GroundTruthLOONG64 assembles the given .s file with the Go toolchain in +// LoongArch cross-assembly mode (GOARCH=loong64). +func GroundTruthLOONG64(path string) (map[string][]byte, error) { + return groundTruthArch(path, "loong64") +} + func groundTruthArch(path, goarch string) (map[string][]byte, error) { goroot := runtime.GOROOT() asmBin := filepath.Join(goroot, "pkg", "tool", runtime.GOOS+"_"+runtime.GOARCH, "asm") @@ -51,6 +57,7 @@ func groundTruthArch(path, goarch string) (map[string][]byte, error) { pkg := strings.TrimSuffix(base, ".s") pkg = strings.TrimSuffix(pkg, "_amd64") pkg = strings.TrimSuffix(pkg, "_riscv64") + pkg = strings.TrimSuffix(pkg, "_loong64") cmd := exec.Command(asmBin, "-I", includeDir, "-p", pkg, "-o", objPath, path) if goarch != "" { diff --git a/verify/l64_groundtruth_test.go b/verify/l64_groundtruth_test.go new file mode 100644 index 0000000..d0f1b07 --- /dev/null +++ b/verify/l64_groundtruth_test.go @@ -0,0 +1,97 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package verify + +import ( + "bytes" + "fmt" + "os" + "testing" + + "sourcedock.dev/petrbalvin/gasm-devkit/asm" + "sourcedock.dev/petrbalvin/gasm-devkit/parser" +) + +// TestGroundTruthLOONG64 assembles the loong64 test kernels with gasm and +// compares them byte-for-byte against `go tool asm` (GOARCH=loong64). The +// relocation fields of static-symbol references are masked before the +// comparison, since the toolchain leaves them zero for the linker. +func TestGroundTruthLOONG64(t *testing.T) { + for _, path := range []string{ + "../testdata/verify/basic_loong64.s", + "../testdata/verify/fp_loong64.s", + } { + t.Run(path, func(t *testing.T) { + src, err := os.ReadFile(path) + if err != nil { + t.Fatalf("read: %v", err) + } + f, errs := parser.Parse(path, string(src)) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := asm.AssembleFileLOONG64(f) + if err != nil { + t.Fatalf("AssembleFileLOONG64: %v", err) + } + gt, err := GroundTruthLOONG64(path) + if err != nil { + t.Fatalf("GroundTruthLOONG64: %v", err) + } + + matched := 0 + for _, fn := range img.Funcs { + gasmCode := maskRelocs(append([]byte(nil), img.Code[fn.Offset:fn.Offset+fn.Size]...), fn.Relocs) + goCode, ok := gt[fn.Name] + if !ok { + t.Errorf("%s: not in ground truth (%d functions)", fn.Name, len(gt)) + continue + } + goCode = maskRelocs(goCode, fn.Relocs) + if !bytes.Equal(gasmCode, goCode) { + t.Errorf("%s: MISMATCH gasm=%d go=%d bytes\n%s", fn.Name, len(gasmCode), len(goCode), diffHex(gasmCode, goCode)) + continue + } + matched++ + t.Logf("%s: MATCH (%d bytes)", fn.Name, fn.Size) + } + if matched == 0 { + t.Fatal("no functions matched") + } + }) + } +} + +// maskRelocs zeroes the 4-byte immediate fields of the relocation sites. +func maskRelocs(code []byte, relocs []asm.Reloc) []byte { + for _, r := range relocs { + for j := r.Off; j < r.Off+4 && j < len(code); j++ { + code[j] = 0 + } + } + return code +} + +func diffHex(a, b []byte) string { + var out bytes.Buffer + n := len(a) + if len(b) > n { + n = len(b) + } + for i := 0; i < n; i += 4 { + ab, bb := "??", "??" + if i < len(a) { + ab = fmt.Sprintf("%02x%02x%02x%02x", a[i], a[i+1], a[i+2], a[i+3]) + } + if i < len(b) { + bb = fmt.Sprintf("%02x%02x%02x%02x", b[i], b[i+1], b[i+2], b[i+3]) + } + mark := " " + if ab != bb { + mark = "!" + } + fmt.Fprintf(&out, "%04x: %s %s %s\n", i, ab, bb, mark) + } + return out.String() +}