Files
gasm-sdk/asm/loong64_assemble.go
T

1415 lines
44 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"math/bits"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and
// the immediate-arithmetic forms expand to 2-5 instructions when the
// immediate does not fit, so the layout is computed in two passes (sizes,
// then encoding with resolved branch targets).
//
// The emitted bytes match the Go toolchain's loong64 assembler, which is the
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch
// encodings and the MOV immediate expansions all follow cmd/internal/obj/
// loong64's asmout cases. One deliberate difference: the stack-growth guard
// (the morestack check in the prologue and the call back into the runtime in
// the epilogue) is not emitted, so the bytes match only for NOSPLIT functions
// or zero-frame leaves, where the toolchain emits no guard either.
func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := loong64ComputeFrame(t)
prologue := loong64Prologue(fi)
guardLen := loong64GuardLen(fi)
chain := loong64JumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
return r
}
return name
}
var relocs []Reloc
var spadj []SpadjStep
// The prologue (3 instructions when a frame is present) raises the SP
// delta by autosize; the boundary is reported at the third instruction's
// pc, exactly as the toolchain's pctospadj does.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: guardLen + 8, Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
pos += loong64InstrSize(s, fi)
}
}
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, which the first
// pass has sized.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += loong64InstrSize(in, fi)
}
}
bodyLen = p - (guardLen + len(prologue))
}
var out []byte
if fi.needSplit {
out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero
// after the addi.d (one instruction for a leaf, two for a non-leaf
// with the LR restore).
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
epi := 4
if !fi.leaf {
epi = 8
}
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
block, blReloc := loong64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil
}
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
// branch-chasing pass: a label whose first instruction is an unconditional
// local jump redirects its own jumpers to the ultimate target. The Go
// toolchain chases these chains before it encodes branches, so matching its
// bytes requires the same redirection.
func loong64JumpChain(t *ast.Text) map[string]string {
leadsTo := map[string]string{}
for i, stmt := range t.Body {
l, ok := stmt.(*ast.Label)
if !ok {
continue
}
j := i + 1
for j < len(t.Body) {
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
break
}
j++
}
if j >= len(t.Body) {
continue
}
in, ok := t.Body[j].(*ast.Instr)
if !ok {
continue
}
mnem := strings.ToUpper(in.Mnemonic.Text)
if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 {
continue
}
if name, ok := l64LabelOK(in.Operands[0]); ok {
leadsTo[l.Name.Text] = name
}
}
chain := map[string]string{}
for name := range leadsTo {
visited := map[string]bool{name: true}
cur := name
for {
next, ok := leadsTo[cur]
if !ok || visited[next] {
break
}
visited[next] = true
cur = next
}
if cur != name {
chain[name] = cur
}
}
return chain
}
// l64LabelOK returns the local label name of a jump operand.
func l64LabelOK(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
if mnem == "RET" {
return len(loong64Return(fi))
}
switch mnem {
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := l64Imm64(ops[0])
if v == 0 {
return 4 // folds into the 3R form (rk = R0)
}
switch mnem {
case "ADD", "ADDW", "ADDV", "ADDVU", "SGT", "SGTU":
// C_US12CON (−2048..0x7ff) encodes directly as addi/slti.
if v >= -2048 && v <= 0x7ff {
return 4
}
// C_U12CON (0x800..0xfff) → ori r30, r0, v; op rd, rj, r30.
if v >= 0x800 && v <= 0xfff {
return 8
}
default: // AND/OR/XOR
// C_UU12CON (0..0x7ff) encodes directly as andi/ori/xori.
if v >= 0 && v <= 0x7ff {
return 4
}
// C_S12CON (−2048..−1) → addi.d r30, r0, v; op rd, rj, r30.
if v >= -2048 && v < 0 {
return 8
}
}
// 0x800..0xfff for AND/OR/XOR and the 32/64-bit ranges go through
// the lu12i.w materialisation.
if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
return 8 // lu12i.w r30, v>>12; op rd, rj, r30
}
return 12 // lu12i.w r30, v>>12; ori r30, r30, v; op rd, rj, r30
}
return 4 * (len(l64DconMovWords(0, v)) + 1) // dcon materialisation + op
}
}
return 4
}
// encodeLOONG64Instr encodes a single LoongArch instruction.
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
// Pseudo-instructions and the branches first.
switch mnem {
case "RET":
return loong64Return(fi), nil
case "NOP", "NOOP":
// andi r0, r0, 0
return l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)), nil
case "UNDEF":
// break 0
return l64wordLE(l64i15(l64InstrTable["BREAK"].op, 0)), nil
case "WORD":
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
}
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
if op, ok := l64branchTable[mnem]; ok {
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
}
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
if op, ok := l64branch21Table[mnem]; ok {
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve)
}
// B/BL aliases reached only via JMP/JAL above.
// The dual-form arithmetic mnemonics: register (3R) or immediate (2RI12).
if de, ok := l64DualTable[mnem]; ok {
if len(ops) >= 2 && isImmOperand(ops[0]) {
if de.shift {
// INSTR $shamt, rd or INSTR $shamt, rj, rd.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
shamt := int(immFromOperand(ops[0]))
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// $0 folds into the register form (the toolchain matches the
// zero constant against the 3R optab entry first).
if shamt == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
// The .d variants take a 6-bit amount, the .w variants 5 bits.
if isLoong64ShiftD(de.imm) {
shamt &= 0x3f
} else {
shamt &= 0x1f
}
return l64wordLE(l64irr(de.imm, shamt, rj, rd)), nil
}
return encodeLOONG64ImmArith(mnem, de, ops)
}
// Register form: 3R.
if len(ops) == 3 {
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(de.rrr, rk, rj, rd)), nil
}
if len(ops) == 2 {
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rk < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(de.rrr, rk, rd, rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
enc, ok := l64InstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
}
switch enc.format {
case l64Frrr:
// INSTR rk, rj, rd (3 operands) or INSTR rk, rd (rj = rd).
switch len(ops) {
case 3:
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
case 2:
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rk < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(enc.op, rk, rd, rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 register operands, got %d", mnem, len(ops))
case l64Frr:
// INSTR rj, rd.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
imm := int(immFromOperand(ops[0]))
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irr(enc.op, imm, rj, rd)), nil
case l64Firr16:
// ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be
// a multiple of 65536 and is shifted right by 16.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
v := int(immFromOperand(ops[0]))
if v&0xFFFF != 0 {
return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem)
}
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irr16(enc.op, v>>16, rj, rd)), nil
case l64Firr14:
// LL/SC/MOVWP: INSTR mem, rd (load) or INSTR rd, mem (store); the
// 14-bit offset is scaled by 4 (byte offset >> 2).
rd, rj, off, load, err := l64MemOperands(ops, fi)
if err != nil {
return nil, err
}
op := enc.op
if load && (mnem == "MOVWP" || mnem == "MOVVP") {
// ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field.
op -= 1 << 24
}
return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil
case l64Fir20:
// LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rd := l64Reg(ops[0])
if rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64ir(enc.op, int(immFromOperand(ops[1])), rd)), nil
case l64Frrrr:
// FMADD/FMSUB/FNMADD/FNMSUB: INSTR fa, fk, fj, fd (4 operands) or
// INSTR fa, fk, fd (fj = fd).
fa, fk, fj, fd, err := l64FmaOperands(ops)
if err != nil {
return nil, err
}
return l64wordLE(l64rrrr(enc.op, fa, fk, fj, fd)), nil
case l64Firir:
// BSTRINS/BSTRPICK: INSTR $msb, rj, $lsb, rd (or $msb, rj, rd with
// lsb = 0).
if len(ops) != 4 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops))
}
msb := int(immFromOperand(ops[0]))
lsb := 0
rj := l64Reg(ops[1])
rd := l64Reg(ops[len(ops)-1])
if len(ops) == 4 {
lsb = int(immFromOperand(ops[2]))
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
case l64Firrr:
// ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in
// the second register position); the source amount is 1-4, encoded
// as sa-1.
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
sa := int(immFromOperand(ops[0])) - 1
rj, rk, rd := l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
if sa < 0 || sa > 3 {
return nil, fmt.Errorf("shift amount out of range [1, 4]")
}
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irrr(enc.op, sa, rk, rj, rd)), nil
case l64Fi15:
// SYSCALL/BREAK/DBAR: no operands, or SYSCALL $code / BREAK $code.
code := 0
if len(ops) == 1 {
code = int(immFromOperand(ops[0]))
} else if len(ops) > 1 {
return nil, fmt.Errorf("%s expects at most 1 operand, got %d", mnem, len(ops))
}
return l64wordLE(l64i15(enc.op, code)), nil
case l64Fam:
// AM* val, (addr), result.
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rk := l64Reg(ops[0])
rj, _ := l64Mem(ops[1])
rd := l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
case l64Frdtime:
// RDTIME* rd, rj (rd at bits [9:5], rj at bits [4:0]).
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rd, rj := l64Reg(ops[0]), l64Reg(ops[1])
if rd < 0 || rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rr(enc.op, rd, rj)), nil
case l64Fpreld:
// PRELD off(rj), $hint.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rj, off := l64Mem(ops[0])
hint := int(immFromOperand(ops[1]))
if rj < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return l64wordLE(l64irr5i(enc.op, int(off), rj, hint)), nil
}
return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops))
}
// encodeLOONG64Branch encodes a label or indirect jump/call:
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
}
op := instr.Operands[0]
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
// Indirect: (rj) → jirl.
rj := loong64RegNum(op.Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
rd := 0
if link {
rd = 1 // link register
}
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
}
// Direct: label → b/bl.
target := resolve(l64Label(op))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
if v < -1<<25 || v >= 1<<25 {
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
}
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[len(ops)-1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
if len(ops) == 2 {
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
// BGEZ-family aliases encoded with rj in the rj field.
rj := l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
zop := l64branch21Table["BEQZ"]
if mnem == "BNE" {
zop = l64branch21Table["BNEZ"]
}
if mnem == "BLT" || mnem == "BLTZ" || mnem == "BGTZ" {
zop = l64branch21Table["BLTZ"]
}
if mnem == "BGE" || mnem == "BGEZ" || mnem == "BLEZ" {
zop = l64branch21Table["BGEZ"]
}
return l64wordLE(l64ir21(zop, v, rj)), nil
}
// Two registers: BEQ rj, rd, label. When one is R0 the toolchain
// re-encodes as the 21-bit BEQZ/BNEZ form.
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if rj == 0 {
rj, rd = rd, 0
}
if rd == 0 {
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
zop := l64branch21Table["BEQZ"]
if mnem == "BNE" {
zop = l64branch21Table["BNEZ"]
}
return l64wordLE(l64ir21(zop, v, rj)), nil
}
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, rj, rd)), nil
}
// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
rj := 0 // BFPT/BFPF default to FCC0
if mnem != "BFPT" && mnem != "BFPF" {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
if mnem == "BGTZ" || mnem == "BLEZ" {
// The toolchain swaps the register into the rd field and keeps the
// 16-bit offset form.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, 0, rj)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
return l64wordLE(l64ir21(op, v, rj)), nil
}
// encodeLOONG64ImmArith encodes an immediate arithmetic/logic instruction,
// expanding the immediate exactly as the toolchain's aclass classifies it:
//
// ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes)
// 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8)
// AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4)
// −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8)
// 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12)
// 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20)
func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := l64Imm64(ops[0])
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// The two immediate families classify differently.
additive := mnem == "ADD" || mnem == "ADDW" || mnem == "ADDV" || mnem == "ADDVU" || mnem == "SGT" || mnem == "SGTU"
if additive {
if v == 0 {
// $0 folds into the 3R form (rk = R0), matching the toolchain's
// optab matching of the zero constant against the register form.
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= -2048 && v <= 0x7ff {
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= 0x800 && v <= 0xfff {
return l64WordsLE(
l64irr(0x00e<<22, int(v), 0, 30), // ori r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
} else {
if v == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= 0 && v <= 0x7ff {
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= -2048 && v < 0 {
return l64WordsLE(
l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
}
// 32/64-bit constants are materialised in R30 (the assembler temp),
// using the same dcon classification as the toolchain's case 24/60/70/
// 71/72 sequences.
const (
lu12iw = 0x0a << 25
ori = 0x00e << 22
)
if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
return l64WordsLE(
l64ir(lu12iw, int(int32(v)>>12), 30),
l64rrr(de.rrr, 30, rj, rd),
), nil
}
return l64WordsLE(
l64ir(lu12iw, int(int32(v)>>12), 30),
l64irr(ori, int(v), 30, 30),
l64rrr(de.rrr, 30, rj, rd),
), nil
}
words := l64DconMovWords(30, v)
words = append(words, l64rrr(de.rrr, 30, rj, rd))
return l64WordsLE(words...), nil
}
// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of
// the 6-bit (.d) variants, the toolchain distinguishes them by the bit
// position of the opcode field (bits [25:16]).
func isLoong64ShiftD(op uint32) bool {
return op&0x03ff0000 != 0 && op>>25 == 0
}
// l64FmaOperands extracts the four fused-multiply-add operands:
// INSTR fa, fk, fj, fd, or INSTR fa, fk, fd with fj = fd.
func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) {
switch len(ops) {
case 4:
fa, fk, fj, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
case 3:
fa, fk, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
fj = fd
default:
return 0, 0, 0, 0, fmt.Errorf("expected 3 or 4 operands, got %d", len(ops))
}
if fa < 0 || fk < 0 || fj < 0 || fd < 0 {
return 0, 0, 0, 0, fmt.Errorf("invalid register operand")
}
return fa, fk, fj, fd, nil
}
// l64MemOperands extracts (rd, rj, off, load) from a load/store instruction:
// INSTR mem, rd is a load, INSTR rd, mem a store.
func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) {
if len(ops) != 2 {
return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops))
}
if isMemOperand(ops[0]) {
rd = l64Reg(ops[1])
rj, off = l64MemWithFrame(ops[0], fi)
load = true
} else if isMemOperand(ops[1]) {
rd = l64Reg(ops[0])
rj, off = l64MemWithFrame(ops[1], fi)
} else {
return 0, 0, 0, false, fmt.Errorf("expected a memory operand")
}
if rd < 0 || rj < 0 {
return 0, 0, 0, false, fmt.Errorf("invalid operand")
}
return rd, rj, off, load, nil
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width
// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the
// access width). The forms, mirroring the toolchain:
//
// MOVx $imm, rd load immediate (addi/lu12i+ori/lu32i/lu52i)
// MOVx mem, rd load from memory
// MOVx rd, mem store to memory
// MOVx rs, rd register move (incl. the FP-bank specials)
// MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d)
// MOVx sym(SB), rd load from a static symbol (pcalau12i+ld)
// MOVx rd, sym(SB) store to a static symbol (pcalau12i+st)
func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if mnem == "MOV" {
mnem = "MOVV"
}
src, dst := ops[0], ops[1]
// Immediate → register.
if isImmOperand(src) && !isMemOperand(src) {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
}
// MOVF/MOVD $imm, Fd → materialise in R30, then movgr2fr.{w,d}.
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
return encodeLOONG64ImmToFp(rd, l64Imm64(src), mnem), nil
}
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
}
// Static symbol load/store via pcalau12i.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem)
}
return encodeLOONG64SBLoad(src.Addr.Sym, rd, mnem, relocs), nil
}
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) {
rs := l64Reg(src)
if rs < 0 {
return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem)
}
return encodeLOONG64SBStore(dst.Addr.Sym, rs, mnem, relocs), nil
}
// Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk).
if src.Addr.Index != "" && !isMemOperand(dst) {
rd := l64Reg(dst)
rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index)
if rd < 0 || rj < 0 || rk < 0 {
return nil, fmt.Errorf("%s (rj)(rk): invalid register operand", mnem)
}
op, ok := l64IndexedTable[mnem]
if !ok {
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
}
return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil
}
if dst.Addr.Index != "" && !isMemOperand(src) {
rs := l64Reg(src)
rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index)
if rs < 0 || rj < 0 || rk < 0 {
return nil, fmt.Errorf("%s rd, (rj)(rk): invalid register operand", mnem)
}
op, ok := l64IndexedTable[mnem]
if !ok {
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
}
return l64wordLE(l64rrr(op.st, rk, rj, rs)), nil
}
// Memory load/store with a 12-bit (or larger, via expansion) offset.
if isMemOperand(src) && !isMemOperand(dst) {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s: invalid destination register", mnem)
}
return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi)
}
if !isMemOperand(src) && isMemOperand(dst) {
rs := l64Reg(src)
if rs < 0 {
return nil, fmt.Errorf("%s: invalid source register", mnem)
}
return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi)
}
// Register → register.
return encodeLOONG64RegMove(mnem, src, dst)
}
// loong64MovSize returns the encoded size of a MOV instruction.
func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if mnem == "MOV" {
mnem = "MOVV"
}
if len(ops) != 2 {
return 4
}
src, dst := ops[0], ops[1]
switch {
case isImmOperand(src):
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // addi/ori r30 + movgr2fr
}
v := l64Imm64(src)
if v == 0 {
return 4
}
if v > 0 && v <= 0xfff {
return 4 // ori rd, r0, v
}
if v >= -2048 && v < 0 {
return 4 // addi.d rd, r0, v
}
if v == int64(int32(v)) {
if v&0xfff == 0 {
return 4 // lu12i.w
}
return 8 // lu12i.w + ori
}
return 4 * len(l64DconMovWords(0, v))
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
return 8 // pcalau12i + ld
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
return 8 // pcalau12i + st
case src.Addr.Index != "" || dst.Addr.Index != "":
return 4 // ldx/stx
case isMemOperand(src) || isMemOperand(dst):
// A 12-bit offset fits in one instruction; larger offsets expand
// to lu12i.w + add.d + the access.
mem := src
if !isMemOperand(src) {
mem = dst
}
if l64MemOffset(mem, fi) >= -2048 && l64MemOffset(mem, fi) < 2048 {
return 4
}
return 12
default:
return 4 // register move
}
}
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it to
// an F register (the toolchain's case 34: movgr2fr.w/movgr2fr.d).
func encodeLOONG64ImmToFp(fd int, v int64, mnem string) []byte {
// ori for positive constants, addi.d for zero/negative.
op := uint32(0x00b << 22)
if v > 0 {
op = 0x00e << 22
}
mov := uint32(0x452a << 10) // movgr2fr.d
if mnem == "MOVF" {
mov = 0x4529 << 10 // movgr2fr.w
}
return l64WordsLE(
l64irr(op, int(v), 0, 30),
l64rr(mov, 30, fd),
)
}
// ---- 64-bit immediate classification ----
// The dcon classes classify a 64-bit constant by which of the four
// materialisation instructions (lu12i.w, ori, lu32i.d, lu52i.d) can be
// dropped, mirroring the toolchain's dconClass: a field is ALL1/ALL0 when
// it is all ones/zeros (fillable by sign/zero extension) or ST1/ST0 when it
// starts with a 1/0 but is mixed.
const (
l64All1 = iota
l64All0
l64St1
l64St0
l64dcon120
l64dcon1220s
l64dcon20s20
l64dcon1212s
l64dcon20s12s
l64dcon20s0
l64dcon1212u
l64dcon20s12u
l64dcon3212s
l64dcon320
l64dcon3220
l64dcon1232s
l64dcon20s32
l64dcon3212u
l64Dcon
)
// l64BitField classifies the bit field of v at [suf+len-1 : suf].
func l64BitField(v int64, suf, ln int8) int {
var mask1, mask2 uint64
if ln == 12 {
if suf == 0 {
mask1, mask2 = 0xfff, 0x800
} else {
mask1, mask2 = 0xfff0000000000000, 0x8000000000000000
}
} else {
if suf == 12 {
mask1, mask2 = 0xfffff000, 0x80000000
} else {
mask1, mask2 = 0xfffff00000000, 0x8000000000000
}
}
u := uint64(v)
switch {
case u&mask1 == mask1:
return l64All1
case u&mask1 == 0:
return l64All0
case u&mask2 == mask2:
return l64St1
}
return l64St0
}
// l64DconClass returns the materialisation class of a 64-bit constant,
// transcribed from cmd/internal/obj/loong64's dconClass.
func l64DconClass(v int64) int {
tzb := bits.TrailingZeros64(uint64(v))
hi12 := l64BitField(v, 52, 12)
hi20 := l64BitField(v, 32, 20)
lo20 := l64BitField(v, 12, 20)
lo12 := l64BitField(v, 0, 12)
if tzb >= 52 {
return l64dcon120
}
if tzb >= 32 {
if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) {
return l64dcon20s0
}
return l64dcon320
}
if tzb >= 12 {
if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1220s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s20
}
return l64dcon3220
}
if hi20 == l64All0 {
return l64dcon1220s
}
if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) {
return l64dcon20s20
}
return l64dcon3220
}
if lo12 == l64St1 || lo12 == l64All1 {
if lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1212s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s12s
}
return l64dcon3212s
}
if lo20 == l64St1 {
if hi20 == l64All1 {
return l64dcon1232s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if lo20 == l64All0 {
if hi20 == l64All0 {
return l64dcon1212u
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s12u
}
return l64dcon3212u
}
if hi20 == l64All0 {
return l64dcon1232s
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if lo20 == l64All0 {
if hi20 == l64All0 {
return l64dcon1212u
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s12u
}
return l64dcon3212u
}
if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1232s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if hi20 == l64All0 {
return l64dcon1232s
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
// l64DconMovWords returns the materialisation words for a 64-bit constant
// into rd, per the toolchain's case 67/68/69/59 sequences.
func l64DconMovWords(rd int, v int64) []uint32 {
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
addiw = 0x00a << 22
addid = 0x00b << 22
ori = 0x00e << 22
)
switch l64DconClass(v) {
case l64dcon120:
return []uint32{l64irr(lu52id, int(v>>52), 0, rd)}
case l64dcon1220s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s20:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon1212s:
return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s12s, l64dcon20s0:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon1212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s12u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon3212s, l64dcon320:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon3220:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon1232s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s32:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon3212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
default:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
}
}
// encodeLOONG64LoadImm loads an immediate into a register, matching the
// toolchain's MOVV/MOVW case 3/19/25/59 expansion:
//
// $0: or rd, r0, r0 (MOVW: sll.w rd, r0, r0)
// 1..0xfff: ori rd, r0, imm
// −2048..−1: addi.d rd, r0, imm
// 32-bit (low 12 zero): lu12i.w rd, imm>>12
// 32-bit: lu12i.w rd, imm>>12; ori rd, rd, imm
// 64-bit: lu12i.w rd, imm>>12; ori rd, rd, imm;
// lu32i.d rd, imm>>32; lu52i.d rd, rd, imm>>52
func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte {
if v == 0 {
// The zero constant matches the register-form optab entry: MOVV →
// or rd, r0, r0, MOVW → sll.w rd, r0, r0.
op := l64movRegTable["MOVV"].op
if mnem == "MOVW" {
op = l64movRegTable["MOVW"].op
}
return l64wordLE(l64rrr(op, 0, 0, rd))
}
if v > 0 && v <= 0xfff {
return l64wordLE(l64irr(l64DualTable["OR"].imm, int(v), 0, rd))
}
if v >= -2048 && v < 0 {
// Both MOVV and MOVW use addi.d for negative constants.
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(v), 0, rd))
}
if v == int64(int32(v)) {
if v&0xfff == 0 {
return l64wordLE(l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd))
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd),
l64irr(l64DualTable["OR"].imm, int(v), rd, rd),
)
}
// 64-bit constants use the shortest materialisation the bit pattern
// admits (dcon classification).
return l64WordsLE(l64DconMovWords(rd, v)...)
}
// encodeLOONG64MemOp encodes a memory load (load = true) or store with a
// 12-bit offset, or the 3-instruction expansion for larger offsets:
// lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30).
func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo) ([]byte, error) {
rj, off := l64MemWithFrame(mem, fi)
if rj < 0 {
return nil, fmt.Errorf("invalid memory operand")
}
ls, ok := l64loadStoreTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported MOV width %q", mnem)
}
op := ls.st
if load {
op = ls.ld
}
if off >= -2048 && off < 2048 {
return l64wordLE(l64irr(op, int(off), rj, reg)), nil
}
// Large offset: materialise the base in R30 (the assembler temp).
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int((off+0x800)>>12), 30),
l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30),
l64irr(op, int(off), 30, reg),
), nil
}
// encodeLOONG64RegMove encodes a register-to-register move: the width
// extensions (ext.w.b, ext.w.h, sll.w, or, andi, bstrpick.d) between GPRs,
// fmov between F registers, and the special moves across the GPR/FP/FCC/FCSR
// banks.
func encodeLOONG64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
rs, rd := l64Reg(src), l64Reg(dst)
if rs < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sc, dc := loong64RegClass(operandRegName(src)), loong64RegClass(operandRegName(dst))
// FP-bank specials (MOVV/MOVW between GPR/FCC/FCSR and F registers).
if key, ok := l64FpMoveKey(mnem, sc, dc); ok {
op, ok := l64FpMovTable[key]
if !ok {
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
}
return l64wordLE(l64rr(op, rs, rd)), nil
}
// GPR → GPR.
if sc == l64ClsGR && dc == l64ClsGR {
switch mnem {
case "MOVHU":
// bstrpick.d rd, rj, $15, $0
return l64wordLE(l64irir(0x3<<22, 15, rs, 0, rd)), nil
case "MOVWU":
// bstrpick.d rd, rj, $31, $0
return l64wordLE(l64irir(0x3<<22, 31, rs, 0, rd)), nil
}
if e, ok := l64movRegTable[mnem]; ok {
if e.rr {
return l64wordLE(l64rr(e.op, rs, rd)), nil
}
if e.imm != 0 {
return l64wordLE(l64irr(e.op, e.imm, rs, rd)), nil
}
// 3R with rk = r0: or rd, rj, r0 / sll.w rd, rj, r0.
return l64wordLE(l64rrr(e.op, 0, rs, rd)), nil
}
}
// F → F.
if sc == l64ClsFP && dc == l64ClsFP {
if op, ok := l64movFpRegTable[mnem]; ok {
return l64wordLE(l64rr(op, rs, rd)), nil
}
}
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
}
// l64FpMoveKey builds the l64FpMovTable key for a cross-bank move, reporting
// whether the move is a cross-bank special at all.
func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) {
bank := func(c l64RegClass) string {
switch c {
case l64ClsFP:
return "F"
case l64ClsFCC:
return "FCC"
case l64ClsFCSR:
return "FCSR"
default:
return "R"
}
}
if sc == dc {
return "", false
}
if mnem != "MOVV" && mnem != "MOVW" {
return "", false
}
key := mnem + "." + bank(sc) + "." + bank(dc)
_, ok := l64FpMovTable[key]
return key, ok
}
// ---- static symbol references (pcalau12i + offset) ----
// encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the
// R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address.
func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, rd),
l64irr(l64DualTable["ADDV"].imm, 0, rd, rd),
)
}
// encodeLOONG64SBLoad emits pcalau12i r30, 0; ld rd, 0(r30) with the
// R_LOONG64_ADDR_HI/LO pair, loading from a static symbol.
func encodeLOONG64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte {
ls := l64loadStoreTable[mnem]
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
l64irr(ls.ld, 0, 30, rd),
)
}
// encodeLOONG64SBStore emits pcalau12i r30, 0; st rd, 0(r30) with the
// R_LOONG64_ADDR_HI/LO pair, storing to a static symbol.
func encodeLOONG64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) []byte {
ls := l64loadStoreTable[mnem]
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
l64irr(ls.st, 0, 30, rs),
)
}
// ---- operand helpers ----
// l64IndexedTable holds the register-indexed load/store (ldx/stx) opcodes.
var l64IndexedTable = map[string]struct{ ld, st uint32 }{
"MOVB": {0x07000 << 15, 0x07020 << 15},
"MOVH": {0x07008 << 15, 0x07028 << 15},
"MOVW": {0x07010 << 15, 0x07030 << 15},
"MOVV": {0x07018 << 15, 0x07038 << 15},
"MOVBU": {0x07040 << 15, 0x07020 << 15},
"MOVHU": {0x07048 << 15, 0x07028 << 15},
"MOVWU": {0x07050 << 15, 0x07030 << 15},
"MOVF": {0x07060 << 15, 0x07070 << 15},
"MOVD": {0x07068 << 15, 0x07078 << 15},
}
// operandRegName returns the register name of an operand, or "".
func operandRegName(op *ast.Operand) string {
if op.Addr.Base != "" {
return op.Addr.Base
}
if op.Addr.Sym != nil && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name
}
return ""
}
// l64Reg returns the register number of an operand, or -1.
func l64Reg(op *ast.Operand) int {
return loong64RegNum(operandRegName(op))
}
// l64Imm64 returns the full 64-bit immediate value of an operand.
func l64Imm64(op *ast.Operand) int64 {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
return 0
}
// l64Mem returns the base register and byte offset of a memory operand.
func l64Mem(op *ast.Operand) (rj int, off int32) {
rj = loong64RegNum(op.Addr.Base)
off = int32(op.Addr.Offset)
return
}
// l64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
// registers via the frame mapping.
func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return loong64ResolvePseudo(op.Addr.Sym, fi)
}
return l64Mem(op)
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
return off
}
// l64Label returns the label name of an operand.
func l64Label(op *ast.Operand) string {
if op.Addr.Sym != nil {
return op.Addr.Sym.Name
}
return op.Raw
}