1422 lines
44 KiB
Go
1422 lines
44 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||
// SPDX-License-Identifier: BSD-3-Clause
|
||
|
||
package asm
|
||
|
||
import (
|
||
"fmt"
|
||
"math/bits"
|
||
"strings"
|
||
|
||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||
)
|
||
|
||
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
|
||
// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and
|
||
// the immediate-arithmetic forms expand to 2-5 instructions when the
|
||
// immediate does not fit, so the layout is computed in two passes (sizes,
|
||
// then encoding with resolved branch targets).
|
||
//
|
||
// The emitted bytes match the Go toolchain's loong64 assembler, which is the
|
||
// ground-truth oracle: prologue/epilogue (including the large-frame R30
|
||
// materialisations), FP/SP frame mapping, the stack-split guard classes, and
|
||
// branch encodings all follow cmd/internal/obj/loong64. The morestack block
|
||
// at the end of split functions carries the runtime.morestack_noctxt call.
|
||
func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
|
||
fi := loong64ComputeFrame(t)
|
||
prologue := loong64Prologue(fi)
|
||
guardLen := loong64GuardLen(fi)
|
||
chain := loong64JumpChain(t)
|
||
resolve := func(name string) string {
|
||
if r, ok := chain[name]; ok {
|
||
return r
|
||
}
|
||
return name
|
||
}
|
||
|
||
var relocs []Reloc
|
||
var spadj []SpadjStep
|
||
|
||
// The prologue raises the SP delta by autosize; the boundary is reported
|
||
// after the SP adjust instruction, exactly as the toolchain's pctospadj
|
||
// does. The prologue (3 instructions when a frame is present) may
|
||
// materialise its store or adjust through R30, which widens it.
|
||
if fi.autosize != 0 {
|
||
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
|
||
}
|
||
|
||
// Pass 1: label offsets from the instruction sizes.
|
||
offsets := map[string]int{}
|
||
pos := guardLen + len(prologue)
|
||
for _, stmt := range t.Body {
|
||
switch s := stmt.(type) {
|
||
case *ast.Label:
|
||
offsets[s.Name.Text] = pos
|
||
case *ast.Instr:
|
||
pos += loong64InstrSize(s, fi)
|
||
}
|
||
}
|
||
|
||
// Pass 2: encode. The guard prefix precedes the prologue; its branches
|
||
// target the morestack block at the end of the function, which the first
|
||
// pass has sized.
|
||
bodyLen := 0
|
||
{
|
||
p := guardLen + len(prologue)
|
||
for _, stmt := range t.Body {
|
||
if in, ok := stmt.(*ast.Instr); ok {
|
||
p += loong64InstrSize(in, fi)
|
||
}
|
||
}
|
||
bodyLen = p - (guardLen + len(prologue))
|
||
}
|
||
var out []byte
|
||
if fi.needSplit {
|
||
out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
|
||
}
|
||
out = append(out, prologue...)
|
||
pc := guardLen + len(prologue)
|
||
preCount := len(relocs)
|
||
var lines []LineEntry
|
||
for _, stmt := range t.Body {
|
||
in, ok := stmt.(*ast.Instr)
|
||
if !ok {
|
||
continue
|
||
}
|
||
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve)
|
||
if err != nil {
|
||
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
|
||
}
|
||
for j := preCount; j < len(relocs); j++ {
|
||
// Make the relocation offsets function-relative: each instruction
|
||
// records its reloc offset relative to its own start, and pc is
|
||
// that instruction's offset from the function start (prologue
|
||
// included). After shifts by the same amount.
|
||
relocs[j].Off += pc
|
||
relocs[j].After += pc
|
||
}
|
||
preCount = len(relocs)
|
||
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
|
||
// The RET's epilogue closes the frame: the SP delta returns to zero
|
||
// after the frame-deallocating ADDV.
|
||
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
|
||
spadj = append(spadj, SpadjStep{PC: pc + loong64EpilogueWords(fi)*4, Value: 0})
|
||
}
|
||
out = append(out, code...)
|
||
pc += len(code)
|
||
}
|
||
if fi.needSplit {
|
||
block, blReloc := loong64MoreStackBlock(pc)
|
||
out = append(out, block...)
|
||
relocs = append(relocs, blReloc)
|
||
pc += len(block)
|
||
}
|
||
return out, offsets, relocs, lines, spadj, nil
|
||
}
|
||
|
||
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
|
||
// branch-chasing pass: a label whose first instruction is an unconditional
|
||
// local jump redirects its own jumpers to the ultimate target. The Go
|
||
// toolchain chases these chains before it encodes branches, so matching its
|
||
// bytes requires the same redirection.
|
||
func loong64JumpChain(t *ast.Text) map[string]string {
|
||
leadsTo := map[string]string{}
|
||
for i, stmt := range t.Body {
|
||
l, ok := stmt.(*ast.Label)
|
||
if !ok {
|
||
continue
|
||
}
|
||
j := i + 1
|
||
for j < len(t.Body) {
|
||
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
|
||
break
|
||
}
|
||
j++
|
||
}
|
||
if j >= len(t.Body) {
|
||
continue
|
||
}
|
||
in, ok := t.Body[j].(*ast.Instr)
|
||
if !ok {
|
||
continue
|
||
}
|
||
mnem := strings.ToUpper(in.Mnemonic.Text)
|
||
if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 {
|
||
continue
|
||
}
|
||
if name, ok := l64LabelOK(in.Operands[0]); ok {
|
||
leadsTo[l.Name.Text] = name
|
||
}
|
||
}
|
||
chain := map[string]string{}
|
||
for name := range leadsTo {
|
||
visited := map[string]bool{name: true}
|
||
cur := name
|
||
for {
|
||
next, ok := leadsTo[cur]
|
||
if !ok || visited[next] {
|
||
break
|
||
}
|
||
visited[next] = true
|
||
cur = next
|
||
}
|
||
if cur != name {
|
||
chain[name] = cur
|
||
}
|
||
}
|
||
return chain
|
||
}
|
||
|
||
// l64LabelOK returns the local label name of a jump operand.
|
||
func l64LabelOK(op *ast.Operand) (string, bool) {
|
||
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
||
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
|
||
return op.Addr.Sym.Name, true
|
||
}
|
||
return "", false
|
||
}
|
||
|
||
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
|
||
// most, more for the multi-instruction expansions.
|
||
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
|
||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||
ops := instr.Operands
|
||
|
||
if mnem == "RET" {
|
||
return len(loong64Return(fi))
|
||
}
|
||
switch mnem {
|
||
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
|
||
return loong64MovSize(mnem, ops, fi)
|
||
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
|
||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||
v := l64Imm64(ops[0])
|
||
if v == 0 {
|
||
return 4 // folds into the 3R form (rk = R0)
|
||
}
|
||
switch mnem {
|
||
case "ADD", "ADDW", "ADDV", "ADDVU", "SGT", "SGTU":
|
||
// C_US12CON (−2048..0x7ff) encodes directly as addi/slti.
|
||
if v >= -2048 && v <= 0x7ff {
|
||
return 4
|
||
}
|
||
// C_U12CON (0x800..0xfff) → ori r30, r0, v; op rd, rj, r30.
|
||
if v >= 0x800 && v <= 0xfff {
|
||
return 8
|
||
}
|
||
default: // AND/OR/XOR
|
||
// C_UU12CON (0..0x7ff) encodes directly as andi/ori/xori.
|
||
if v >= 0 && v <= 0x7ff {
|
||
return 4
|
||
}
|
||
// C_S12CON (−2048..−1) → addi.d r30, r0, v; op rd, rj, r30.
|
||
if v >= -2048 && v < 0 {
|
||
return 8
|
||
}
|
||
}
|
||
// 0x800..0xfff for AND/OR/XOR and the 32/64-bit ranges go through
|
||
// the lu12i.w materialisation.
|
||
if v == int64(int32(v)) {
|
||
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
|
||
return 8 // lu12i.w r30, v>>12; op rd, rj, r30
|
||
}
|
||
return 12 // lu12i.w r30, v>>12; ori r30, r30, v; op rd, rj, r30
|
||
}
|
||
return 4 * (len(l64DconMovWords(0, v)) + 1) // dcon materialisation + op
|
||
}
|
||
}
|
||
return 4
|
||
}
|
||
|
||
// encodeLOONG64Instr encodes a single LoongArch instruction.
|
||
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
|
||
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
||
ops := instr.Operands
|
||
|
||
// Pseudo-instructions and the branches first.
|
||
switch mnem {
|
||
case "RET":
|
||
return loong64Return(fi), nil
|
||
case "NOP", "NOOP":
|
||
// andi r0, r0, 0
|
||
return l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)), nil
|
||
case "UNDEF":
|
||
// break 0
|
||
return l64wordLE(l64i15(l64InstrTable["BREAK"].op, 0)), nil
|
||
case "WORD":
|
||
if len(ops) != 1 {
|
||
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
||
}
|
||
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
|
||
case "JMP", "B":
|
||
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
|
||
case "JAL", "CALL", "BL":
|
||
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
|
||
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
|
||
return encodeLOONG64Mov(instr, mnem, fi, relocs)
|
||
}
|
||
|
||
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
|
||
if op, ok := l64branchTable[mnem]; ok {
|
||
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
|
||
}
|
||
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
|
||
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
|
||
if op, ok := l64branch21Table[mnem]; ok {
|
||
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve)
|
||
}
|
||
// B/BL aliases reached only via JMP/JAL above.
|
||
|
||
// The dual-form arithmetic mnemonics: register (3R) or immediate (2RI12).
|
||
if de, ok := l64DualTable[mnem]; ok {
|
||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||
if de.shift {
|
||
// INSTR $shamt, rd or INSTR $shamt, rj, rd.
|
||
if len(ops) != 2 && len(ops) != 3 {
|
||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||
}
|
||
shamt := int(immFromOperand(ops[0]))
|
||
rd := l64Reg(ops[len(ops)-1])
|
||
rj := rd
|
||
if len(ops) == 3 {
|
||
rj = l64Reg(ops[1])
|
||
}
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
// $0 folds into the register form (the toolchain matches the
|
||
// zero constant against the 3R optab entry first).
|
||
if shamt == 0 {
|
||
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
|
||
}
|
||
// The .d variants take a 6-bit amount, the .w variants 5 bits.
|
||
if isLoong64ShiftD(de.imm) {
|
||
shamt &= 0x3f
|
||
} else {
|
||
shamt &= 0x1f
|
||
}
|
||
return l64wordLE(l64irr(de.imm, shamt, rj, rd)), nil
|
||
}
|
||
return encodeLOONG64ImmArith(mnem, de, ops)
|
||
}
|
||
// Register form: 3R.
|
||
if len(ops) == 3 {
|
||
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
|
||
if rk < 0 || rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64rrr(de.rrr, rk, rj, rd)), nil
|
||
}
|
||
if len(ops) == 2 {
|
||
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
|
||
if rk < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64rrr(de.rrr, rk, rd, rd)), nil
|
||
}
|
||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||
}
|
||
|
||
enc, ok := l64InstrTable[mnem]
|
||
if !ok {
|
||
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
|
||
}
|
||
|
||
switch enc.format {
|
||
case l64Frrr:
|
||
// INSTR rk, rj, rd (3 operands) or INSTR rk, rd (rj = rd).
|
||
switch len(ops) {
|
||
case 3:
|
||
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
|
||
if rk < 0 || rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
|
||
case 2:
|
||
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
|
||
if rk < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64rrr(enc.op, rk, rd, rd)), nil
|
||
}
|
||
return nil, fmt.Errorf("%s expects 2 or 3 register operands, got %d", mnem, len(ops))
|
||
|
||
case l64Frr:
|
||
// INSTR rj, rd.
|
||
if len(ops) != 2 {
|
||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||
}
|
||
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64rr(enc.op, rj, rd)), nil
|
||
|
||
case l64Firr:
|
||
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
|
||
if len(ops) < 2 || !isImmOperand(ops[0]) {
|
||
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
|
||
}
|
||
imm := int(immFromOperand(ops[0]))
|
||
rd := l64Reg(ops[len(ops)-1])
|
||
rj := rd
|
||
if len(ops) == 3 {
|
||
rj = l64Reg(ops[1])
|
||
}
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64irr(enc.op, imm, rj, rd)), nil
|
||
|
||
case l64Firr16:
|
||
// ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be
|
||
// a multiple of 65536 and is shifted right by 16.
|
||
if len(ops) < 2 || !isImmOperand(ops[0]) {
|
||
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
|
||
}
|
||
v := int(immFromOperand(ops[0]))
|
||
if v&0xFFFF != 0 {
|
||
return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem)
|
||
}
|
||
rd := l64Reg(ops[len(ops)-1])
|
||
rj := rd
|
||
if len(ops) == 3 {
|
||
rj = l64Reg(ops[1])
|
||
}
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64irr16(enc.op, v>>16, rj, rd)), nil
|
||
|
||
case l64Firr14:
|
||
// LL/SC/MOVWP: INSTR mem, rd (load) or INSTR rd, mem (store); the
|
||
// 14-bit offset is scaled by 4 (byte offset >> 2).
|
||
rd, rj, off, load, err := l64MemOperands(ops, fi)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
op := enc.op
|
||
if load && (mnem == "MOVWP" || mnem == "MOVVP") {
|
||
// ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field.
|
||
op -= 1 << 24
|
||
}
|
||
return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil
|
||
|
||
case l64Fir20:
|
||
// LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm.
|
||
if len(ops) != 2 {
|
||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||
}
|
||
rd := l64Reg(ops[0])
|
||
if rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64ir(enc.op, int(immFromOperand(ops[1])), rd)), nil
|
||
|
||
case l64Frrrr:
|
||
// FMADD/FMSUB/FNMADD/FNMSUB: INSTR fa, fk, fj, fd (4 operands) or
|
||
// INSTR fa, fk, fd (fj = fd).
|
||
fa, fk, fj, fd, err := l64FmaOperands(ops)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
return l64wordLE(l64rrrr(enc.op, fa, fk, fj, fd)), nil
|
||
|
||
case l64Firir:
|
||
// BSTRINS/BSTRPICK: INSTR $msb, rj, $lsb, rd (or $msb, rj, rd with
|
||
// lsb = 0).
|
||
if len(ops) != 4 && len(ops) != 3 {
|
||
return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops))
|
||
}
|
||
msb := int(immFromOperand(ops[0]))
|
||
lsb := 0
|
||
rj := l64Reg(ops[1])
|
||
rd := l64Reg(ops[len(ops)-1])
|
||
if len(ops) == 4 {
|
||
lsb = int(immFromOperand(ops[2]))
|
||
}
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
|
||
|
||
case l64Firrr:
|
||
// ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in
|
||
// the second register position); the source amount is 1-4, encoded
|
||
// as sa-1.
|
||
if len(ops) != 4 {
|
||
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
||
}
|
||
sa := int(immFromOperand(ops[0])) - 1
|
||
rj, rk, rd := l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
|
||
if sa < 0 || sa > 3 {
|
||
return nil, fmt.Errorf("shift amount out of range [1, 4]")
|
||
}
|
||
if rk < 0 || rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64irrr(enc.op, sa, rk, rj, rd)), nil
|
||
|
||
case l64Fi15:
|
||
// SYSCALL/BREAK/DBAR: no operands, or SYSCALL $code / BREAK $code.
|
||
code := 0
|
||
if len(ops) == 1 {
|
||
code = int(immFromOperand(ops[0]))
|
||
} else if len(ops) > 1 {
|
||
return nil, fmt.Errorf("%s expects at most 1 operand, got %d", mnem, len(ops))
|
||
}
|
||
return l64wordLE(l64i15(enc.op, code)), nil
|
||
|
||
case l64Fam:
|
||
// AM* val, (addr), result.
|
||
if len(ops) != 3 {
|
||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||
}
|
||
rk := l64Reg(ops[0])
|
||
rj, _ := l64Mem(ops[1])
|
||
rd := l64Reg(ops[2])
|
||
if rk < 0 || rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||
}
|
||
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
|
||
|
||
case l64Frdtime:
|
||
// RDTIME* rd, rj (rd at bits [9:5], rj at bits [4:0]).
|
||
if len(ops) != 2 {
|
||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||
}
|
||
rd, rj := l64Reg(ops[0]), l64Reg(ops[1])
|
||
if rd < 0 || rj < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
return l64wordLE(l64rr(enc.op, rd, rj)), nil
|
||
|
||
case l64Fpreld:
|
||
// PRELD off(rj), $hint.
|
||
if len(ops) != 2 {
|
||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||
}
|
||
rj, off := l64Mem(ops[0])
|
||
hint := int(immFromOperand(ops[1]))
|
||
if rj < 0 {
|
||
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
||
}
|
||
return l64wordLE(l64irr5i(enc.op, int(off), rj, hint)), nil
|
||
}
|
||
return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops))
|
||
}
|
||
|
||
// encodeLOONG64Branch encodes a label or indirect jump/call:
|
||
//
|
||
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
|
||
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
|
||
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
|
||
if len(instr.Operands) != 1 {
|
||
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
|
||
}
|
||
op := instr.Operands[0]
|
||
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
|
||
// Indirect: (rj) → jirl.
|
||
rj := loong64RegNum(op.Addr.Base)
|
||
if rj < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
rd := 0
|
||
if link {
|
||
rd = 1 // link register
|
||
}
|
||
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
|
||
}
|
||
// Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation
|
||
// (the linker fills the offset), as the toolchain does for CALL/BL/JAL
|
||
// and for tail-calling JMP.
|
||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||
opc := l64jumpTable["B"]
|
||
if link {
|
||
opc = l64jumpTable["BL"]
|
||
}
|
||
if relocs != nil {
|
||
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset})
|
||
}
|
||
return l64wordLE(l64bbl(opc, 0)), nil
|
||
}
|
||
// Direct: label → b/bl.
|
||
target := resolve(l64Label(op))
|
||
targetOff, ok := offsets[target]
|
||
if !ok {
|
||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||
}
|
||
v := (targetOff - pc) >> 2
|
||
if v < -1<<25 || v >= 1<<25 {
|
||
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
|
||
}
|
||
opc := l64jumpTable[mnem]
|
||
return l64wordLE(l64bbl(opc, v)), nil
|
||
}
|
||
|
||
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
|
||
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
|
||
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
|
||
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
||
if len(ops) != 2 && len(ops) != 3 {
|
||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||
}
|
||
target := resolve(l64Label(ops[len(ops)-1]))
|
||
targetOff, ok := offsets[target]
|
||
if !ok {
|
||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||
}
|
||
v := (targetOff - pc) >> 2
|
||
if len(ops) == 2 {
|
||
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
|
||
// BGEZ-family aliases encoded with rj in the rj field.
|
||
rj := l64Reg(ops[0])
|
||
if rj < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
if (v<<11)>>11 != v {
|
||
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
|
||
}
|
||
zop := l64branch21Table["BEQZ"]
|
||
if mnem == "BNE" {
|
||
zop = l64branch21Table["BNEZ"]
|
||
}
|
||
if mnem == "BLT" || mnem == "BLTZ" || mnem == "BGTZ" {
|
||
zop = l64branch21Table["BLTZ"]
|
||
}
|
||
if mnem == "BGE" || mnem == "BGEZ" || mnem == "BLEZ" {
|
||
zop = l64branch21Table["BGEZ"]
|
||
}
|
||
return l64wordLE(l64ir21(zop, v, rj)), nil
|
||
}
|
||
// Two registers: BEQ rj, rd, label. When one is R0 the toolchain
|
||
// re-encodes as the 21-bit BEQZ/BNEZ form.
|
||
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
if rj == 0 {
|
||
rj, rd = rd, 0
|
||
}
|
||
if rd == 0 {
|
||
if (v<<11)>>11 != v {
|
||
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
|
||
}
|
||
zop := l64branch21Table["BEQZ"]
|
||
if mnem == "BNE" {
|
||
zop = l64branch21Table["BNEZ"]
|
||
}
|
||
return l64wordLE(l64ir21(zop, v, rj)), nil
|
||
}
|
||
if (v<<16)>>16 != v {
|
||
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
|
||
}
|
||
return l64wordLE(l64irr16(op, v, rj, rd)), nil
|
||
}
|
||
|
||
// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and
|
||
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
|
||
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
|
||
// and a 16-bit offset, are handled separately.
|
||
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
||
if len(ops) != 2 {
|
||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||
}
|
||
target := resolve(l64Label(ops[1]))
|
||
targetOff, ok := offsets[target]
|
||
if !ok {
|
||
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
|
||
}
|
||
v := (targetOff - pc) >> 2
|
||
rj := 0 // BFPT/BFPF default to FCC0
|
||
if mnem != "BFPT" && mnem != "BFPF" {
|
||
rj = l64Reg(ops[0])
|
||
if rj < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
}
|
||
if mnem == "BGTZ" || mnem == "BLEZ" {
|
||
// The toolchain swaps the register into the rd field and keeps the
|
||
// 16-bit offset form.
|
||
if (v<<16)>>16 != v {
|
||
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
|
||
}
|
||
return l64wordLE(l64irr16(op, v, 0, rj)), nil
|
||
}
|
||
if (v<<11)>>11 != v {
|
||
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
|
||
}
|
||
return l64wordLE(l64ir21(op, v, rj)), nil
|
||
}
|
||
|
||
// encodeLOONG64ImmArith encodes an immediate arithmetic/logic instruction,
|
||
// expanding the immediate exactly as the toolchain's aclass classifies it:
|
||
//
|
||
// ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes)
|
||
// 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8)
|
||
// AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4)
|
||
// −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8)
|
||
// 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12)
|
||
// 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20)
|
||
func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) {
|
||
if len(ops) != 2 && len(ops) != 3 {
|
||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||
}
|
||
v := l64Imm64(ops[0])
|
||
rd := l64Reg(ops[len(ops)-1])
|
||
rj := rd
|
||
if len(ops) == 3 {
|
||
rj = l64Reg(ops[1])
|
||
}
|
||
if rj < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
|
||
// The two immediate families classify differently.
|
||
additive := mnem == "ADD" || mnem == "ADDW" || mnem == "ADDV" || mnem == "ADDVU" || mnem == "SGT" || mnem == "SGTU"
|
||
if additive {
|
||
if v == 0 {
|
||
// $0 folds into the 3R form (rk = R0), matching the toolchain's
|
||
// optab matching of the zero constant against the register form.
|
||
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
|
||
}
|
||
if v >= -2048 && v <= 0x7ff {
|
||
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
|
||
}
|
||
if v >= 0x800 && v <= 0xfff {
|
||
return l64WordsLE(
|
||
l64irr(0x00e<<22, int(v), 0, 30), // ori r30, r0, v
|
||
l64rrr(de.rrr, 30, rj, rd),
|
||
), nil
|
||
}
|
||
} else {
|
||
if v == 0 {
|
||
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
|
||
}
|
||
if v >= 0 && v <= 0x7ff {
|
||
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
|
||
}
|
||
if v >= -2048 && v < 0 {
|
||
return l64WordsLE(
|
||
l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v
|
||
l64rrr(de.rrr, 30, rj, rd),
|
||
), nil
|
||
}
|
||
}
|
||
|
||
// 32/64-bit constants are materialised in R30 (the assembler temp),
|
||
// using the same dcon classification as the toolchain's case 24/60/70/
|
||
// 71/72 sequences.
|
||
const (
|
||
lu12iw = 0x0a << 25
|
||
ori = 0x00e << 22
|
||
)
|
||
if v == int64(int32(v)) {
|
||
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
|
||
return l64WordsLE(
|
||
l64ir(lu12iw, int(int32(v)>>12), 30),
|
||
l64rrr(de.rrr, 30, rj, rd),
|
||
), nil
|
||
}
|
||
return l64WordsLE(
|
||
l64ir(lu12iw, int(int32(v)>>12), 30),
|
||
l64irr(ori, int(v), 30, 30),
|
||
l64rrr(de.rrr, 30, rj, rd),
|
||
), nil
|
||
}
|
||
words := l64DconMovWords(30, v)
|
||
words = append(words, l64rrr(de.rrr, 30, rj, rd))
|
||
return l64WordsLE(words...), nil
|
||
}
|
||
|
||
// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of
|
||
// the 6-bit (.d) variants, the toolchain distinguishes them by the bit
|
||
// position of the opcode field (bits [25:16]).
|
||
func isLoong64ShiftD(op uint32) bool {
|
||
return op&0x03ff0000 != 0 && op>>25 == 0
|
||
}
|
||
|
||
// l64FmaOperands extracts the four fused-multiply-add operands:
|
||
// INSTR fa, fk, fj, fd, or INSTR fa, fk, fd with fj = fd.
|
||
func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) {
|
||
switch len(ops) {
|
||
case 4:
|
||
fa, fk, fj, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
|
||
case 3:
|
||
fa, fk, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
|
||
fj = fd
|
||
default:
|
||
return 0, 0, 0, 0, fmt.Errorf("expected 3 or 4 operands, got %d", len(ops))
|
||
}
|
||
if fa < 0 || fk < 0 || fj < 0 || fd < 0 {
|
||
return 0, 0, 0, 0, fmt.Errorf("invalid register operand")
|
||
}
|
||
return fa, fk, fj, fd, nil
|
||
}
|
||
|
||
// l64MemOperands extracts (rd, rj, off, load) from a load/store instruction:
|
||
// INSTR mem, rd is a load, INSTR rd, mem a store.
|
||
func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) {
|
||
if len(ops) != 2 {
|
||
return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops))
|
||
}
|
||
if isMemOperand(ops[0]) {
|
||
rd = l64Reg(ops[1])
|
||
rj, off = l64MemWithFrame(ops[0], fi)
|
||
load = true
|
||
} else if isMemOperand(ops[1]) {
|
||
rd = l64Reg(ops[0])
|
||
rj, off = l64MemWithFrame(ops[1], fi)
|
||
} else {
|
||
return 0, 0, 0, false, fmt.Errorf("expected a memory operand")
|
||
}
|
||
if rd < 0 || rj < 0 {
|
||
return 0, 0, 0, false, fmt.Errorf("invalid operand")
|
||
}
|
||
return rd, rj, off, load, nil
|
||
}
|
||
|
||
// ---- the MOV pseudo-instruction ----
|
||
|
||
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
|
||
// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width
|
||
// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the
|
||
// access width). The forms, mirroring the toolchain:
|
||
//
|
||
// MOVx $imm, rd load immediate (addi/lu12i+ori/lu32i/lu52i)
|
||
// MOVx mem, rd load from memory
|
||
// MOVx rd, mem store to memory
|
||
// MOVx rs, rd register move (incl. the FP-bank specials)
|
||
// MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d)
|
||
// MOVx sym(SB), rd load from a static symbol (pcalau12i+ld)
|
||
// MOVx rd, sym(SB) store to a static symbol (pcalau12i+st)
|
||
func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) {
|
||
ops := instr.Operands
|
||
if len(ops) != 2 {
|
||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||
}
|
||
if mnem == "MOV" {
|
||
mnem = "MOVV"
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
|
||
// Immediate → register.
|
||
if isImmOperand(src) && !isMemOperand(src) {
|
||
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
||
rd := l64Reg(dst)
|
||
if rd < 0 {
|
||
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
|
||
}
|
||
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
|
||
}
|
||
rd := l64Reg(dst)
|
||
if rd < 0 {
|
||
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
|
||
}
|
||
// MOVF/MOVD $imm, Fd → materialise in R30, then movgr2fr.{w,d}.
|
||
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
||
return encodeLOONG64ImmToFp(rd, l64Imm64(src), mnem), nil
|
||
}
|
||
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
|
||
}
|
||
|
||
// Static symbol load/store via pcalau12i.
|
||
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) {
|
||
rd := l64Reg(dst)
|
||
if rd < 0 {
|
||
return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem)
|
||
}
|
||
return encodeLOONG64SBLoad(src.Addr.Sym, rd, mnem, relocs), nil
|
||
}
|
||
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) {
|
||
rs := l64Reg(src)
|
||
if rs < 0 {
|
||
return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem)
|
||
}
|
||
return encodeLOONG64SBStore(dst.Addr.Sym, rs, mnem, relocs), nil
|
||
}
|
||
|
||
// Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk).
|
||
if src.Addr.Index != "" && !isMemOperand(dst) {
|
||
rd := l64Reg(dst)
|
||
rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index)
|
||
if rd < 0 || rj < 0 || rk < 0 {
|
||
return nil, fmt.Errorf("%s (rj)(rk): invalid register operand", mnem)
|
||
}
|
||
op, ok := l64IndexedTable[mnem]
|
||
if !ok {
|
||
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
|
||
}
|
||
return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil
|
||
}
|
||
if dst.Addr.Index != "" && !isMemOperand(src) {
|
||
rs := l64Reg(src)
|
||
rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index)
|
||
if rs < 0 || rj < 0 || rk < 0 {
|
||
return nil, fmt.Errorf("%s rd, (rj)(rk): invalid register operand", mnem)
|
||
}
|
||
op, ok := l64IndexedTable[mnem]
|
||
if !ok {
|
||
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
|
||
}
|
||
return l64wordLE(l64rrr(op.st, rk, rj, rs)), nil
|
||
}
|
||
|
||
// Memory load/store with a 12-bit (or larger, via expansion) offset.
|
||
if isMemOperand(src) && !isMemOperand(dst) {
|
||
rd := l64Reg(dst)
|
||
if rd < 0 {
|
||
return nil, fmt.Errorf("%s: invalid destination register", mnem)
|
||
}
|
||
return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi)
|
||
}
|
||
if !isMemOperand(src) && isMemOperand(dst) {
|
||
rs := l64Reg(src)
|
||
if rs < 0 {
|
||
return nil, fmt.Errorf("%s: invalid source register", mnem)
|
||
}
|
||
return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi)
|
||
}
|
||
|
||
// Register → register.
|
||
return encodeLOONG64RegMove(mnem, src, dst)
|
||
}
|
||
|
||
// loong64MovSize returns the encoded size of a MOV instruction.
|
||
func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
|
||
if mnem == "MOV" {
|
||
mnem = "MOVV"
|
||
}
|
||
if len(ops) != 2 {
|
||
return 4
|
||
}
|
||
src, dst := ops[0], ops[1]
|
||
switch {
|
||
case isImmOperand(src):
|
||
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
||
return 8 // pcalau12i + addi.d
|
||
}
|
||
if (mnem == "MOVF" || mnem == "MOVD") && loong64RegClass(operandRegName(dst)) == l64ClsFP {
|
||
return 8 // addi/ori r30 + movgr2fr
|
||
}
|
||
v := l64Imm64(src)
|
||
if v == 0 {
|
||
return 4
|
||
}
|
||
if v > 0 && v <= 0xfff {
|
||
return 4 // ori rd, r0, v
|
||
}
|
||
if v >= -2048 && v < 0 {
|
||
return 4 // addi.d rd, r0, v
|
||
}
|
||
if v == int64(int32(v)) {
|
||
if v&0xfff == 0 {
|
||
return 4 // lu12i.w
|
||
}
|
||
return 8 // lu12i.w + ori
|
||
}
|
||
return 4 * len(l64DconMovWords(0, v))
|
||
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
|
||
return 8 // pcalau12i + ld
|
||
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
|
||
return 8 // pcalau12i + st
|
||
case src.Addr.Index != "" || dst.Addr.Index != "":
|
||
return 4 // ldx/stx
|
||
case isMemOperand(src) || isMemOperand(dst):
|
||
// A 12-bit offset fits in one instruction; larger offsets expand
|
||
// to lu12i.w + add.d + the access.
|
||
mem := src
|
||
if !isMemOperand(src) {
|
||
mem = dst
|
||
}
|
||
if l64MemOffset(mem, fi) >= -2048 && l64MemOffset(mem, fi) < 2048 {
|
||
return 4
|
||
}
|
||
return 12
|
||
default:
|
||
return 4 // register move
|
||
}
|
||
}
|
||
|
||
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it to
|
||
// an F register (the toolchain's case 34: movgr2fr.w/movgr2fr.d).
|
||
func encodeLOONG64ImmToFp(fd int, v int64, mnem string) []byte {
|
||
// ori for positive constants, addi.d for zero/negative.
|
||
op := uint32(0x00b << 22)
|
||
if v > 0 {
|
||
op = 0x00e << 22
|
||
}
|
||
mov := uint32(0x452a << 10) // movgr2fr.d
|
||
if mnem == "MOVF" {
|
||
mov = 0x4529 << 10 // movgr2fr.w
|
||
}
|
||
return l64WordsLE(
|
||
l64irr(op, int(v), 0, 30),
|
||
l64rr(mov, 30, fd),
|
||
)
|
||
}
|
||
|
||
// ---- 64-bit immediate classification ----
|
||
|
||
// The dcon classes classify a 64-bit constant by which of the four
|
||
// materialisation instructions (lu12i.w, ori, lu32i.d, lu52i.d) can be
|
||
// dropped, mirroring the toolchain's dconClass: a field is ALL1/ALL0 when
|
||
// it is all ones/zeros (fillable by sign/zero extension) or ST1/ST0 when it
|
||
// starts with a 1/0 but is mixed.
|
||
const (
|
||
l64All1 = iota
|
||
l64All0
|
||
l64St1
|
||
l64St0
|
||
|
||
l64dcon120
|
||
l64dcon1220s
|
||
l64dcon20s20
|
||
l64dcon1212s
|
||
l64dcon20s12s
|
||
l64dcon20s0
|
||
l64dcon1212u
|
||
l64dcon20s12u
|
||
l64dcon3212s
|
||
l64dcon320
|
||
l64dcon3220
|
||
l64dcon1232s
|
||
l64dcon20s32
|
||
l64dcon3212u
|
||
l64Dcon
|
||
)
|
||
|
||
// l64BitField classifies the bit field of v at [suf+len-1 : suf].
|
||
func l64BitField(v int64, suf, ln int8) int {
|
||
var mask1, mask2 uint64
|
||
if ln == 12 {
|
||
if suf == 0 {
|
||
mask1, mask2 = 0xfff, 0x800
|
||
} else {
|
||
mask1, mask2 = 0xfff0000000000000, 0x8000000000000000
|
||
}
|
||
} else {
|
||
if suf == 12 {
|
||
mask1, mask2 = 0xfffff000, 0x80000000
|
||
} else {
|
||
mask1, mask2 = 0xfffff00000000, 0x8000000000000
|
||
}
|
||
}
|
||
u := uint64(v)
|
||
switch {
|
||
case u&mask1 == mask1:
|
||
return l64All1
|
||
case u&mask1 == 0:
|
||
return l64All0
|
||
case u&mask2 == mask2:
|
||
return l64St1
|
||
}
|
||
return l64St0
|
||
}
|
||
|
||
// l64DconClass returns the materialisation class of a 64-bit constant,
|
||
// transcribed from cmd/internal/obj/loong64's dconClass.
|
||
func l64DconClass(v int64) int {
|
||
tzb := bits.TrailingZeros64(uint64(v))
|
||
hi12 := l64BitField(v, 52, 12)
|
||
hi20 := l64BitField(v, 32, 20)
|
||
lo20 := l64BitField(v, 12, 20)
|
||
lo12 := l64BitField(v, 0, 12)
|
||
if tzb >= 52 {
|
||
return l64dcon120
|
||
}
|
||
if tzb >= 32 {
|
||
if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) {
|
||
return l64dcon20s0
|
||
}
|
||
return l64dcon320
|
||
}
|
||
if tzb >= 12 {
|
||
if lo20 == l64St1 || lo20 == l64All1 {
|
||
if hi20 == l64All1 {
|
||
return l64dcon1220s
|
||
}
|
||
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
|
||
return l64dcon20s20
|
||
}
|
||
return l64dcon3220
|
||
}
|
||
if hi20 == l64All0 {
|
||
return l64dcon1220s
|
||
}
|
||
if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) {
|
||
return l64dcon20s20
|
||
}
|
||
return l64dcon3220
|
||
}
|
||
if lo12 == l64St1 || lo12 == l64All1 {
|
||
if lo20 == l64All1 {
|
||
if hi20 == l64All1 {
|
||
return l64dcon1212s
|
||
}
|
||
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
|
||
return l64dcon20s12s
|
||
}
|
||
return l64dcon3212s
|
||
}
|
||
if lo20 == l64St1 {
|
||
if hi20 == l64All1 {
|
||
return l64dcon1232s
|
||
}
|
||
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
|
||
return l64dcon20s32
|
||
}
|
||
return l64Dcon
|
||
}
|
||
if lo20 == l64All0 {
|
||
if hi20 == l64All0 {
|
||
return l64dcon1212u
|
||
}
|
||
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
|
||
return l64dcon20s12u
|
||
}
|
||
return l64dcon3212u
|
||
}
|
||
if hi20 == l64All0 {
|
||
return l64dcon1232s
|
||
}
|
||
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
|
||
return l64dcon20s32
|
||
}
|
||
return l64Dcon
|
||
}
|
||
if lo20 == l64All0 {
|
||
if hi20 == l64All0 {
|
||
return l64dcon1212u
|
||
}
|
||
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
|
||
return l64dcon20s12u
|
||
}
|
||
return l64dcon3212u
|
||
}
|
||
if lo20 == l64St1 || lo20 == l64All1 {
|
||
if hi20 == l64All1 {
|
||
return l64dcon1232s
|
||
}
|
||
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
|
||
return l64dcon20s32
|
||
}
|
||
return l64Dcon
|
||
}
|
||
if hi20 == l64All0 {
|
||
return l64dcon1232s
|
||
}
|
||
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
|
||
return l64dcon20s32
|
||
}
|
||
return l64Dcon
|
||
}
|
||
|
||
// l64DconMovWords returns the materialisation words for a 64-bit constant
|
||
// into rd, per the toolchain's case 67/68/69/59 sequences.
|
||
func l64DconMovWords(rd int, v int64) []uint32 {
|
||
const (
|
||
lu12iw = 0x0a << 25
|
||
lu32id = 0x0b << 25
|
||
lu52id = 0x00c << 22
|
||
addiw = 0x00a << 22
|
||
addid = 0x00b << 22
|
||
ori = 0x00e << 22
|
||
)
|
||
switch l64DconClass(v) {
|
||
case l64dcon120:
|
||
return []uint32{l64irr(lu52id, int(v>>52), 0, rd)}
|
||
case l64dcon1220s:
|
||
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
case l64dcon20s20:
|
||
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)}
|
||
case l64dcon1212s:
|
||
return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
case l64dcon20s12s, l64dcon20s0:
|
||
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
|
||
case l64dcon1212u:
|
||
return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
case l64dcon20s12u:
|
||
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
|
||
case l64dcon3212s, l64dcon320:
|
||
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
case l64dcon3220:
|
||
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
case l64dcon1232s:
|
||
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
case l64dcon20s32:
|
||
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)}
|
||
case l64dcon3212u:
|
||
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
default:
|
||
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
|
||
}
|
||
}
|
||
|
||
// encodeLOONG64LoadImm loads an immediate into a register, matching the
|
||
// toolchain's MOVV/MOVW case 3/19/25/59 expansion:
|
||
//
|
||
// $0: or rd, r0, r0 (MOVW: sll.w rd, r0, r0)
|
||
// 1..0xfff: ori rd, r0, imm
|
||
// −2048..−1: addi.d rd, r0, imm
|
||
// 32-bit (low 12 zero): lu12i.w rd, imm>>12
|
||
// 32-bit: lu12i.w rd, imm>>12; ori rd, rd, imm
|
||
// 64-bit: lu12i.w rd, imm>>12; ori rd, rd, imm;
|
||
// lu32i.d rd, imm>>32; lu52i.d rd, rd, imm>>52
|
||
func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte {
|
||
if v == 0 {
|
||
// The zero constant matches the register-form optab entry: MOVV →
|
||
// or rd, r0, r0, MOVW → sll.w rd, r0, r0.
|
||
op := l64movRegTable["MOVV"].op
|
||
if mnem == "MOVW" {
|
||
op = l64movRegTable["MOVW"].op
|
||
}
|
||
return l64wordLE(l64rrr(op, 0, 0, rd))
|
||
}
|
||
if v > 0 && v <= 0xfff {
|
||
return l64wordLE(l64irr(l64DualTable["OR"].imm, int(v), 0, rd))
|
||
}
|
||
if v >= -2048 && v < 0 {
|
||
// Both MOVV and MOVW use addi.d for negative constants.
|
||
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(v), 0, rd))
|
||
}
|
||
if v == int64(int32(v)) {
|
||
if v&0xfff == 0 {
|
||
return l64wordLE(l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd))
|
||
}
|
||
return l64WordsLE(
|
||
l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd),
|
||
l64irr(l64DualTable["OR"].imm, int(v), rd, rd),
|
||
)
|
||
}
|
||
// 64-bit constants use the shortest materialisation the bit pattern
|
||
// admits (dcon classification).
|
||
return l64WordsLE(l64DconMovWords(rd, v)...)
|
||
}
|
||
|
||
// encodeLOONG64MemOp encodes a memory load (load = true) or store with a
|
||
// 12-bit offset, or the 3-instruction expansion for larger offsets:
|
||
// lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30).
|
||
func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo) ([]byte, error) {
|
||
rj, off := l64MemWithFrame(mem, fi)
|
||
if rj < 0 {
|
||
return nil, fmt.Errorf("invalid memory operand")
|
||
}
|
||
ls, ok := l64loadStoreTable[mnem]
|
||
if !ok {
|
||
return nil, fmt.Errorf("unsupported MOV width %q", mnem)
|
||
}
|
||
op := ls.st
|
||
if load {
|
||
op = ls.ld
|
||
}
|
||
if off >= -2048 && off < 2048 {
|
||
return l64wordLE(l64irr(op, int(off), rj, reg)), nil
|
||
}
|
||
// Large offset: materialise the base in R30 (the assembler temp).
|
||
return l64WordsLE(
|
||
l64ir(l64InstrTable["LU12IW"].op, int((off+0x800)>>12), 30),
|
||
l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30),
|
||
l64irr(op, int(off), 30, reg),
|
||
), nil
|
||
}
|
||
|
||
// encodeLOONG64RegMove encodes a register-to-register move: the width
|
||
// extensions (ext.w.b, ext.w.h, sll.w, or, andi, bstrpick.d) between GPRs,
|
||
// fmov between F registers, and the special moves across the GPR/FP/FCC/FCSR
|
||
// banks.
|
||
func encodeLOONG64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
|
||
rs, rd := l64Reg(src), l64Reg(dst)
|
||
if rs < 0 || rd < 0 {
|
||
return nil, fmt.Errorf("invalid register operand")
|
||
}
|
||
sc, dc := loong64RegClass(operandRegName(src)), loong64RegClass(operandRegName(dst))
|
||
|
||
// FP-bank specials (MOVV/MOVW between GPR/FCC/FCSR and F registers).
|
||
if key, ok := l64FpMoveKey(mnem, sc, dc); ok {
|
||
op, ok := l64FpMovTable[key]
|
||
if !ok {
|
||
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
|
||
}
|
||
return l64wordLE(l64rr(op, rs, rd)), nil
|
||
}
|
||
|
||
// GPR → GPR.
|
||
if sc == l64ClsGR && dc == l64ClsGR {
|
||
switch mnem {
|
||
case "MOVHU":
|
||
// bstrpick.d rd, rj, $15, $0
|
||
return l64wordLE(l64irir(0x3<<22, 15, rs, 0, rd)), nil
|
||
case "MOVWU":
|
||
// bstrpick.d rd, rj, $31, $0
|
||
return l64wordLE(l64irir(0x3<<22, 31, rs, 0, rd)), nil
|
||
}
|
||
if e, ok := l64movRegTable[mnem]; ok {
|
||
if e.rr {
|
||
return l64wordLE(l64rr(e.op, rs, rd)), nil
|
||
}
|
||
if e.imm != 0 {
|
||
return l64wordLE(l64irr(e.op, e.imm, rs, rd)), nil
|
||
}
|
||
// 3R with rk = r0: or rd, rj, r0 / sll.w rd, rj, r0.
|
||
return l64wordLE(l64rrr(e.op, 0, rs, rd)), nil
|
||
}
|
||
}
|
||
|
||
// F → F.
|
||
if sc == l64ClsFP && dc == l64ClsFP {
|
||
if op, ok := l64movFpRegTable[mnem]; ok {
|
||
return l64wordLE(l64rr(op, rs, rd)), nil
|
||
}
|
||
}
|
||
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
|
||
}
|
||
|
||
// l64FpMoveKey builds the l64FpMovTable key for a cross-bank move, reporting
|
||
// whether the move is a cross-bank special at all.
|
||
func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) {
|
||
bank := func(c l64RegClass) string {
|
||
switch c {
|
||
case l64ClsFP:
|
||
return "F"
|
||
case l64ClsFCC:
|
||
return "FCC"
|
||
case l64ClsFCSR:
|
||
return "FCSR"
|
||
default:
|
||
return "R"
|
||
}
|
||
}
|
||
if sc == dc {
|
||
return "", false
|
||
}
|
||
if mnem != "MOVV" && mnem != "MOVW" {
|
||
return "", false
|
||
}
|
||
key := mnem + "." + bank(sc) + "." + bank(dc)
|
||
_, ok := l64FpMovTable[key]
|
||
return key, ok
|
||
}
|
||
|
||
// ---- static symbol references (pcalau12i + offset) ----
|
||
|
||
// encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the
|
||
// R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address.
|
||
func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
|
||
if relocs != nil {
|
||
*relocs = append(*relocs,
|
||
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
|
||
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
|
||
)
|
||
}
|
||
return l64WordsLE(
|
||
l64ir(l64InstrTable["PCALAU12I"].op, 0, rd),
|
||
l64irr(l64DualTable["ADDV"].imm, 0, rd, rd),
|
||
)
|
||
}
|
||
|
||
// encodeLOONG64SBLoad emits pcalau12i r30, 0; ld rd, 0(r30) with the
|
||
// R_LOONG64_ADDR_HI/LO pair, loading from a static symbol.
|
||
func encodeLOONG64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte {
|
||
ls := l64loadStoreTable[mnem]
|
||
if relocs != nil {
|
||
*relocs = append(*relocs,
|
||
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
|
||
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
|
||
)
|
||
}
|
||
return l64WordsLE(
|
||
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
|
||
l64irr(ls.ld, 0, 30, rd),
|
||
)
|
||
}
|
||
|
||
// encodeLOONG64SBStore emits pcalau12i r30, 0; st rd, 0(r30) with the
|
||
// R_LOONG64_ADDR_HI/LO pair, storing to a static symbol.
|
||
func encodeLOONG64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) []byte {
|
||
ls := l64loadStoreTable[mnem]
|
||
if relocs != nil {
|
||
*relocs = append(*relocs,
|
||
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
|
||
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
|
||
)
|
||
}
|
||
return l64WordsLE(
|
||
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
|
||
l64irr(ls.st, 0, 30, rs),
|
||
)
|
||
}
|
||
|
||
// ---- operand helpers ----
|
||
|
||
// l64IndexedTable holds the register-indexed load/store (ldx/stx) opcodes.
|
||
var l64IndexedTable = map[string]struct{ ld, st uint32 }{
|
||
"MOVB": {0x07000 << 15, 0x07020 << 15},
|
||
"MOVH": {0x07008 << 15, 0x07028 << 15},
|
||
"MOVW": {0x07010 << 15, 0x07030 << 15},
|
||
"MOVV": {0x07018 << 15, 0x07038 << 15},
|
||
"MOVBU": {0x07040 << 15, 0x07020 << 15},
|
||
"MOVHU": {0x07048 << 15, 0x07028 << 15},
|
||
"MOVWU": {0x07050 << 15, 0x07030 << 15},
|
||
"MOVF": {0x07060 << 15, 0x07070 << 15},
|
||
"MOVD": {0x07068 << 15, 0x07078 << 15},
|
||
}
|
||
|
||
// operandRegName returns the register name of an operand, or "".
|
||
func operandRegName(op *ast.Operand) string {
|
||
if op.Addr.Base != "" {
|
||
return op.Addr.Base
|
||
}
|
||
if op.Addr.Sym != nil && op.Addr.Sym.Name != "" {
|
||
return op.Addr.Sym.Name
|
||
}
|
||
return ""
|
||
}
|
||
|
||
// l64Reg returns the register number of an operand, or -1.
|
||
func l64Reg(op *ast.Operand) int {
|
||
return loong64RegNum(operandRegName(op))
|
||
}
|
||
|
||
// l64Imm64 returns the full 64-bit immediate value of an operand.
|
||
func l64Imm64(op *ast.Operand) int64 {
|
||
if op.Imm.HasVal {
|
||
v := op.Imm.Val
|
||
if op.Imm.Neg {
|
||
v = -v
|
||
}
|
||
return v
|
||
}
|
||
return 0
|
||
}
|
||
|
||
// l64Mem returns the base register and byte offset of a memory operand.
|
||
func l64Mem(op *ast.Operand) (rj int, off int32) {
|
||
rj = loong64RegNum(op.Addr.Base)
|
||
off = int32(op.Addr.Offset)
|
||
return
|
||
}
|
||
|
||
// l64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
|
||
// registers via the frame mapping.
|
||
func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
|
||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
|
||
return loong64ResolvePseudo(op.Addr.Sym, fi)
|
||
}
|
||
return l64Mem(op)
|
||
}
|
||
|
||
// l64MemOffset returns the resolved byte offset of a memory operand.
|
||
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
|
||
_, off := l64MemWithFrame(op, fi)
|
||
return off
|
||
}
|
||
|
||
// l64Label returns the label name of an operand.
|
||
func l64Label(op *ast.Operand) string {
|
||
if op.Addr.Sym != nil {
|
||
return op.Addr.Sym.Name
|
||
}
|
||
return op.Raw
|
||
}
|