Files
gasm-sdk/asm/loong64_assemble.go
2026-09-20 19:14:43 +02:00

2320 lines
73 KiB
Go
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"math/bits"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and
// the immediate-arithmetic forms expand to 2-5 instructions when the
// immediate does not fit, so the layout is computed in two passes (sizes,
// then encoding with resolved branch targets).
//
// The emitted bytes match the Go toolchain's loong64 assembler, which is the
// ground-truth oracle: prologue/epilogue (including the large-frame R30
// materialisations), FP/SP frame mapping, the stack-split guard classes, and
// branch encodings all follow cmd/internal/obj/loong64. The morestack block
// at the end of split functions carries the runtime.morestack_noctxt call.
func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := loong64ComputeFrame(t)
prologue := loong64Prologue(fi)
guardLen := loong64GuardLen(fi)
chain := loong64JumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
return r
}
return name
}
var relocs []Reloc
var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported
// after the SP adjust instruction, exactly as the toolchain's pctospadj
// does. The prologue (3 instructions when a frame is present) may
// materialise its store or adjust through R30, which widens it.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
}
// The toolchain's parser counts N(PC) displacements over the source
// instructions at a uniform 4 bytes each, so a PC-relative branch
// resolves to the instruction N slots away in body order; the resolved
// target then participates in layout and loop-head padding like any
// branch target.
instrs := make([]*ast.Instr, 0, len(t.Body))
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok && strings.ToUpper(in.Mnemonic.Text) != "PCALIGN" {
instrs = append(instrs, in)
}
}
parseIndex := make(map[*ast.Instr]int, len(instrs))
for i, in := range instrs {
parseIndex[in] = i
}
pcRelTarget := make(map[*ast.Instr]*ast.Instr)
for _, in := range instrs {
off, ok := loong64PCRelOffset(in)
if !ok {
continue
}
tgt := parseIndex[in] + off
if tgt < 0 || tgt >= len(instrs) {
continue
}
pcRelTarget[in] = instrs[tgt]
}
// Pass 1: label offsets from the instruction sizes. PCALIGN contributes
// only its padding. On top of the explicit PCALIGNs, the toolchain pads
// every backward-branch target (loop head) to a 16-byte boundary, so the
// layout runs to a fixpoint over the alignment set.
loopAligns := map[string]bool{}
alignInstrs := map[*ast.Instr]bool{}
for {
offsets, _, pcs, _ := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
changed := false
for _, in := range instrs {
// A backward PC-relative target is the resolved instruction.
if tgt, ok := pcRelTarget[in]; ok && pcs[tgt] < pcs[in] && !alignInstrs[tgt] {
alignInstrs[tgt] = true
changed = true
}
target, ok := loong64BranchTarget(in)
if !ok {
continue
}
tOff, ok := offsets[target]
if !ok || tOff >= pcs[in] || loopAligns[target] {
continue
}
loopAligns[target] = true
changed = true
}
if !changed {
break
}
}
// Final layout with the complete alignment set.
offsets, alignPad, pcs, bodyEnd := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs)
bodyLen := bodyEnd - (guardLen + len(prologue))
pcRelPcs := make(map[*ast.Instr]int, len(pcRelTarget))
for in, tgt := range pcRelTarget {
pcRelPcs[in] = pcs[tgt]
}
var out []byte
if fi.needSplit {
out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
// PCALIGN pads to the requested boundary with andi $0, $0, 0, the
// architecture's NOP, and encodes to nothing itself.
if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" {
pad := loong64PCAlignPad(pc, in)
out = append(out, loong64PadBytes(pad)...)
pc += pad
continue
}
// Loop-head alignment padding precedes the instruction.
if pad := alignPad[in]; pad > 0 {
out = append(out, loong64PadBytes(pad)...)
pc += pad
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve, pcRelPcs)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero
// after the frame-deallocating ADDV.
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + loong64EpilogueWords(fi)*4, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
block, blReloc := loong64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil
}
// loong64PCRelOffset reports the N of a branch operand spelled N(PC): the
// displacement counted in source instructions from the branch itself.
func loong64PCRelOffset(instr *ast.Instr) (int, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
branch := false
switch mnem {
case "JMP":
branch = len(instr.Operands) == 1
case "JAL", "CALL", "BL":
branch = len(instr.Operands) == 1 || len(instr.Operands) == 2
case "BFPT", "BFPF":
branch = len(instr.Operands) == 1
case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU",
"BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ":
branch = len(instr.Operands) >= 2
}
if !branch {
return 0, false
}
op := instr.Operands[len(instr.Operands)-1]
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" {
return int(op.Addr.Offset), true
}
return 0, false
}
// loong64Layout walks the function body once and returns the label offsets,
// the loop-alignment padding due before each instruction (a pad of 0 needs
// nothing), the pc each instruction starts at (its padding included) and the
// first pc past the body. Explicit PCALIGN pads, the alignment pads for the
// labels in aligns and those for the instructions in alignInstrs (backward
// PC-relative targets) all contribute, mirroring the toolchain's layout
// pass.
func loong64Layout(t *ast.Text, start int, fi loong64FrameInfo, aligns map[string]bool, alignInstrs map[*ast.Instr]bool) (map[string]int, map[*ast.Instr]int, map[*ast.Instr]int, int) {
offsets := map[string]int{}
alignPad := map[*ast.Instr]int{}
pcs := map[*ast.Instr]int{}
pos := start
pendingAlign := false
var pendingNames []string
explicit := false
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
if aligns[s.Name.Text] {
pendingAlign = true
}
pendingNames = append(pendingNames, s.Name.Text)
// Provisional: a branch to the label lands here unless a loop
// alignment pad follows, in which case the label resolves to the
// padded instruction (the toolchain's labels bind to the branch
// target instruction, which the padding pass precedes).
offsets[s.Name.Text] = pos
case *ast.Instr:
if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" {
pos += loong64PCAlignPad(pos, s)
explicit = true
continue
}
if pendingAlign {
pendingAlign = false
if pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
}
if alignInstrs[s] && pos&15 != 0 {
alignPad[s] = 16 - pos&15
}
if !explicit {
for _, n := range pendingNames {
offsets[n] = pos + alignPad[s]
}
}
pendingNames = nil
explicit = false
pcs[s] = pos + alignPad[s]
pos += alignPad[s] + loong64InstrSize(s, fi)
}
}
return offsets, alignPad, pcs, pos
}
// loong64BranchTarget reports the local label a branch-like instruction
// transfers to, the loop-head signal the toolchain derives from backward
// branch targets.
func loong64BranchTarget(instr *ast.Instr) (string, bool) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var op *ast.Operand
switch {
case mnem == "JMP" || mnem == "JAL" || mnem == "BFPT" || mnem == "BFPF":
if len(ops) != 1 {
return "", false
}
op = ops[0]
case mnem == "TEQ" || mnem == "TNE":
return "", false
case len(ops) >= 2:
op = ops[len(ops)-1]
default:
return "", false
}
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
// branch-chasing pass: a label whose first instruction is an unconditional
// local jump redirects its own jumpers to the ultimate target. The Go
// toolchain chases these chains before it encodes branches, so matching its
// bytes requires the same redirection.
func loong64JumpChain(t *ast.Text) map[string]string {
leadsTo := map[string]string{}
for i, stmt := range t.Body {
l, ok := stmt.(*ast.Label)
if !ok {
continue
}
j := i + 1
for j < len(t.Body) {
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
break
}
j++
}
if j >= len(t.Body) {
continue
}
in, ok := t.Body[j].(*ast.Instr)
if !ok {
continue
}
mnem := strings.ToUpper(in.Mnemonic.Text)
if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 {
continue
}
if name, ok := l64LabelOK(in.Operands[0]); ok {
leadsTo[l.Name.Text] = name
}
}
chain := map[string]string{}
for name := range leadsTo {
visited := map[string]bool{name: true}
cur := name
for {
next, ok := leadsTo[cur]
if !ok || visited[next] {
break
}
visited[next] = true
cur = next
}
if cur != name {
chain[name] = cur
}
}
return chain
}
// l64LabelOK returns the local label name of a jump operand.
func l64LabelOK(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// l64SubToAdd rewrites the SUB family with an immediate first operand onto
// its ADD counterpart with the negated immediate: LoongArch has no
// subtract-immediate instructions, and the toolchain folds SUB $v into the
// ADD immediate form through the same optab matching (the $0 fold into 3R
// and the large-constant materialisations included). The negation is the
// second result; the operand is left untouched because the size pass
// normalises the same instruction.
func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) {
if len(ops) >= 2 && isImmOperand(ops[0]) {
switch mnem {
case "SUB":
return "ADD", true
case "SUBW":
return "ADDW", true
case "SUBV", "SUBVU":
return "ADDV", true
}
}
return mnem, false
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
var neg bool
mnem, neg = l64SubToAdd(mnem, ops)
if mnem == "RET" {
return len(loong64Return(fi))
}
switch mnem {
case "END", "FUNCDATA", "PCDATA":
return 0 // bookkeeping statements contribute no bytes
case "GETCALLERPC":
return 4 // or rd, r1, r0
case "TEQ", "TNE":
return 8 // bne/beq over the BREAK, then BREAK
case "PRELDX":
return 20 // the four-instruction constant materialisation + preldx
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := l64Imm64(ops[0])
if neg {
v = -v
}
if v == 0 {
return 4 // folds into the 3R form (rk = R0)
}
switch mnem {
case "ADD", "ADDW", "ADDV", "ADDVU", "SGT", "SGTU":
// C_US12CON (−2048..0x7ff) encodes directly as addi/slti.
if v >= -2048 && v <= 0x7ff {
return 4
}
// C_U12CON (0x800..0xfff) → ori r30, r0, v; op rd, rj, r30.
if v >= 0x800 && v <= 0xfff {
return 8
}
default: // AND/OR/XOR
// C_UU12CON (0..0x7ff) encodes directly as andi/ori/xori.
if v >= 0 && v <= 0x7ff {
return 4
}
// C_S12CON (−2048..−1) → addi.d r30, r0, v; op rd, rj, r30.
if v >= -2048 && v < 0 {
return 8
}
}
// 0x800..0xfff for AND/OR/XOR and the 32/64-bit ranges go through
// the lu12i.w materialisation.
if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
return 8 // lu12i.w r30, v>>12; op rd, rj, r30
}
return 12 // lu12i.w r30, v>>12; ori r30, r30, v; op rd, rj, r30
}
return 4 * (len(l64DconMovWords(0, v)) + 1) // dcon materialisation + op
}
}
return 4
}
// loong64PCAlignPad returns the padding PCALIGN inserts before the next
// instruction so that it starts at the requested boundary relative to the
// function start. The boundary must be a power of two between 8 and 2048, as
// the toolchain requires; anything else pads nothing.
func loong64PCAlignPad(pos int, instr *ast.Instr) int {
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
return 0
}
align := int(immFromOperand(instr.Operands[0]))
if align < 8 || align > 2048 || align&(align-1) != 0 {
return 0
}
return (align - pos%align) % align
}
// loong64PadBytes renders PCALIGN padding: the toolchain emits andi $0, $0, 0
// (the architecture's NOP) for every full 4 bytes of pad.
func loong64PadBytes(pad int) []byte {
nop := l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0))
out := make([]byte, 0, pad/4*len(nop))
for i := 0; i < pad/4; i++ {
out = append(out, nop...)
}
return out
}
// encodeLOONG64Instr encodes a single LoongArch instruction.
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
// The SUB family with an immediate first operand folds onto the ADD
// immediate form with the negated immediate; the negation happens on a
// copy of the operand, never on the shared syntax tree.
mnem, neg := l64SubToAdd(mnem, ops)
if neg {
c := *ops[0]
c.Imm.Val = -c.Imm.Val
ops2 := make([]*ast.Operand, len(ops))
ops2[0] = &c
copy(ops2[1:], ops[1:])
ops = ops2
}
// Pseudo-instructions and the branches first.
switch mnem {
case "RET":
return loong64Return(fi), nil
case "NOP", "NOOP":
// andi r0, r0, 0
return l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)), nil
case "UNDEF":
// break 0
return l64wordLE(l64i15(l64InstrTable["BREAK"].op, 0)), nil
case "WORD":
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "END", "FUNCDATA", "PCDATA", "GETCALLERPC":
// The assembler's bookkeeping statements. END, FUNCDATA and PCDATA
// contribute no bytes, the same shapes GOARCH=loong64 go tool asm
// accepts and emits nothing for; GETCALLERPC reads the caller's
// address out of R1 (RA) as or rd, r1, r0.
switch mnem {
case "END":
if len(ops) != 0 {
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
}
return nil, nil
case "FUNCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
}
return nil, nil
case "PCDATA":
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
return nil, fmt.Errorf("PCDATA expects $n, $n")
}
return nil, nil
}
if len(ops) != 1 || isMemOperand(ops[0]) || loong64RegClass(operandRegName(ops[0])) != l64ClsGR {
return nil, fmt.Errorf("GETCALLERPC expects a general register")
}
rd := loong64RegNum(operandRegName(ops[0]))
if rd < 0 {
return nil, fmt.Errorf("GETCALLERPC: invalid register operand")
}
return l64wordLE(l64rrr(l64movRegTable["MOVV"].op, 0, 1, rd)), nil
case "NEGW", "NEGV":
// The integer negation pseudo is a subtract from zero:
// NEGW src, dst → sub.w r0, src, dst.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
src, dst := l64Reg(ops[0]), l64Reg(ops[1])
if src < 0 || dst < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sub := l64InstrTable["SUBW"].op
if mnem == "NEGV" {
sub = l64InstrTable["SUBV"].op
}
return l64wordLE(l64rrr(sub, src, 0, dst)), nil
case "TEQ", "TNE":
// The trap pseudo expands to two instructions: bne/beq rj, rd over
// the BREAK (offset 2 instruction units), then BREAK $code.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
code := int(immFromOperand(ops[0]))
rj, rd := 0, l64Reg(ops[len(ops)-1])
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
bop := l64branchTable["BNE"]
if mnem == "TNE" {
bop = l64branchTable["BEQ"]
}
return l64WordsLE(
l64irr16(bop, 2, rj, rd),
l64i15(l64InstrTable["BREAK"].op, code),
), nil
case "PRELDX":
// preldx offset(Rbase), $n, $hint: the 64-bit descriptor n packs
// (addrSeq, blockSize, blockNums, stride); the constant v built from
// it materialises in R30 across four instructions, then the preldx.
if len(ops) != 3 || !isMemOperand(ops[0]) || !isImmOperand(ops[1]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("PRELDX expects offset(reg), $n, $hint")
}
rj := loong64RegNum(ops[0].Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
n := uint64(l64Imm64(ops[1]))
hint := int(l64Imm64(ops[2]))
addrSeq := (n >> 0) & 0x1
blkSize := (n >> 1) & 0x7ff
blkNums := (n >> 12) & 0x1ff
stride := (n >> 21) & 0xffff
v := uint64(ops[0].Addr.Offset)&0xffff + addrSeq<<16 +
((blkSize/16)-1)<<20 + (blkNums-1)<<32 + stride<<44
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
ori = 0x00e << 22
preldx = 0x7058 << 15
)
return l64WordsLE(
l64ir(lu12iw, int(uint32(v>>12)), 30),
l64irr(ori, int(uint32(v)), 30, 30),
l64ir(lu32id, int(uint32(v>>32)), 30),
l64irr(lu52id, int(uint32(v>>52)), 30, 30),
l64rrr(preldx, 30, rj, hint),
), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
}
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
if op, ok := l64branchTable[mnem]; ok {
if mnem == "JIRL" {
return encodeLOONG64Jirl(op, ops)
}
return encodeLOONG64Branch16(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
}
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
if op, ok := l64branch21Table[mnem]; ok {
return encodeLOONG64Branch21(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs)
}
// B/BL aliases reached only via JMP/JAL above.
// The dual-form arithmetic mnemonics: register (3R) or immediate (2RI12).
if de, ok := l64DualTable[mnem]; ok {
if len(ops) >= 2 && isImmOperand(ops[0]) {
if de.shift {
// INSTR $shamt, rd or INSTR $shamt, rj, rd.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
shamt := int(immFromOperand(ops[0]))
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// $0 folds into the register form (the toolchain matches the
// zero constant against the 3R optab entry first).
if shamt == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
// The .d variants take a 6-bit amount, the .w variants 5 bits.
if isLoong64ShiftD(de.imm) {
shamt &= 0x3f
} else {
shamt &= 0x1f
}
return l64wordLE(l64irr(de.imm, shamt, rj, rd)), nil
}
return encodeLOONG64ImmArith(mnem, de, ops)
}
// Register form: 3R.
if len(ops) == 3 {
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(de.rrr, rk, rj, rd)), nil
}
if len(ops) == 2 {
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rk < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(de.rrr, rk, rd, rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
// the integer/FP table (their mnemonics overlap the table's 2R format
// but resolve vector-bank registers).
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := l64InstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
}
switch enc.format {
case l64Frrr:
// INSTR rk, rj, rd (3 operands) or INSTR rk, rd (rj = rd).
switch len(ops) {
case 3:
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
case 2:
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rk < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(enc.op, rk, rd, rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 register operands, got %d", mnem, len(ops))
case l64Frr:
// INSTR rj, rd.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
imm := int(immFromOperand(ops[0]))
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irr(enc.op, imm, rj, rd)), nil
case l64Firr16:
// ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be
// a multiple of 65536 and is shifted right by 16.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
v := int(immFromOperand(ops[0]))
if v&0xFFFF != 0 {
return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem)
}
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irr16(enc.op, v>>16, rj, rd)), nil
case l64Firr14:
// LL/SC/MOVWP: INSTR mem, rd (load) or INSTR rd, mem (store); the
// 14-bit offset is scaled by 4 (byte offset >> 2).
rd, rj, off, load, err := l64MemOperands(ops, fi)
if err != nil {
return nil, err
}
op := enc.op
if load && (mnem == "MOVWP" || mnem == "MOVVP") {
// ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field.
op -= 1 << 24
}
return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil
case l64Fir20:
// LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rd := l64Reg(ops[0])
if rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64ir(enc.op, int(immFromOperand(ops[1])), rd)), nil
case l64Frrrr:
// FMADD/FMSUB/FNMADD/FNMSUB: INSTR fa, fk, fj, fd (4 operands) or
// INSTR fa, fk, fd (fj = fd).
fa, fk, fj, fd, err := l64FmaOperands(ops)
if err != nil {
return nil, err
}
return l64wordLE(l64rrrr(enc.op, fa, fk, fj, fd)), nil
case l64Firir:
// BSTRINS/BSTRPICK: INSTR $msb, rj, $lsb, rd (or $msb, rj, rd with
// lsb = 0).
if len(ops) != 4 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops))
}
msb := int(immFromOperand(ops[0]))
lsb := 0
rj := l64Reg(ops[1])
rd := l64Reg(ops[len(ops)-1])
if len(ops) == 4 {
lsb = int(immFromOperand(ops[2]))
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// The toolchain validates the bit numbers ("illegal bit number"):
// 0..31 for the .w forms, 0..63 for the .d forms, lsb <= msb.
b := 64
if strings.HasSuffix(mnem, "W") {
b = 32
}
if msb < 0 || msb >= b || lsb < 0 || lsb >= b || lsb > msb {
return nil, fmt.Errorf("%s: illegal bit number (msb %d, lsb %d)", mnem, msb, lsb)
}
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
case l64Firrr:
// ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in
// the second register position); the source amount is 1-4, encoded
// as sa-1.
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
sa := int(immFromOperand(ops[0])) - 1
rj, rk, rd := l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
if sa < 0 || sa > 3 {
return nil, fmt.Errorf("shift amount out of range [1, 4]")
}
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irrr(enc.op, sa, rk, rj, rd)), nil
case l64Fi15:
// SYSCALL/BREAK/DBAR: no operands, or SYSCALL $code / BREAK $code.
code := 0
if len(ops) == 1 {
code = int(immFromOperand(ops[0]))
} else if len(ops) > 1 {
return nil, fmt.Errorf("%s expects at most 1 operand, got %d", mnem, len(ops))
}
return l64wordLE(l64i15(enc.op, code)), nil
case l64Fam:
// AM* val, (addr), result.
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rk := l64Reg(ops[0])
rj, _ := l64Mem(ops[1])
rd := l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
case l64Frdtime:
// RDTIME* rd, rj (rd at bits [9:5], rj at bits [4:0]).
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rd, rj := l64Reg(ops[0]), l64Reg(ops[1])
if rd < 0 || rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rr(enc.op, rd, rj)), nil
case l64Fpreld:
// PRELD off(rj), $hint.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rj, off := l64Mem(ops[0])
hint := int(immFromOperand(ops[1]))
if rj < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return l64wordLE(l64irr5i(enc.op, int(off), rj, hint)), nil
}
return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops))
}
// encodeLOONG64Branch encodes a label or indirect jump/call:
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
}
op := instr.Operands[0]
// PC-relative displacement: N(PC) resolves to the instruction N slots
// away in source order (the toolchain's parse-time count), and the field
// carries the final pc distance in instruction units.
if op.Addr.Sym == nil && op.Addr.Base == "PC" {
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, op.Addr.Offset)
}
v := (targetPc - pc) >> 2
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
// Indirect: (rj) → jirl.
rj := loong64RegNum(op.Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
rd := 0
if link {
rd = 1 // link register
}
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
}
// Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation
// (the linker fills the offset), as the toolchain does for CALL/BL/JAL
// and for tail-calling JMP.
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
opc := l64jumpTable["B"]
if link {
opc = l64jumpTable["BL"]
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset})
}
return l64wordLE(l64bbl(opc, 0)), nil
}
// Direct: label → b/bl.
target := resolve(l64Label(op))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
if v < -1<<25 || v >= 1<<25 {
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
}
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
// encodeLOONG64Jirl encodes the raw JIRL spelling, JIRL rd, rj, offset, the
// form the verify trampolines use. The (rj) indirect form without an offset
// is handled by encodeLOONG64Branch.
func encodeLOONG64Jirl(op uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("JIRL expects 3 operands, got %d", len(ops))
}
rd := l64Reg(ops[0])
rj := l64Reg(ops[1])
if rd < 0 || rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
off, ok := l64offsetOperand(ops[2])
if !ok {
return nil, fmt.Errorf("JIRL expects an immediate offset, got %q", ops[2].Raw)
}
if (int64(off)<<16)>>16 != int64(off) {
return nil, fmt.Errorf("JIRL offset %d out of the 16-bit range", off)
}
return l64wordLE(l64irr16(op, int(off), rj, rd)), nil
}
// l64offsetOperand reads a bare numeric branch offset: an immediate ($n) or a
// plain number, which parses as an empty address carrying the digits in Raw.
func l64offsetOperand(op *ast.Operand) (int32, bool) {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), true
}
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "" && op.Addr.Index == "" {
if v, err := strconv.ParseInt(op.Raw, 0, 64); err == nil {
return int32(v), true
}
}
return 0, false
}
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
func encodeLOONG64Branch16(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
var target string
var v int
lastOp := ops[len(ops)-1]
if lastOp.Kind == ast.OpAddr && lastOp.Addr.Sym == nil && lastOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, lastOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target = resolve(l64Label(lastOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
if len(ops) == 2 {
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
// BGEZ-family aliases encoded with rj in the rj field.
rj := l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if mnem == "BLTU" || mnem == "BGEU" {
// The unsigned compares have no single-register pseudo: the
// toolchain keeps the register-register form with rd = R0
// (bltu rj, r0 is never taken), not a sometimes-taken beqz.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, rj, 0)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
zop := l64branch21Table["BEQZ"]
if mnem == "BNE" {
zop = l64branch21Table["BNEZ"]
}
if mnem == "BLT" || mnem == "BLTZ" || mnem == "BGTZ" {
zop = l64branch21Table["BLTZ"]
}
if mnem == "BGE" || mnem == "BGEZ" || mnem == "BLEZ" {
zop = l64branch21Table["BGEZ"]
}
return l64wordLE(l64ir21(zop, v, rj)), nil
}
// Two registers: BEQ rj, rd, label. When one is R0 the toolchain
// re-encodes as the 21-bit BEQZ/BNEZ form.
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if rj == 0 {
rj, rd = rd, 0
}
if rd == 0 {
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
zop := l64branch21Table["BEQZ"]
if mnem == "BNE" {
zop = l64branch21Table["BNEZ"]
}
return l64wordLE(l64ir21(zop, v, rj)), nil
}
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, rj, rd)), nil
}
// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) {
isBF := mnem == "BFPT" || mnem == "BFPF"
if len(ops) != 2 && !(isBF && (len(ops) == 1 || len(ops) == 2)) {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
var rj int
tgtOp := ops[len(ops)-1]
if isBF {
// BFPT/BFPF test an FCC condition register, defaulting to FCC0 when
// spelled without one.
rj = 0
if len(ops) == 2 {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
} else {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
var v int
if tgtOp.Kind == ast.OpAddr && tgtOp.Addr.Sym == nil && tgtOp.Addr.Base == "PC" {
// N(PC) resolves to the instruction N slots away in source order.
targetPc, ok := pcRelPcs[instr]
if !ok {
return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, tgtOp.Addr.Offset)
}
v = (targetPc - pc) >> 2
} else {
target := resolve(l64Label(tgtOp))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v = (targetOff - pc) >> 2
}
if mnem == "BGTZ" || mnem == "BLEZ" {
// The toolchain swaps the register into the rd field and keeps the
// 16-bit offset form.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch %d too far (16-bit range)", v)
}
return l64wordLE(l64irr16(op, v, 0, rj)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch %d too far (21-bit range)", v)
}
return l64wordLE(l64ir21(op, v, rj)), nil
}
// encodeLOONG64ImmArith encodes an immediate arithmetic/logic instruction,
// expanding the immediate exactly as the toolchain's aclass classifies it:
//
// ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes)
// 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8)
// AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4)
// −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8)
// 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12)
// 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20)
func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := l64Imm64(ops[0])
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// The two immediate families classify differently.
additive := mnem == "ADD" || mnem == "ADDW" || mnem == "ADDV" || mnem == "ADDVU" || mnem == "SGT" || mnem == "SGTU"
if additive {
if v == 0 {
// $0 folds into the 3R form (rk = R0), matching the toolchain's
// optab matching of the zero constant against the register form.
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= -2048 && v <= 0x7ff {
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= 0x800 && v <= 0xfff {
return l64WordsLE(
l64irr(0x00e<<22, int(v), 0, 30), // ori r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
} else {
if v == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= 0 && v <= 0x7ff {
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= -2048 && v < 0 {
return l64WordsLE(
l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
}
// 32/64-bit constants are materialised in R30 (the assembler temp),
// using the same dcon classification as the toolchain's case 24/60/70/
// 71/72 sequences.
const (
lu12iw = 0x0a << 25
ori = 0x00e << 22
)
if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
return l64WordsLE(
l64ir(lu12iw, int(int32(v)>>12), 30),
l64rrr(de.rrr, 30, rj, rd),
), nil
}
return l64WordsLE(
l64ir(lu12iw, int(int32(v)>>12), 30),
l64irr(ori, int(v), 30, 30),
l64rrr(de.rrr, 30, rj, rd),
), nil
}
words := l64DconMovWords(30, v)
words = append(words, l64rrr(de.rrr, 30, rj, rd))
return l64WordsLE(words...), nil
}
// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of
// the 6-bit (.d) variants, the toolchain distinguishes them by the bit
// position of the opcode field (bits [25:16]).
func isLoong64ShiftD(op uint32) bool {
return op&0x03ff0000 != 0 && op>>25 == 0
}
// l64FmaOperands extracts the four fused-multiply-add operands:
// INSTR fa, fk, fj, fd, or INSTR fa, fk, fd with fj = fd.
func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) {
switch len(ops) {
case 4:
fa, fk, fj, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
case 3:
fa, fk, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
fj = fd
default:
return 0, 0, 0, 0, fmt.Errorf("expected 3 or 4 operands, got %d", len(ops))
}
if fa < 0 || fk < 0 || fj < 0 || fd < 0 {
return 0, 0, 0, 0, fmt.Errorf("invalid register operand")
}
return fa, fk, fj, fd, nil
}
// l64MemOperands extracts (rd, rj, off, load) from a load/store instruction:
// INSTR mem, rd is a load, INSTR rd, mem a store.
func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) {
if len(ops) != 2 {
return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops))
}
if isMemOperand(ops[0]) {
rd = l64Reg(ops[1])
rj, off = l64MemWithFrame(ops[0], fi)
load = true
} else if isMemOperand(ops[1]) {
rd = l64Reg(ops[0])
rj, off = l64MemWithFrame(ops[1], fi)
} else {
return 0, 0, 0, false, fmt.Errorf("expected a memory operand")
}
if rd < 0 || rj < 0 {
return 0, 0, 0, false, fmt.Errorf("invalid operand")
}
return rd, rj, off, load, nil
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width
// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the
// access width). The forms, mirroring the toolchain:
//
// MOVx $imm, rd load immediate (addi/lu12i+ori/lu32i/lu52i)
// MOVx mem, rd load from memory
// MOVx rd, mem store to memory
// MOVx rs, rd register move (incl. the FP-bank specials)
// MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d)
// MOVx sym(SB), rd load from a static symbol (pcalau12i+ld)
// MOVx rd, sym(SB) store to a static symbol (pcalau12i+st)
func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if mnem == "MOV" {
mnem = "MOVV"
}
src, dst := ops[0], ops[1]
// Immediate → register.
if isImmOperand(src) && !isMemOperand(src) {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
}
// MOVW $imm, Fd is the only immediate-to-F form the toolchain's optab
// accepts (AMOVW's C_12CON against C_FREG): it materialises the
// constant in R30 and moves it across with movgr2fr.w. MOVV/MOVF/
// MOVD are illegal combinations there, and are diagnosed here rather
// than silently written into the GPR of the register's number.
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
if mnem != "MOVW" {
return nil, fmt.Errorf("%s $imm: illegal combination with an F register destination (only MOVW $c, Fd is supported)", mnem)
}
return encodeLOONG64ImmToFp(rd, l64Imm64(src))
}
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
}
// Static symbol load/store via pcalau12i.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem)
}
return encodeLOONG64SBLoad(src.Addr.Sym, rd, mnem, relocs), nil
}
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) {
rs := l64Reg(src)
if rs < 0 {
return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem)
}
return encodeLOONG64SBStore(dst.Addr.Sym, rs, mnem, relocs), nil
}
// Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk).
if src.Addr.Index != "" && !isMemOperand(dst) {
rd := l64Reg(dst)
rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index)
if rd < 0 || rj < 0 || rk < 0 {
return nil, fmt.Errorf("%s (rj)(rk): invalid register operand", mnem)
}
op, ok := l64IndexedTable[mnem]
if !ok {
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
}
return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil
}
if dst.Addr.Index != "" && !isMemOperand(src) {
rs := l64Reg(src)
rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index)
if rs < 0 || rj < 0 || rk < 0 {
return nil, fmt.Errorf("%s rd, (rj)(rk): invalid register operand", mnem)
}
op, ok := l64IndexedTable[mnem]
if !ok {
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
}
return l64wordLE(l64rrr(op.st, rk, rj, rs)), nil
}
// Memory load/store with a 12-bit (or larger, via expansion) offset.
if isMemOperand(src) && !isMemOperand(dst) {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s: invalid destination register", mnem)
}
return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi)
}
if !isMemOperand(src) && isMemOperand(dst) {
rs := l64Reg(src)
if rs < 0 {
return nil, fmt.Errorf("%s: invalid source register", mnem)
}
return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi)
}
// Register → register.
return encodeLOONG64RegMove(mnem, src, dst)
}
// loong64MovSize returns the encoded size of a MOV instruction.
func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if mnem == "MOV" {
mnem = "MOVV"
}
if len(ops) != 2 {
return 4
}
src, dst := ops[0], ops[1]
switch {
case isImmOperand(src):
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
v := l64Imm64(src)
if v == 0 {
return 4
}
if v > 0 && v <= 0xfff {
return 4 // ori rd, r0, v
}
if v >= -2048 && v < 0 {
return 4 // addi.d rd, r0, v
}
if v == int64(int32(v)) {
if v&0xfff == 0 {
return 4 // lu12i.w
}
return 8 // lu12i.w + ori
}
return 4 * len(l64DconMovWords(0, v))
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
return 8 // pcalau12i + ld
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
return 8 // pcalau12i + st
case src.Addr.Index != "" || dst.Addr.Index != "":
return 4 // ldx/stx
case isMemOperand(src) || isMemOperand(dst):
// A 12-bit offset fits in one instruction; larger offsets expand
// to lu12i.w + add.d + the access.
mem := src
if !isMemOperand(src) {
mem = dst
}
if l64MemOffset(mem, fi) >= -2048 && l64MemOffset(mem, fi) < 2048 {
return 4
}
return 12
default:
return 4 // register move
}
}
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it
// to an F register, the toolchain's expansion of MOVW $c, Fd: ori (which
// zero-extends) for the positive span, addi.w for zero and the negative
// span, then movgr2fr.w. The toolchain's optab accepts no wider constant on
// this path (it never materialises one fully first), so values outside
// [-2048, 4095] are diagnosed rather than masked into si12.
func encodeLOONG64ImmToFp(fd int, v int64) ([]byte, error) {
if v < -2048 || v > 4095 {
return nil, fmt.Errorf("MOVW $%d: immediate out of the [-2048, 4095] range for an F register destination", v)
}
op := uint32(0x00a << 22) // addi.w r30, r0, v (sign-extends)
if v > 0 {
op = 0x00e << 22 // ori r30, r0, v (zero-extends)
}
return l64WordsLE(
l64irr(op, int(v), 0, 30),
l64rr(0x4529<<10, 30, fd), // movgr2fr.w fd, r30
), nil
}
// ---- 64-bit immediate classification ----
// The dcon classes classify a 64-bit constant by which of the four
// materialisation instructions (lu12i.w, ori, lu32i.d, lu52i.d) can be
// dropped, mirroring the toolchain's dconClass: a field is ALL1/ALL0 when
// it is all ones/zeros (fillable by sign/zero extension) or ST1/ST0 when it
// starts with a 1/0 but is mixed.
const (
l64All1 = iota
l64All0
l64St1
l64St0
l64dcon120
l64dcon1220s
l64dcon20s20
l64dcon1212s
l64dcon20s12s
l64dcon20s0
l64dcon1212u
l64dcon20s12u
l64dcon3212s
l64dcon320
l64dcon3220
l64dcon1232s
l64dcon20s32
l64dcon3212u
l64Dcon
)
// l64BitField classifies the bit field of v at [suf+len-1 : suf].
func l64BitField(v int64, suf, ln int8) int {
var mask1, mask2 uint64
if ln == 12 {
if suf == 0 {
mask1, mask2 = 0xfff, 0x800
} else {
mask1, mask2 = 0xfff0000000000000, 0x8000000000000000
}
} else {
if suf == 12 {
mask1, mask2 = 0xfffff000, 0x80000000
} else {
mask1, mask2 = 0xfffff00000000, 0x8000000000000
}
}
u := uint64(v)
switch {
case u&mask1 == mask1:
return l64All1
case u&mask1 == 0:
return l64All0
case u&mask2 == mask2:
return l64St1
}
return l64St0
}
// l64DconClass returns the materialisation class of a 64-bit constant,
// transcribed from cmd/internal/obj/loong64's dconClass.
func l64DconClass(v int64) int {
tzb := bits.TrailingZeros64(uint64(v))
hi12 := l64BitField(v, 52, 12)
hi20 := l64BitField(v, 32, 20)
lo20 := l64BitField(v, 12, 20)
lo12 := l64BitField(v, 0, 12)
if tzb >= 52 {
return l64dcon120
}
if tzb >= 32 {
if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) {
return l64dcon20s0
}
return l64dcon320
}
if tzb >= 12 {
if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1220s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s20
}
return l64dcon3220
}
if hi20 == l64All0 {
return l64dcon1220s
}
if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) {
return l64dcon20s20
}
return l64dcon3220
}
if lo12 == l64St1 || lo12 == l64All1 {
if lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1212s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s12s
}
return l64dcon3212s
}
if lo20 == l64St1 {
if hi20 == l64All1 {
return l64dcon1232s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if lo20 == l64All0 {
if hi20 == l64All0 {
return l64dcon1212u
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s12u
}
return l64dcon3212u
}
if hi20 == l64All0 {
return l64dcon1232s
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if lo20 == l64All0 {
if hi20 == l64All0 {
return l64dcon1212u
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s12u
}
return l64dcon3212u
}
if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1232s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if hi20 == l64All0 {
return l64dcon1232s
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
// l64DconMovWords returns the materialisation words for a 64-bit constant
// into rd, per the toolchain's case 67/68/69/59 sequences.
func l64DconMovWords(rd int, v int64) []uint32 {
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
addiw = 0x00a << 22
addid = 0x00b << 22
ori = 0x00e << 22
)
switch l64DconClass(v) {
case l64dcon120:
return []uint32{l64irr(lu52id, int(v>>52), 0, rd)}
case l64dcon1220s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s20:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon1212s:
return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s12s, l64dcon20s0:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon1212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s12u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon3212s, l64dcon320:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon3220:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon1232s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s32:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon3212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
default:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
}
}
// encodeLOONG64LoadImm loads an immediate into a register, matching the
// toolchain's MOVV/MOVW case 3/19/25/59 expansion:
//
// $0: or rd, r0, r0 (MOVW: sll.w rd, r0, r0)
// 1..0xfff: ori rd, r0, imm
// −2048..−1: addi.d rd, r0, imm
// 32-bit (low 12 zero): lu12i.w rd, imm>>12
// 32-bit: lu12i.w rd, imm>>12; ori rd, rd, imm
// 64-bit: lu12i.w rd, imm>>12; ori rd, rd, imm;
// lu32i.d rd, imm>>32; lu52i.d rd, rd, imm>>52
func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte {
if v == 0 {
// The zero constant matches the register-form optab entry: MOVV →
// or rd, r0, r0, MOVW → sll.w rd, r0, r0.
op := l64movRegTable["MOVV"].op
if mnem == "MOVW" {
op = l64movRegTable["MOVW"].op
}
return l64wordLE(l64rrr(op, 0, 0, rd))
}
if v > 0 && v <= 0xfff {
return l64wordLE(l64irr(l64DualTable["OR"].imm, int(v), 0, rd))
}
if v >= -2048 && v < 0 {
// Both MOVV and MOVW use addi.d for negative constants.
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(v), 0, rd))
}
if v == int64(int32(v)) {
if v&0xfff == 0 {
return l64wordLE(l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd))
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd),
l64irr(l64DualTable["OR"].imm, int(v), rd, rd),
)
}
// 64-bit constants use the shortest materialisation the bit pattern
// admits (dcon classification).
return l64WordsLE(l64DconMovWords(rd, v)...)
}
// encodeLOONG64MemOp encodes a memory load (load = true) or store with a
// 12-bit offset, or the 3-instruction expansion for larger offsets:
// lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30).
func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo) ([]byte, error) {
rj, off := l64MemWithFrame(mem, fi)
if rj < 0 {
return nil, fmt.Errorf("invalid memory operand")
}
ls, ok := l64loadStoreTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported MOV width %q", mnem)
}
op := ls.st
if load {
op = ls.ld
}
if off >= -2048 && off < 2048 {
return l64wordLE(l64irr(op, int(off), rj, reg)), nil
}
// Large offset: materialise the base in R30 (the assembler temp).
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int((off+0x800)>>12), 30),
l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30),
l64irr(op, int(off), 30, reg),
), nil
}
// encodeLOONG64RegMove encodes a register-to-register move: the width
// extensions (ext.w.b, ext.w.h, sll.w, or, andi, bstrpick.d) between GPRs,
// fmov between F registers, and the special moves across the GPR/FP/FCC/FCSR
// banks.
func encodeLOONG64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
rs, rd := l64Reg(src), l64Reg(dst)
if rs < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sc, dc := loong64RegClass(operandRegName(src)), loong64RegClass(operandRegName(dst))
// FP-bank specials (MOVV/MOVW between GPR/FCC/FCSR and F registers).
if key, ok := l64FpMoveKey(mnem, sc, dc); ok {
op, ok := l64FpMovTable[key]
if !ok {
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
}
return l64wordLE(l64rr(op, rs, rd)), nil
}
// GPR → GPR.
if sc == l64ClsGR && dc == l64ClsGR {
switch mnem {
case "MOVHU":
// bstrpick.d rd, rj, $15, $0
return l64wordLE(l64irir(0x3<<22, 15, rs, 0, rd)), nil
case "MOVWU":
// bstrpick.d rd, rj, $31, $0
return l64wordLE(l64irir(0x3<<22, 31, rs, 0, rd)), nil
}
if e, ok := l64movRegTable[mnem]; ok {
if e.rr {
return l64wordLE(l64rr(e.op, rs, rd)), nil
}
if e.imm != 0 {
return l64wordLE(l64irr(e.op, e.imm, rs, rd)), nil
}
// 3R with rk = r0: or rd, rj, r0 / sll.w rd, rj, r0.
return l64wordLE(l64rrr(e.op, 0, rs, rd)), nil
}
}
// F → F.
if sc == l64ClsFP && dc == l64ClsFP {
if op, ok := l64movFpRegTable[mnem]; ok {
return l64wordLE(l64rr(op, rs, rd)), nil
}
}
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
}
// l64FpMoveKey builds the l64FpMovTable key for a cross-bank move, reporting
// whether the move is a cross-bank special at all.
func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) {
bank := func(c l64RegClass) string {
switch c {
case l64ClsFP:
return "F"
case l64ClsFCC:
return "FCC"
case l64ClsFCSR:
return "FCSR"
default:
return "R"
}
}
if sc == dc {
return "", false
}
if mnem != "MOVV" && mnem != "MOVW" {
return "", false
}
key := mnem + "." + bank(sc) + "." + bank(dc)
_, ok := l64FpMovTable[key]
return key, ok
}
// ---- static symbol references (pcalau12i + offset) ----
// encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the
// R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address.
func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, rd),
l64irr(l64DualTable["ADDV"].imm, 0, rd, rd),
)
}
// encodeLOONG64SBLoad emits pcalau12i r30, 0; ld rd, 0(r30) with the
// R_LOONG64_ADDR_HI/LO pair, loading from a static symbol.
func encodeLOONG64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte {
ls := l64loadStoreTable[mnem]
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
l64irr(ls.ld, 0, 30, rd),
)
}
// encodeLOONG64SBStore emits pcalau12i r30, 0; st rd, 0(r30) with the
// R_LOONG64_ADDR_HI/LO pair, storing to a static symbol.
func encodeLOONG64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) []byte {
ls := l64loadStoreTable[mnem]
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
l64irr(ls.st, 0, 30, rs),
)
}
// ---- operand helpers ----
// l64IndexedTable holds the register-indexed load/store (ldx/stx) opcodes.
var l64IndexedTable = map[string]struct{ ld, st uint32 }{
"MOVB": {0x07000 << 15, 0x07020 << 15},
"MOVH": {0x07008 << 15, 0x07028 << 15},
"MOVW": {0x07010 << 15, 0x07030 << 15},
"MOVV": {0x07018 << 15, 0x07038 << 15},
"MOVBU": {0x07040 << 15, 0x07020 << 15},
"MOVHU": {0x07048 << 15, 0x07028 << 15},
"MOVWU": {0x07050 << 15, 0x07030 << 15},
"MOVF": {0x07060 << 15, 0x07070 << 15},
"MOVD": {0x07068 << 15, 0x07078 << 15},
}
// operandRegName returns the register name of an operand, or "".
func operandRegName(op *ast.Operand) string {
if op.Addr.Base != "" {
return op.Addr.Base
}
if op.Addr.Sym != nil && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name
}
return ""
}
// l64Reg returns the register number of an operand, or -1.
func l64Reg(op *ast.Operand) int {
return loong64RegNum(operandRegName(op))
}
// l64Imm64 returns the full 64-bit immediate value of an operand.
func l64Imm64(op *ast.Operand) int64 {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
return 0
}
// l64Mem returns the base register and byte offset of a memory operand.
func l64Mem(op *ast.Operand) (rj int, off int32) {
rj = loong64RegNum(op.Addr.Base)
off = int32(op.Addr.Offset)
return
}
// l64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
// registers via the frame mapping.
func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return loong64ResolvePseudo(op.Addr.Sym, fi)
}
return l64Mem(op)
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
return off
}
// l64Label returns the label name of an operand.
func l64Label(op *ast.Operand) string {
if op.Addr.Sym != nil {
return op.Addr.Sym.Name
}
return op.Raw
}
// ---- LSX/LASX (V*/XV*) vector dispatch ----
// l64VecOperand describes a vector register operand: the 5-bit register
// number, its bank and an optional width or element suffix (V0.B16,
// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim
// (the element index survives only in the raw text), so the suffix is
// scanned from op.Raw.
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
unsig bool // the suffix carries the U marker (.BU[0])
hasSuf bool // any suffix present
}
// l64ParseVecOperand parses a vector register operand with an optional
// width or element suffix. ok reports whether the operand names a vector
// register at all (V or X bank, with or without a suffix).
func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
if op.Kind == ast.OpImmediate {
return v, false
}
name := strings.ReplaceAll(op.Raw, " ", "")
if name == "" || (name[0] != 'V' && name[0] != 'X') {
return v, false
}
i := 1
num := 0
for i < len(name) && name[i] >= '0' && name[i] <= '9' {
num = num*10 + int(name[i]-'0')
if num > 31 {
return v, false
}
i++
}
if i == 1 {
return v, false // no register digits
}
v.num, v.lasx = num, name[0] == 'X'
if i == len(name) {
return v, true
}
if name[i] != '.' || i+2 > len(name) {
return v, false
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
return v, false
}
v.width, v.hasSuf = w, true
i++
if i < len(name) && name[i] == 'U' {
v.unsig = true
i++
}
if i < len(name) && name[i] == '[' {
// Element form .T[i]: the closing bracket ends the operand.
if name[len(name)-1] != ']' || i+2 > len(name)-1 {
return v, false
}
idx := 0
for _, c := range name[i+1 : len(name)-1] {
if c < '0' || c > '9' {
return v, false
}
idx = idx*10 + int(c-'0')
if idx > 31 {
return v, false
}
}
v.elem, v.hasEl = idx, true
return v, true
}
// Width form .T<lanes>: the trailing digits give the lane count.
lanes := 0
if i >= len(name) {
return v, false
}
for ; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return v, false
}
lanes = lanes*10 + int(name[i]-'0')
if lanes > 64 {
return v, false
}
}
v.lanes = lanes
return v, true
}
// l64VecSuffixWidth validates a width suffix against the bank (LSX:
// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width
// selector of vreplgr2vr and vldrepl.
func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) {
want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2}
if lasx {
want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4}
}
lanes, ok := want[v.width]
if !ok || lanes != v.lanes {
return 0, false
}
switch v.width {
case 'B':
return 0, true
case 'H':
return 1, true
case 'W':
return 2, true
default:
return 3, true
}
}
// l64VecElementBase validates an element suffix against the bank and
// returns the encoded index field: the index rides in the rk field above a
// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w
// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain
// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`.
func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
limit, base := 0, 0
switch v.width {
case 'B':
if lasx {
return 0, false
}
limit, base = 15, 0
case 'H':
if lasx {
return 0, false
}
limit, base = 7, 16
case 'W':
limit, base = 3, 24
if lasx {
limit, base = 7, 16
}
case 'V':
limit, base = 1, 28
if lasx {
limit, base = 3, 24
}
default:
return 0, false
}
if v.elem > limit {
return 0, false
}
return base + v.elem, true
}
// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
// mnemonic belongs to the vector slice; the operand shapes and opcode
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
return code, true, err
}
lasx, ok := l64VecBank[mnem]
if !ok {
return nil, false, nil
}
ops := instr.Operands
bank := "V"
if lasx {
bank = "X"
}
vec := func(op *ast.Operand) (int, error) {
v, isVec := l64ParseVecOperand(op)
if !isVec || v.lasx != lasx || v.hasSuf {
return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw)
}
return v.num, nil
}
// Two-operand forms (vpcnt.v): INSTR vj, vd.
if l64Vec2R[mnem] {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[1])
if err != nil {
return nil, true, err
}
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil
}
// Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd.
if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) {
if len(ops) > 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < e.min || imm > e.max {
return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max)
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil
}
// Vector-to-condition forms: INSTR vj, FCCn.
if l64InstrTable[mnem].format == l64Fvcf {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC {
return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw)
}
fcc := loong64RegNum(operandRegName(ops[1]))
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil
}
// Four-register forms (vshuf.b): INSTR va, vk, vj, vd.
if l64Vec4R[mnem] {
if len(ops) != 4 {
return nil, true, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
va, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vk, err := vec(ops[1])
if err != nil {
return nil, true, err
}
vj, err := vec(ops[2])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[3])
if err != nil {
return nil, true, err
}
return l64wordLE(l64InstrTable[mnem].op | uint32(va&0x1f)<<15 | uint32(vk&0x1f)<<10 | uint32(vj&0x1f)<<5 | uint32(vd&0x1f)), true, nil
}
// Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd).
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
vk, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil
}
// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic
// covers the whole LSX/LASX transfer surface, dispatched by operand shape
// exactly as the toolchain's table does:
//
// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld
// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx
// VMOVQ off(rj), vd.T vldrepl (load and replicate one element)
// VMOVQ vj, vd vori.b $0 (a register move)
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
if lasx {
bank = "X"
}
if len(ops) != 2 {
return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops))
}
src, srcVec := l64ParseVecOperand(ops[0])
dst, dstVec := l64ParseVecOperand(ops[1])
srcMem := isMemOperand(ops[0])
dstMem := isMemOperand(ops[1])
srcIdx := srcMem && ops[0].Addr.Index != ""
dstIdx := dstMem && ops[1].Addr.Index != ""
intReg := func(op *ast.Operand) (int, error) {
if isMemOperand(op) {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
name := operandRegName(op)
if loong64RegClass(name) != l64ClsGR {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
return loong64RegNum(name), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
if src.hasSuf || dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank)
}
if src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil
}
// Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk).
if srcVec && dstMem {
if src.hasSuf || src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank)
}
if dstIdx {
rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil
}
// Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the
// load-and-replicate form VMOVQ off(rj), vd.T.
if srcMem && dstVec {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
if srcIdx {
if dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank)
}
rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
op := enc.ld
if dst.hasSuf {
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
case 2:
op = enc.replW
default:
op = enc.replD
}
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
// Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned).
if srcVec && src.hasEl && !dstVec && !dstMem {
if src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
idx, ok := l64VecElementBase(lasx, src)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw)
}
rd, err := intReg(ops[1])
if err != nil {
return nil, err
}
op := enc.pickS
if src.unsig {
op = enc.pickU
}
return l64wordLE(l64irr(op, idx, src.num, rd)), nil
}
// Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and
// VMOVQ rj, vd.T (vreplgr2vr).
if !srcVec && !srcMem && dstVec && dst.hasSuf {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
rs, err := intReg(ops[0])
if err != nil {
return nil, err
}
if dst.hasEl {
idx, ok := l64VecElementBase(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil
}
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil
}
return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw)
}