Files
gasm-sdk/asm/loong64_assemble.go
T

1915 lines
59 KiB
Go
Raw Normal View History

// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"fmt"
"math/bits"
"strconv"
"strings"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
)
// assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into
// machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and
// the immediate-arithmetic forms expand to 2-5 instructions when the
// immediate does not fit, so the layout is computed in two passes (sizes,
// then encoding with resolved branch targets).
//
// The emitted bytes match the Go toolchain's loong64 assembler, which is the
// ground-truth oracle: prologue/epilogue (including the large-frame R30
// materialisations), FP/SP frame mapping, the stack-split guard classes, and
// branch encodings all follow cmd/internal/obj/loong64. The morestack block
// at the end of split functions carries the runtime.morestack_noctxt call.
func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) {
fi := loong64ComputeFrame(t)
prologue := loong64Prologue(fi)
guardLen := loong64GuardLen(fi)
chain := loong64JumpChain(t)
resolve := func(name string) string {
if r, ok := chain[name]; ok {
return r
}
return name
}
var relocs []Reloc
var spadj []SpadjStep
// The prologue raises the SP delta by autosize; the boundary is reported
// after the SP adjust instruction, exactly as the toolchain's pctospadj
// does. The prologue (3 instructions when a frame is present) may
// materialise its store or adjust through R30, which widens it.
if fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize})
}
// Pass 1: label offsets from the instruction sizes.
offsets := map[string]int{}
pos := guardLen + len(prologue)
for _, stmt := range t.Body {
switch s := stmt.(type) {
case *ast.Label:
offsets[s.Name.Text] = pos
case *ast.Instr:
pos += loong64InstrSize(s, fi)
}
}
// Pass 2: encode. The guard prefix precedes the prologue; its branches
// target the morestack block at the end of the function, which the first
// pass has sized.
bodyLen := 0
{
p := guardLen + len(prologue)
for _, stmt := range t.Body {
if in, ok := stmt.(*ast.Instr); ok {
p += loong64InstrSize(in, fi)
}
}
bodyLen = p - (guardLen + len(prologue))
}
var out []byte
if fi.needSplit {
out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...)
}
out = append(out, prologue...)
pc := guardLen + len(prologue)
preCount := len(relocs)
var lines []LineEntry
for _, stmt := range t.Body {
in, ok := stmt.(*ast.Instr)
if !ok {
continue
}
code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve)
if err != nil {
return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
}
for j := preCount; j < len(relocs); j++ {
// Make the relocation offsets function-relative: each instruction
// records its reloc offset relative to its own start, and pc is
// that instruction's offset from the function start (prologue
// included). After shifts by the same amount.
relocs[j].Off += pc
relocs[j].After += pc
}
preCount = len(relocs)
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
// The RET's epilogue closes the frame: the SP delta returns to zero
// after the frame-deallocating ADDV.
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
spadj = append(spadj, SpadjStep{PC: pc + loong64EpilogueWords(fi)*4, Value: 0})
}
out = append(out, code...)
pc += len(code)
}
if fi.needSplit {
block, blReloc := loong64MoreStackBlock(pc)
out = append(out, block...)
relocs = append(relocs, blReloc)
pc += len(block)
}
return out, offsets, relocs, lines, spadj, nil
}
// loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's
// branch-chasing pass: a label whose first instruction is an unconditional
// local jump redirects its own jumpers to the ultimate target. The Go
// toolchain chases these chains before it encodes branches, so matching its
// bytes requires the same redirection.
func loong64JumpChain(t *ast.Text) map[string]string {
leadsTo := map[string]string{}
for i, stmt := range t.Body {
l, ok := stmt.(*ast.Label)
if !ok {
continue
}
j := i + 1
for j < len(t.Body) {
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
break
}
j++
}
if j >= len(t.Body) {
continue
}
in, ok := t.Body[j].(*ast.Instr)
if !ok {
continue
}
mnem := strings.ToUpper(in.Mnemonic.Text)
if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 {
continue
}
if name, ok := l64LabelOK(in.Operands[0]); ok {
leadsTo[l.Name.Text] = name
}
}
chain := map[string]string{}
for name := range leadsTo {
visited := map[string]bool{name: true}
cur := name
for {
next, ok := leadsTo[cur]
if !ok || visited[next] {
break
}
visited[next] = true
cur = next
}
if cur != name {
chain[name] = cur
}
}
return chain
}
// l64LabelOK returns the local label name of a jump operand.
func l64LabelOK(op *ast.Operand) (string, bool) {
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name, true
}
return "", false
}
// loong64InstrSize returns the encoded size of an instruction: 4 bytes for
// most, more for the multi-instruction expansions.
func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
if mnem == "RET" {
return len(loong64Return(fi))
}
switch mnem {
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return loong64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := l64Imm64(ops[0])
if v == 0 {
return 4 // folds into the 3R form (rk = R0)
}
switch mnem {
case "ADD", "ADDW", "ADDV", "ADDVU", "SGT", "SGTU":
// C_US12CON (−2048..0x7ff) encodes directly as addi/slti.
if v >= -2048 && v <= 0x7ff {
return 4
}
// C_U12CON (0x800..0xfff) → ori r30, r0, v; op rd, rj, r30.
if v >= 0x800 && v <= 0xfff {
return 8
}
default: // AND/OR/XOR
// C_UU12CON (0..0x7ff) encodes directly as andi/ori/xori.
if v >= 0 && v <= 0x7ff {
return 4
}
// C_S12CON (−2048..−1) → addi.d r30, r0, v; op rd, rj, r30.
if v >= -2048 && v < 0 {
return 8
}
}
// 0x800..0xfff for AND/OR/XOR and the 32/64-bit ranges go through
// the lu12i.w materialisation.
if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
return 8 // lu12i.w r30, v>>12; op rd, rj, r30
}
return 12 // lu12i.w r30, v>>12; ori r30, r30, v; op rd, rj, r30
}
return 4 * (len(l64DconMovWords(0, v)) + 1) // dcon materialisation + op
}
}
return 4
}
// encodeLOONG64Instr encodes a single LoongArch instruction.
func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
mnem := strings.ToUpper(instr.Mnemonic.Text)
ops := instr.Operands
// Pseudo-instructions and the branches first.
switch mnem {
case "RET":
return loong64Return(fi), nil
case "NOP", "NOOP":
// andi r0, r0, 0
return l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)), nil
case "UNDEF":
// break 0
return l64wordLE(l64i15(l64InstrTable["BREAK"].op, 0)), nil
case "WORD":
if len(ops) != 1 {
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return l64wordLE(uint32(immFromOperand(ops[0]))), nil
case "JMP", "B":
return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs)
case "JAL", "CALL", "BL":
return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs)
case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD":
return encodeLOONG64Mov(instr, mnem, fi, relocs)
}
// 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL.
if op, ok := l64branchTable[mnem]; ok {
if mnem == "JIRL" {
return encodeLOONG64Jirl(op, ops)
}
return encodeLOONG64Branch16(mnem, op, ops, pc, offsets, resolve)
}
// Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ,
// BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0).
if op, ok := l64branch21Table[mnem]; ok {
return encodeLOONG64Branch21(mnem, op, ops, pc, offsets, resolve)
}
// B/BL aliases reached only via JMP/JAL above.
// The dual-form arithmetic mnemonics: register (3R) or immediate (2RI12).
if de, ok := l64DualTable[mnem]; ok {
if len(ops) >= 2 && isImmOperand(ops[0]) {
if de.shift {
// INSTR $shamt, rd or INSTR $shamt, rj, rd.
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
shamt := int(immFromOperand(ops[0]))
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// $0 folds into the register form (the toolchain matches the
// zero constant against the 3R optab entry first).
if shamt == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
// The .d variants take a 6-bit amount, the .w variants 5 bits.
if isLoong64ShiftD(de.imm) {
shamt &= 0x3f
} else {
shamt &= 0x1f
}
return l64wordLE(l64irr(de.imm, shamt, rj, rd)), nil
}
return encodeLOONG64ImmArith(mnem, de, ops)
}
// Register form: 3R.
if len(ops) == 3 {
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(de.rrr, rk, rj, rd)), nil
}
if len(ops) == 2 {
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rk < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(de.rrr, rk, rd, rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before
// the integer/FP table (their mnemonics overlap the table's 2R format
// but resolve vector-bank registers).
if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled {
if err != nil {
return nil, err
}
return code, nil
}
enc, ok := l64InstrTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem)
}
switch enc.format {
case l64Frrr:
// INSTR rk, rj, rd (3 operands) or INSTR rk, rd (rj = rd).
switch len(ops) {
case 3:
rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
case 2:
rk, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rk < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rrr(enc.op, rk, rd, rd)), nil
}
return nil, fmt.Errorf("%s expects 2 or 3 register operands, got %d", mnem, len(ops))
case l64Frr:
// INSTR rj, rd.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rr(enc.op, rj, rd)), nil
case l64Firr:
// LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
imm := int(immFromOperand(ops[0]))
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irr(enc.op, imm, rj, rd)), nil
case l64Firr16:
// ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be
// a multiple of 65536 and is shifted right by 16.
if len(ops) < 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects an immediate operand", mnem)
}
v := int(immFromOperand(ops[0]))
if v&0xFFFF != 0 {
return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem)
}
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irr16(enc.op, v>>16, rj, rd)), nil
case l64Firr14:
// LL/SC/MOVWP: INSTR mem, rd (load) or INSTR rd, mem (store); the
// 14-bit offset is scaled by 4 (byte offset >> 2).
rd, rj, off, load, err := l64MemOperands(ops, fi)
if err != nil {
return nil, err
}
op := enc.op
if load && (mnem == "MOVWP" || mnem == "MOVVP") {
// ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field.
op -= 1 << 24
}
return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil
case l64Fir20:
// LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rd := l64Reg(ops[0])
if rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64ir(enc.op, int(immFromOperand(ops[1])), rd)), nil
case l64Frrrr:
// FMADD/FMSUB/FNMADD/FNMSUB: INSTR fa, fk, fj, fd (4 operands) or
// INSTR fa, fk, fd (fj = fd).
fa, fk, fj, fd, err := l64FmaOperands(ops)
if err != nil {
return nil, err
}
return l64wordLE(l64rrrr(enc.op, fa, fk, fj, fd)), nil
case l64Firir:
// BSTRINS/BSTRPICK: INSTR $msb, rj, $lsb, rd (or $msb, rj, rd with
// lsb = 0).
if len(ops) != 4 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops))
}
msb := int(immFromOperand(ops[0]))
lsb := 0
rj := l64Reg(ops[1])
rd := l64Reg(ops[len(ops)-1])
if len(ops) == 4 {
lsb = int(immFromOperand(ops[2]))
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// The toolchain validates the bit numbers ("illegal bit number"):
// 0..31 for the .w forms, 0..63 for the .d forms, lsb <= msb.
b := 64
if strings.HasSuffix(mnem, "W") {
b = 32
}
if msb < 0 || msb >= b || lsb < 0 || lsb >= b || lsb > msb {
return nil, fmt.Errorf("%s: illegal bit number (msb %d, lsb %d)", mnem, msb, lsb)
}
return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil
case l64Firrr:
// ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in
// the second register position); the source amount is 1-4, encoded
// as sa-1.
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
}
sa := int(immFromOperand(ops[0])) - 1
rj, rk, rd := l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
if sa < 0 || sa > 3 {
return nil, fmt.Errorf("shift amount out of range [1, 4]")
}
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64irrr(enc.op, sa, rk, rj, rd)), nil
case l64Fi15:
// SYSCALL/BREAK/DBAR: no operands, or SYSCALL $code / BREAK $code.
code := 0
if len(ops) == 1 {
code = int(immFromOperand(ops[0]))
} else if len(ops) > 1 {
return nil, fmt.Errorf("%s expects at most 1 operand, got %d", mnem, len(ops))
}
return l64wordLE(l64i15(enc.op, code)), nil
case l64Fam:
// AM* val, (addr), result.
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
rk := l64Reg(ops[0])
rj, _ := l64Mem(ops[1])
rd := l64Reg(ops[2])
if rk < 0 || rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil
case l64Frdtime:
// RDTIME* rd, rj (rd at bits [9:5], rj at bits [4:0]).
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rd, rj := l64Reg(ops[0]), l64Reg(ops[1])
if rd < 0 || rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
return l64wordLE(l64rr(enc.op, rd, rj)), nil
case l64Fpreld:
// PRELD off(rj), $hint.
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
rj, off := l64Mem(ops[0])
hint := int(immFromOperand(ops[1]))
if rj < 0 {
return nil, fmt.Errorf("invalid operand in %s", mnem)
}
return l64wordLE(l64irr5i(enc.op, int(off), rj, hint)), nil
}
return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops))
}
// encodeLOONG64Branch encodes a label or indirect jump/call:
//
// JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0
// JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0
func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc) ([]byte, error) {
if len(instr.Operands) != 1 {
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands))
}
op := instr.Operands[0]
if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil {
// Indirect: (rj) → jirl.
rj := loong64RegNum(op.Addr.Base)
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
rd := 0
if link {
rd = 1 // link register
}
return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil
}
// Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation
// (the linker fills the offset), as the toolchain does for CALL/BL/JAL
// and for tail-calling JMP.
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
opc := l64jumpTable["B"]
if link {
opc = l64jumpTable["BL"]
}
if relocs != nil {
*relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset})
}
return l64wordLE(l64bbl(opc, 0)), nil
}
// Direct: label → b/bl.
target := resolve(l64Label(op))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
if v < -1<<25 || v >= 1<<25 {
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
}
opc := l64jumpTable[mnem]
return l64wordLE(l64bbl(opc, v)), nil
}
// encodeLOONG64Jirl encodes the raw JIRL spelling, JIRL rd, rj, offset, the
// form the verify trampolines use. The (rj) indirect form without an offset
// is handled by encodeLOONG64Branch.
func encodeLOONG64Jirl(op uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("JIRL expects 3 operands, got %d", len(ops))
}
rd := l64Reg(ops[0])
rj := l64Reg(ops[1])
if rd < 0 || rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
off, ok := l64offsetOperand(ops[2])
if !ok {
return nil, fmt.Errorf("JIRL expects an immediate offset, got %q", ops[2].Raw)
}
if (int64(off)<<16)>>16 != int64(off) {
return nil, fmt.Errorf("JIRL offset %d out of the 16-bit range", off)
}
return l64wordLE(l64irr16(op, int(off), rj, rd)), nil
}
// l64offsetOperand reads a bare numeric branch offset: an immediate ($n) or a
// plain number, which parses as an empty address carrying the digits in Raw.
func l64offsetOperand(op *ast.Operand) (int32, bool) {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return int32(v), true
}
if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "" && op.Addr.Index == "" {
if v, err := strconv.ParseInt(op.Raw, 0, 64); err == nil {
return int32(v), true
}
}
return 0, false
}
// encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU):
// INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain
// turns into the 21-bit BEQZ/BNEZ form when the register is the only operand.
func encodeLOONG64Branch16(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[len(ops)-1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
if len(ops) == 2 {
// Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/
// BGEZ-family aliases encoded with rj in the rj field.
rj := l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if mnem == "BLTU" || mnem == "BGEU" {
// The unsigned compares have no single-register pseudo: the
// toolchain keeps the register-register form with rd = R0
// (bltu rj, r0 is never taken), not a sometimes-taken beqz.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, rj, 0)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
zop := l64branch21Table["BEQZ"]
if mnem == "BNE" {
zop = l64branch21Table["BNEZ"]
}
if mnem == "BLT" || mnem == "BLTZ" || mnem == "BGTZ" {
zop = l64branch21Table["BLTZ"]
}
if mnem == "BGE" || mnem == "BGEZ" || mnem == "BLEZ" {
zop = l64branch21Table["BGEZ"]
}
return l64wordLE(l64ir21(zop, v, rj)), nil
}
// Two registers: BEQ rj, rd, label. When one is R0 the toolchain
// re-encodes as the 21-bit BEQZ/BNEZ form.
rj, rd := l64Reg(ops[0]), l64Reg(ops[1])
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
if rj == 0 {
rj, rd = rd, 0
}
if rd == 0 {
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
zop := l64branch21Table["BEQZ"]
if mnem == "BNE" {
zop = l64branch21Table["BNEZ"]
}
return l64wordLE(l64ir21(zop, v, rj)), nil
}
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, rj, rd)), nil
}
// encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and
// BFPT/BFPF use the 21-bit offset form (register in the rj field), while
// BGTZ/BLEZ, which the toolchain encodes with the register in the rd field
// and a 16-bit offset, are handled separately.
func encodeLOONG64Branch21(mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
target := resolve(l64Label(ops[1]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets))
}
v := (targetOff - pc) >> 2
rj := 0 // BFPT/BFPF default to FCC0
if mnem != "BFPT" && mnem != "BFPF" {
rj = l64Reg(ops[0])
if rj < 0 {
return nil, fmt.Errorf("invalid register operand")
}
}
if mnem == "BGTZ" || mnem == "BLEZ" {
// The toolchain swaps the register into the rd field and keeps the
// 16-bit offset form.
if (v<<16)>>16 != v {
return nil, fmt.Errorf("branch to %q too far (16-bit range)", target)
}
return l64wordLE(l64irr16(op, v, 0, rj)), nil
}
if (v<<11)>>11 != v {
return nil, fmt.Errorf("branch to %q too far (21-bit range)", target)
}
return l64wordLE(l64ir21(op, v, rj)), nil
}
// encodeLOONG64ImmArith encodes an immediate arithmetic/logic instruction,
// expanding the immediate exactly as the toolchain's aclass classifies it:
//
// ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes)
// 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8)
// AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4)
// −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8)
// 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12)
// 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20)
func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := l64Imm64(ops[0])
rd := l64Reg(ops[len(ops)-1])
rj := rd
if len(ops) == 3 {
rj = l64Reg(ops[1])
}
if rj < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
// The two immediate families classify differently.
additive := mnem == "ADD" || mnem == "ADDW" || mnem == "ADDV" || mnem == "ADDVU" || mnem == "SGT" || mnem == "SGTU"
if additive {
if v == 0 {
// $0 folds into the 3R form (rk = R0), matching the toolchain's
// optab matching of the zero constant against the register form.
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= -2048 && v <= 0x7ff {
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= 0x800 && v <= 0xfff {
return l64WordsLE(
l64irr(0x00e<<22, int(v), 0, 30), // ori r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
} else {
if v == 0 {
return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil
}
if v >= 0 && v <= 0x7ff {
return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil
}
if v >= -2048 && v < 0 {
return l64WordsLE(
l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v
l64rrr(de.rrr, 30, rj, rd),
), nil
}
}
// 32/64-bit constants are materialised in R30 (the assembler temp),
// using the same dcon classification as the toolchain's case 24/60/70/
// 71/72 sequences.
const (
lu12iw = 0x0a << 25
ori = 0x00e << 22
)
if v == int64(int32(v)) {
if v&0xfff == 0 && (v < 0x800 || v > 0xfff) {
return l64WordsLE(
l64ir(lu12iw, int(int32(v)>>12), 30),
l64rrr(de.rrr, 30, rj, rd),
), nil
}
return l64WordsLE(
l64ir(lu12iw, int(int32(v)>>12), 30),
l64irr(ori, int(v), 30, 30),
l64rrr(de.rrr, 30, rj, rd),
), nil
}
words := l64DconMovWords(30, v)
words = append(words, l64rrr(de.rrr, 30, rj, rd))
return l64WordsLE(words...), nil
}
// isLoong64ShiftD reports whether a shift-immediate opcode constant is one of
// the 6-bit (.d) variants, the toolchain distinguishes them by the bit
// position of the opcode field (bits [25:16]).
func isLoong64ShiftD(op uint32) bool {
return op&0x03ff0000 != 0 && op>>25 == 0
}
// l64FmaOperands extracts the four fused-multiply-add operands:
// INSTR fa, fk, fj, fd, or INSTR fa, fk, fd with fj = fd.
func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) {
switch len(ops) {
case 4:
fa, fk, fj, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3])
case 3:
fa, fk, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2])
fj = fd
default:
return 0, 0, 0, 0, fmt.Errorf("expected 3 or 4 operands, got %d", len(ops))
}
if fa < 0 || fk < 0 || fj < 0 || fd < 0 {
return 0, 0, 0, 0, fmt.Errorf("invalid register operand")
}
return fa, fk, fj, fd, nil
}
// l64MemOperands extracts (rd, rj, off, load) from a load/store instruction:
// INSTR mem, rd is a load, INSTR rd, mem a store.
func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) {
if len(ops) != 2 {
return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops))
}
if isMemOperand(ops[0]) {
rd = l64Reg(ops[1])
rj, off = l64MemWithFrame(ops[0], fi)
load = true
} else if isMemOperand(ops[1]) {
rd = l64Reg(ops[0])
rj, off = l64MemWithFrame(ops[1], fi)
} else {
return 0, 0, 0, false, fmt.Errorf("expected a memory operand")
}
if rd < 0 || rj < 0 {
return 0, 0, 0, false, fmt.Errorf("invalid operand")
}
return rd, rj, off, load, nil
}
// ---- the MOV pseudo-instruction ----
// encodeLOONG64Mov encodes the MOV family, the load/store/immediate
// workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width
// mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the
// access width). The forms, mirroring the toolchain:
//
// MOVx $imm, rd load immediate (addi/lu12i+ori/lu32i/lu52i)
// MOVx mem, rd load from memory
// MOVx rd, mem store to memory
// MOVx rs, rd register move (incl. the FP-bank specials)
// MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d)
// MOVx sym(SB), rd load from a static symbol (pcalau12i+ld)
// MOVx rd, sym(SB) store to a static symbol (pcalau12i+st)
func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) {
ops := instr.Operands
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
if mnem == "MOV" {
mnem = "MOVV"
}
src, dst := ops[0], ops[1]
// Immediate → register.
if isImmOperand(src) && !isMemOperand(src) {
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
}
return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil
}
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
}
// MOVW $imm, Fd is the only immediate-to-F form the toolchain's optab
// accepts (AMOVW's C_12CON against C_FREG): it materialises the
// constant in R30 and moves it across with movgr2fr.w. MOVV/MOVF/
// MOVD are illegal combinations there, and are diagnosed here rather
// than silently written into the GPR of the register's number.
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
if mnem != "MOVW" {
return nil, fmt.Errorf("%s $imm: illegal combination with an F register destination (only MOVW $c, Fd is supported)", mnem)
}
return encodeLOONG64ImmToFp(rd, l64Imm64(src))
}
return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil
}
// Static symbol load/store via pcalau12i.
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem)
}
return encodeLOONG64SBLoad(src.Addr.Sym, rd, mnem, relocs), nil
}
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) {
rs := l64Reg(src)
if rs < 0 {
return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem)
}
return encodeLOONG64SBStore(dst.Addr.Sym, rs, mnem, relocs), nil
}
// Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk).
if src.Addr.Index != "" && !isMemOperand(dst) {
rd := l64Reg(dst)
rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index)
if rd < 0 || rj < 0 || rk < 0 {
return nil, fmt.Errorf("%s (rj)(rk): invalid register operand", mnem)
}
op, ok := l64IndexedTable[mnem]
if !ok {
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
}
return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil
}
if dst.Addr.Index != "" && !isMemOperand(src) {
rs := l64Reg(src)
rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index)
if rs < 0 || rj < 0 || rk < 0 {
return nil, fmt.Errorf("%s rd, (rj)(rk): invalid register operand", mnem)
}
op, ok := l64IndexedTable[mnem]
if !ok {
return nil, fmt.Errorf("%s: no register-indexed form", mnem)
}
return l64wordLE(l64rrr(op.st, rk, rj, rs)), nil
}
// Memory load/store with a 12-bit (or larger, via expansion) offset.
if isMemOperand(src) && !isMemOperand(dst) {
rd := l64Reg(dst)
if rd < 0 {
return nil, fmt.Errorf("%s: invalid destination register", mnem)
}
return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi)
}
if !isMemOperand(src) && isMemOperand(dst) {
rs := l64Reg(src)
if rs < 0 {
return nil, fmt.Errorf("%s: invalid source register", mnem)
}
return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi)
}
// Register → register.
return encodeLOONG64RegMove(mnem, src, dst)
}
// loong64MovSize returns the encoded size of a MOV instruction.
func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int {
if mnem == "MOV" {
mnem = "MOVV"
}
if len(ops) != 2 {
return 4
}
src, dst := ops[0], ops[1]
switch {
case isImmOperand(src):
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
return 8 // pcalau12i + addi.d
}
if loong64RegClass(operandRegName(dst)) == l64ClsFP {
return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid)
}
v := l64Imm64(src)
if v == 0 {
return 4
}
if v > 0 && v <= 0xfff {
return 4 // ori rd, r0, v
}
if v >= -2048 && v < 0 {
return 4 // addi.d rd, r0, v
}
if v == int64(int32(v)) {
if v&0xfff == 0 {
return 4 // lu12i.w
}
return 8 // lu12i.w + ori
}
return 4 * len(l64DconMovWords(0, v))
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
return 8 // pcalau12i + ld
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
return 8 // pcalau12i + st
case src.Addr.Index != "" || dst.Addr.Index != "":
return 4 // ldx/stx
case isMemOperand(src) || isMemOperand(dst):
// A 12-bit offset fits in one instruction; larger offsets expand
// to lu12i.w + add.d + the access.
mem := src
if !isMemOperand(src) {
mem = dst
}
if l64MemOffset(mem, fi) >= -2048 && l64MemOffset(mem, fi) < 2048 {
return 4
}
return 12
default:
return 4 // register move
}
}
// encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it
// to an F register, the toolchain's expansion of MOVW $c, Fd: ori (which
// zero-extends) for the positive span, addi.w for zero and the negative
// span, then movgr2fr.w. The toolchain's optab accepts no wider constant on
// this path (it never materialises one fully first), so values outside
// [-2048, 4095] are diagnosed rather than masked into si12.
func encodeLOONG64ImmToFp(fd int, v int64) ([]byte, error) {
if v < -2048 || v > 4095 {
return nil, fmt.Errorf("MOVW $%d: immediate out of the [-2048, 4095] range for an F register destination", v)
}
op := uint32(0x00a << 22) // addi.w r30, r0, v (sign-extends)
if v > 0 {
op = 0x00e << 22 // ori r30, r0, v (zero-extends)
}
return l64WordsLE(
l64irr(op, int(v), 0, 30),
l64rr(0x4529<<10, 30, fd), // movgr2fr.w fd, r30
), nil
}
// ---- 64-bit immediate classification ----
// The dcon classes classify a 64-bit constant by which of the four
// materialisation instructions (lu12i.w, ori, lu32i.d, lu52i.d) can be
// dropped, mirroring the toolchain's dconClass: a field is ALL1/ALL0 when
// it is all ones/zeros (fillable by sign/zero extension) or ST1/ST0 when it
// starts with a 1/0 but is mixed.
const (
l64All1 = iota
l64All0
l64St1
l64St0
l64dcon120
l64dcon1220s
l64dcon20s20
l64dcon1212s
l64dcon20s12s
l64dcon20s0
l64dcon1212u
l64dcon20s12u
l64dcon3212s
l64dcon320
l64dcon3220
l64dcon1232s
l64dcon20s32
l64dcon3212u
l64Dcon
)
// l64BitField classifies the bit field of v at [suf+len-1 : suf].
func l64BitField(v int64, suf, ln int8) int {
var mask1, mask2 uint64
if ln == 12 {
if suf == 0 {
mask1, mask2 = 0xfff, 0x800
} else {
mask1, mask2 = 0xfff0000000000000, 0x8000000000000000
}
} else {
if suf == 12 {
mask1, mask2 = 0xfffff000, 0x80000000
} else {
mask1, mask2 = 0xfffff00000000, 0x8000000000000
}
}
u := uint64(v)
switch {
case u&mask1 == mask1:
return l64All1
case u&mask1 == 0:
return l64All0
case u&mask2 == mask2:
return l64St1
}
return l64St0
}
// l64DconClass returns the materialisation class of a 64-bit constant,
// transcribed from cmd/internal/obj/loong64's dconClass.
func l64DconClass(v int64) int {
tzb := bits.TrailingZeros64(uint64(v))
hi12 := l64BitField(v, 52, 12)
hi20 := l64BitField(v, 32, 20)
lo20 := l64BitField(v, 12, 20)
lo12 := l64BitField(v, 0, 12)
if tzb >= 52 {
return l64dcon120
}
if tzb >= 32 {
if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) {
return l64dcon20s0
}
return l64dcon320
}
if tzb >= 12 {
if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1220s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s20
}
return l64dcon3220
}
if hi20 == l64All0 {
return l64dcon1220s
}
if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) {
return l64dcon20s20
}
return l64dcon3220
}
if lo12 == l64St1 || lo12 == l64All1 {
if lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1212s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s12s
}
return l64dcon3212s
}
if lo20 == l64St1 {
if hi20 == l64All1 {
return l64dcon1232s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if lo20 == l64All0 {
if hi20 == l64All0 {
return l64dcon1212u
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s12u
}
return l64dcon3212u
}
if hi20 == l64All0 {
return l64dcon1232s
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if lo20 == l64All0 {
if hi20 == l64All0 {
return l64dcon1212u
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s12u
}
return l64dcon3212u
}
if lo20 == l64St1 || lo20 == l64All1 {
if hi20 == l64All1 {
return l64dcon1232s
}
if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
if hi20 == l64All0 {
return l64dcon1232s
}
if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) {
return l64dcon20s32
}
return l64Dcon
}
// l64DconMovWords returns the materialisation words for a 64-bit constant
// into rd, per the toolchain's case 67/68/69/59 sequences.
func l64DconMovWords(rd int, v int64) []uint32 {
const (
lu12iw = 0x0a << 25
lu32id = 0x0b << 25
lu52id = 0x00c << 22
addiw = 0x00a << 22
addid = 0x00b << 22
ori = 0x00e << 22
)
switch l64DconClass(v) {
case l64dcon120:
return []uint32{l64irr(lu52id, int(v>>52), 0, rd)}
case l64dcon1220s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s20:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon1212s:
return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s12s, l64dcon20s0:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon1212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s12u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon3212s, l64dcon320:
return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon3220:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon1232s:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)}
case l64dcon20s32:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)}
case l64dcon3212u:
return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
default:
return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)}
}
}
// encodeLOONG64LoadImm loads an immediate into a register, matching the
// toolchain's MOVV/MOVW case 3/19/25/59 expansion:
//
// $0: or rd, r0, r0 (MOVW: sll.w rd, r0, r0)
// 1..0xfff: ori rd, r0, imm
// −2048..−1: addi.d rd, r0, imm
// 32-bit (low 12 zero): lu12i.w rd, imm>>12
// 32-bit: lu12i.w rd, imm>>12; ori rd, rd, imm
// 64-bit: lu12i.w rd, imm>>12; ori rd, rd, imm;
// lu32i.d rd, imm>>32; lu52i.d rd, rd, imm>>52
func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte {
if v == 0 {
// The zero constant matches the register-form optab entry: MOVV →
// or rd, r0, r0, MOVW → sll.w rd, r0, r0.
op := l64movRegTable["MOVV"].op
if mnem == "MOVW" {
op = l64movRegTable["MOVW"].op
}
return l64wordLE(l64rrr(op, 0, 0, rd))
}
if v > 0 && v <= 0xfff {
return l64wordLE(l64irr(l64DualTable["OR"].imm, int(v), 0, rd))
}
if v >= -2048 && v < 0 {
// Both MOVV and MOVW use addi.d for negative constants.
return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(v), 0, rd))
}
if v == int64(int32(v)) {
if v&0xfff == 0 {
return l64wordLE(l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd))
}
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd),
l64irr(l64DualTable["OR"].imm, int(v), rd, rd),
)
}
// 64-bit constants use the shortest materialisation the bit pattern
// admits (dcon classification).
return l64WordsLE(l64DconMovWords(rd, v)...)
}
// encodeLOONG64MemOp encodes a memory load (load = true) or store with a
// 12-bit offset, or the 3-instruction expansion for larger offsets:
// lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30).
func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo) ([]byte, error) {
rj, off := l64MemWithFrame(mem, fi)
if rj < 0 {
return nil, fmt.Errorf("invalid memory operand")
}
ls, ok := l64loadStoreTable[mnem]
if !ok {
return nil, fmt.Errorf("unsupported MOV width %q", mnem)
}
op := ls.st
if load {
op = ls.ld
}
if off >= -2048 && off < 2048 {
return l64wordLE(l64irr(op, int(off), rj, reg)), nil
}
// Large offset: materialise the base in R30 (the assembler temp).
return l64WordsLE(
l64ir(l64InstrTable["LU12IW"].op, int((off+0x800)>>12), 30),
l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30),
l64irr(op, int(off), 30, reg),
), nil
}
// encodeLOONG64RegMove encodes a register-to-register move: the width
// extensions (ext.w.b, ext.w.h, sll.w, or, andi, bstrpick.d) between GPRs,
// fmov between F registers, and the special moves across the GPR/FP/FCC/FCSR
// banks.
func encodeLOONG64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
rs, rd := l64Reg(src), l64Reg(dst)
if rs < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand")
}
sc, dc := loong64RegClass(operandRegName(src)), loong64RegClass(operandRegName(dst))
// FP-bank specials (MOVV/MOVW between GPR/FCC/FCSR and F registers).
if key, ok := l64FpMoveKey(mnem, sc, dc); ok {
op, ok := l64FpMovTable[key]
if !ok {
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
}
return l64wordLE(l64rr(op, rs, rd)), nil
}
// GPR → GPR.
if sc == l64ClsGR && dc == l64ClsGR {
switch mnem {
case "MOVHU":
// bstrpick.d rd, rj, $15, $0
return l64wordLE(l64irir(0x3<<22, 15, rs, 0, rd)), nil
case "MOVWU":
// bstrpick.d rd, rj, $31, $0
return l64wordLE(l64irir(0x3<<22, 31, rs, 0, rd)), nil
}
if e, ok := l64movRegTable[mnem]; ok {
if e.rr {
return l64wordLE(l64rr(e.op, rs, rd)), nil
}
if e.imm != 0 {
return l64wordLE(l64irr(e.op, e.imm, rs, rd)), nil
}
// 3R with rk = r0: or rd, rj, r0 / sll.w rd, rj, r0.
return l64wordLE(l64rrr(e.op, 0, rs, rd)), nil
}
}
// F → F.
if sc == l64ClsFP && dc == l64ClsFP {
if op, ok := l64movFpRegTable[mnem]; ok {
return l64wordLE(l64rr(op, rs, rd)), nil
}
}
return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst))
}
// l64FpMoveKey builds the l64FpMovTable key for a cross-bank move, reporting
// whether the move is a cross-bank special at all.
func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) {
bank := func(c l64RegClass) string {
switch c {
case l64ClsFP:
return "F"
case l64ClsFCC:
return "FCC"
case l64ClsFCSR:
return "FCSR"
default:
return "R"
}
}
if sc == dc {
return "", false
}
if mnem != "MOVV" && mnem != "MOVW" {
return "", false
}
key := mnem + "." + bank(sc) + "." + bank(dc)
_, ok := l64FpMovTable[key]
return key, ok
}
// ---- static symbol references (pcalau12i + offset) ----
// encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the
// R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address.
func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, rd),
l64irr(l64DualTable["ADDV"].imm, 0, rd, rd),
)
}
// encodeLOONG64SBLoad emits pcalau12i r30, 0; ld rd, 0(r30) with the
// R_LOONG64_ADDR_HI/LO pair, loading from a static symbol.
func encodeLOONG64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte {
ls := l64loadStoreTable[mnem]
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
l64irr(ls.ld, 0, 30, rd),
)
}
// encodeLOONG64SBStore emits pcalau12i r30, 0; st rd, 0(r30) with the
// R_LOONG64_ADDR_HI/LO pair, storing to a static symbol.
func encodeLOONG64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) []byte {
ls := l64loadStoreTable[mnem]
if relocs != nil {
*relocs = append(*relocs,
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset},
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset},
)
}
return l64WordsLE(
l64ir(l64InstrTable["PCALAU12I"].op, 0, 30),
l64irr(ls.st, 0, 30, rs),
)
}
// ---- operand helpers ----
// l64IndexedTable holds the register-indexed load/store (ldx/stx) opcodes.
var l64IndexedTable = map[string]struct{ ld, st uint32 }{
"MOVB": {0x07000 << 15, 0x07020 << 15},
"MOVH": {0x07008 << 15, 0x07028 << 15},
"MOVW": {0x07010 << 15, 0x07030 << 15},
"MOVV": {0x07018 << 15, 0x07038 << 15},
"MOVBU": {0x07040 << 15, 0x07020 << 15},
"MOVHU": {0x07048 << 15, 0x07028 << 15},
"MOVWU": {0x07050 << 15, 0x07030 << 15},
"MOVF": {0x07060 << 15, 0x07070 << 15},
"MOVD": {0x07068 << 15, 0x07078 << 15},
}
// operandRegName returns the register name of an operand, or "".
func operandRegName(op *ast.Operand) string {
if op.Addr.Base != "" {
return op.Addr.Base
}
if op.Addr.Sym != nil && op.Addr.Sym.Name != "" {
return op.Addr.Sym.Name
}
return ""
}
// l64Reg returns the register number of an operand, or -1.
func l64Reg(op *ast.Operand) int {
return loong64RegNum(operandRegName(op))
}
// l64Imm64 returns the full 64-bit immediate value of an operand.
func l64Imm64(op *ast.Operand) int64 {
if op.Imm.HasVal {
v := op.Imm.Val
if op.Imm.Neg {
v = -v
}
return v
}
return 0
}
// l64Mem returns the base register and byte offset of a memory operand.
func l64Mem(op *ast.Operand) (rj int, off int32) {
rj = loong64RegNum(op.Addr.Base)
off = int32(op.Addr.Offset)
return
}
// l64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
// registers via the frame mapping.
func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) {
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
return loong64ResolvePseudo(op.Addr.Sym, fi)
}
return l64Mem(op)
}
// l64MemOffset returns the resolved byte offset of a memory operand.
func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 {
_, off := l64MemWithFrame(op, fi)
return off
}
// l64Label returns the label name of an operand.
func l64Label(op *ast.Operand) string {
if op.Addr.Sym != nil {
return op.Addr.Sym.Name
}
return op.Raw
}
// ---- LSX/LASX (V*/XV*) vector dispatch ----
// l64VecOperand describes a vector register operand: the 5-bit register
// number, its bank and an optional width or element suffix (V0.B16,
// V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim
// (the element index survives only in the raw text), so the suffix is
// scanned from op.Raw.
type l64VecOperand struct {
num int // 5-bit register number
lasx bool // X bank (LASX) rather than V (LSX)
width byte // suffix width letter (B/H/W/V), 0 on a bare register
lanes int // lane count of a width suffix (B16 → 16)
elem int // element index of a .T[i] suffix
hasEl bool // the suffix names an element (.T[i])
unsig bool // the suffix carries the U marker (.BU[0])
hasSuf bool // any suffix present
}
// l64ParseVecOperand parses a vector register operand with an optional
// width or element suffix. ok reports whether the operand names a vector
// register at all (V or X bank, with or without a suffix).
func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) {
if op.Kind == ast.OpImmediate {
return v, false
}
name := strings.ReplaceAll(op.Raw, " ", "")
if name == "" || (name[0] != 'V' && name[0] != 'X') {
return v, false
}
i := 1
num := 0
for i < len(name) && name[i] >= '0' && name[i] <= '9' {
num = num*10 + int(name[i]-'0')
if num > 31 {
return v, false
}
i++
}
if i == 1 {
return v, false // no register digits
}
v.num, v.lasx = num, name[0] == 'X'
if i == len(name) {
return v, true
}
if name[i] != '.' || i+2 > len(name) {
return v, false
}
i++
w := name[i]
if w != 'B' && w != 'H' && w != 'W' && w != 'V' {
return v, false
}
v.width, v.hasSuf = w, true
i++
if i < len(name) && name[i] == 'U' {
v.unsig = true
i++
}
if i < len(name) && name[i] == '[' {
// Element form .T[i]: the closing bracket ends the operand.
if name[len(name)-1] != ']' || i+2 > len(name)-1 {
return v, false
}
idx := 0
for _, c := range name[i+1 : len(name)-1] {
if c < '0' || c > '9' {
return v, false
}
idx = idx*10 + int(c-'0')
if idx > 31 {
return v, false
}
}
v.elem, v.hasEl = idx, true
return v, true
}
// Width form .T<lanes>: the trailing digits give the lane count.
lanes := 0
if i >= len(name) {
return v, false
}
for ; i < len(name); i++ {
if name[i] < '0' || name[i] > '9' {
return v, false
}
lanes = lanes*10 + int(name[i]-'0')
if lanes > 64 {
return v, false
}
}
v.lanes = lanes
return v, true
}
// l64VecSuffixWidth validates a width suffix against the bank (LSX:
// B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width
// selector of vreplgr2vr and vldrepl.
func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) {
want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2}
if lasx {
want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4}
}
lanes, ok := want[v.width]
if !ok || lanes != v.lanes {
return 0, false
}
switch v.width {
case 'B':
return 0, true
case 'H':
return 1, true
case 'W':
return 2, true
default:
return 3, true
}
}
// l64VecElementBase validates an element suffix against the bank and
// returns the encoded index field: the index rides in the rk field above a
// per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w
// and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain
// rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`.
func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) {
limit, base := 0, 0
switch v.width {
case 'B':
if lasx {
return 0, false
}
limit, base = 15, 0
case 'H':
if lasx {
return 0, false
}
limit, base = 7, 16
case 'W':
limit, base = 3, 24
if lasx {
limit, base = 7, 16
}
case 'V':
limit, base = 1, 28
if lasx {
limit, base = 3, 24
}
default:
return 0, false
}
if v.elem > limit {
return 0, false
}
return base + v.elem, true
}
// encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as
// vector plus the VMOVQ/XVMOVQ move family. handled reports whether the
// mnemonic belongs to the vector slice; the operand shapes and opcode
// constants reproduce GOARCH=loong64 `go tool asm` exactly.
func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) {
if mnem == "VMOVQ" || mnem == "XVMOVQ" {
code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi)
return code, true, err
}
lasx, ok := l64VecBank[mnem]
if !ok {
return nil, false, nil
}
ops := instr.Operands
bank := "V"
if lasx {
bank = "X"
}
vec := func(op *ast.Operand) (int, error) {
v, isVec := l64ParseVecOperand(op)
if !isVec || v.lasx != lasx || v.hasSuf {
return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw)
}
return v.num, nil
}
// Two-operand forms (vpcnt.v): INSTR vj, vd.
if l64Vec2R[mnem] {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[1])
if err != nil {
return nil, true, err
}
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil
}
// Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd.
if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) {
if len(ops) > 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
imm := int(immFromOperand(ops[0]))
if imm < e.min || imm > e.max {
return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max)
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil
}
// Vector-to-condition forms: INSTR vj, FCCn.
if l64InstrTable[mnem].format == l64Fvcf {
if len(ops) != 2 {
return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vj, err := vec(ops[0])
if err != nil {
return nil, true, err
}
if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC {
return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw)
}
fcc := loong64RegNum(operandRegName(ops[1]))
return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil
}
// Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd).
if len(ops) != 2 && len(ops) != 3 {
return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
vk, err := vec(ops[0])
if err != nil {
return nil, true, err
}
vd, err := vec(ops[len(ops)-1])
if err != nil {
return nil, true, err
}
vj := vd
if len(ops) == 3 {
if vj, err = vec(ops[1]); err != nil {
return nil, true, err
}
}
return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil
}
// encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic
// covers the whole LSX/LASX transfer surface, dispatched by operand shape
// exactly as the toolchain's table does:
//
// VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld
// VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx
// VMOVQ off(rj), vd.T vldrepl (load and replicate one element)
// VMOVQ vj, vd vori.b $0 (a register move)
// VMOVQ rj, vd.T vreplgr2vr (duplicate a general register)
// VMOVQ vj.T[i], rd vpickve2gr (extract one element)
// VMOVQ rj, vd.T[i] vinsgr2vr (insert one element)
func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) {
enc := l64VmovqTable[lasx]
bank := "V"
if lasx {
bank = "X"
}
if len(ops) != 2 {
return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops))
}
src, srcVec := l64ParseVecOperand(ops[0])
dst, dstVec := l64ParseVecOperand(ops[1])
srcMem := isMemOperand(ops[0])
dstMem := isMemOperand(ops[1])
srcIdx := srcMem && ops[0].Addr.Index != ""
dstIdx := dstMem && ops[1].Addr.Index != ""
intReg := func(op *ast.Operand) (int, error) {
if isMemOperand(op) {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
name := operandRegName(op)
if loong64RegClass(name) != l64ClsGR {
return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw)
}
return loong64RegNum(name), nil
}
// Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant),
// both operands bare registers of the same bank.
if srcVec && dstVec {
if src.hasSuf || dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank)
}
if src.lasx != lasx || dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil
}
// Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk).
if srcVec && dstMem {
if src.hasSuf || src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank)
}
if dstIdx {
rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil
}
rj, off := l64MemWithFrame(ops[1], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]")
}
return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil
}
// Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the
// load-and-replicate form VMOVQ off(rj), vd.T.
if srcMem && dstVec {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
if srcIdx {
if dst.hasSuf {
return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank)
}
rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index)
if rj < 0 || rk < 0 {
return nil, fmt.Errorf("VMOVQ: invalid register operand")
}
return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil
}
rj, off := l64MemWithFrame(ops[0], fi)
if rj < 0 || off < -2048 || off > 2047 {
return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]")
}
op := enc.ld
if dst.hasSuf {
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw)
}
switch w {
case 0:
op = enc.replB
case 1:
op = enc.replH
case 2:
op = enc.replW
default:
op = enc.replD
}
}
return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil
}
// Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned).
if srcVec && src.hasEl && !dstVec && !dstMem {
if src.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
idx, ok := l64VecElementBase(lasx, src)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw)
}
rd, err := intReg(ops[1])
if err != nil {
return nil, err
}
op := enc.pickS
if src.unsig {
op = enc.pickU
}
return l64wordLE(l64irr(op, idx, src.num, rd)), nil
}
// Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and
// VMOVQ rj, vd.T (vreplgr2vr).
if !srcVec && !srcMem && dstVec && dst.hasSuf {
if dst.lasx != lasx {
return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank)
}
rs, err := intReg(ops[0])
if err != nil {
return nil, err
}
if dst.hasEl {
idx, ok := l64VecElementBase(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil
}
w, ok := l64VecSuffixWidth(lasx, dst)
if !ok {
return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw)
}
return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil
}
return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw)
}