// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm import ( "fmt" "math/bits" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-sdk/ast" ) // assembleLOONG64 assembles a LoongArch (loong64) TEXT function body into // machine code. Every instruction is 4 bytes; the MOV pseudo-instruction and // the immediate-arithmetic forms expand to 2-5 instructions when the // immediate does not fit, so the layout is computed in two passes (sizes, // then encoding with resolved branch targets). // // The emitted bytes match the Go toolchain's loong64 assembler, which is the // ground-truth oracle: prologue/epilogue (including the large-frame R30 // materialisations), FP/SP frame mapping, the stack-split guard classes, and // branch encodings all follow cmd/internal/obj/loong64. The morestack block // at the end of split functions carries the runtime.morestack_noctxt call. func assembleLOONG64(t *ast.Text) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, error) { fi := loong64ComputeFrame(t) prologue := loong64Prologue(fi) guardLen := loong64GuardLen(fi) chain := loong64JumpChain(t) resolve := func(name string) string { if r, ok := chain[name]; ok { return r } return name } var relocs []Reloc var spadj []SpadjStep // The prologue raises the SP delta by autosize; the boundary is reported // after the SP adjust instruction, exactly as the toolchain's pctospadj // does. The prologue (3 instructions when a frame is present) may // materialise its store or adjust through R30, which widens it. if fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: guardLen + (loong64StoreWords(fi.autosize)+loong64AdjustWords(-int64(fi.autosize)))*4, Value: fi.autosize}) } // The toolchain's parser counts N(PC) displacements over the source // instructions at a uniform 4 bytes each, so a PC-relative branch // resolves to the instruction N slots away in body order; the resolved // target then participates in layout and loop-head padding like any // branch target. instrs := make([]*ast.Instr, 0, len(t.Body)) for _, stmt := range t.Body { if in, ok := stmt.(*ast.Instr); ok && strings.ToUpper(in.Mnemonic.Text) != "PCALIGN" { instrs = append(instrs, in) } } parseIndex := make(map[*ast.Instr]int, len(instrs)) for i, in := range instrs { parseIndex[in] = i } pcRelTarget := make(map[*ast.Instr]*ast.Instr) for _, in := range instrs { off, ok := loong64PCRelOffset(in) if !ok { continue } tgt := parseIndex[in] + off if tgt < 0 || tgt >= len(instrs) { continue } pcRelTarget[in] = instrs[tgt] } // Pass 1: label offsets from the instruction sizes. PCALIGN contributes // only its padding. On top of the explicit PCALIGNs, the toolchain pads // every backward-branch target (loop head) to a 16-byte boundary, so the // layout runs to a fixpoint over the alignment set. loopAligns := map[string]bool{} alignInstrs := map[*ast.Instr]bool{} for { offsets, _, pcs, _ := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs) changed := false for _, in := range instrs { // A backward PC-relative target is the resolved instruction. if tgt, ok := pcRelTarget[in]; ok && pcs[tgt] < pcs[in] && !alignInstrs[tgt] { alignInstrs[tgt] = true changed = true } target, ok := loong64BranchTarget(in) if !ok { continue } tOff, ok := offsets[target] if !ok || tOff >= pcs[in] || loopAligns[target] { continue } loopAligns[target] = true changed = true } if !changed { break } } // Final layout with the complete alignment set. offsets, alignPad, pcs, bodyEnd := loong64Layout(t, guardLen+len(prologue), fi, loopAligns, alignInstrs) bodyLen := bodyEnd - (guardLen + len(prologue)) pcRelPcs := make(map[*ast.Instr]int, len(pcRelTarget)) for in, tgt := range pcRelTarget { pcRelPcs[in] = pcs[tgt] } var out []byte if fi.needSplit { out = append(out, loong64GuardBytes(fi, guardLen+len(prologue)+bodyLen)...) } out = append(out, prologue...) pc := guardLen + len(prologue) preCount := len(relocs) var lines []LineEntry for _, stmt := range t.Body { in, ok := stmt.(*ast.Instr) if !ok { continue } // PCALIGN pads to the requested boundary with andi $0, $0, 0, the // architecture's NOP, and encodes to nothing itself. if strings.ToUpper(in.Mnemonic.Text) == "PCALIGN" { pad := loong64PCAlignPad(pc, in) out = append(out, loong64PadBytes(pad)...) pc += pad continue } // Loop-head alignment padding precedes the instruction. if pad := alignPad[in]; pad > 0 { out = append(out, loong64PadBytes(pad)...) pc += pad } code, err := encodeLOONG64Instr(in, pc, offsets, fi, &relocs, resolve, pcRelPcs) if err != nil { return nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err) } for j := preCount; j < len(relocs); j++ { // Make the relocation offsets function-relative: each instruction // records its reloc offset relative to its own start, and pc is // that instruction's offset from the function start (prologue // included). After shifts by the same amount. relocs[j].Off += pc relocs[j].After += pc } preCount = len(relocs) lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line}) // The RET's epilogue closes the frame: the SP delta returns to zero // after the frame-deallocating ADDV. if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 { spadj = append(spadj, SpadjStep{PC: pc + loong64EpilogueWords(fi)*4, Value: 0}) } out = append(out, code...) pc += len(code) } if fi.needSplit { block, blReloc := loong64MoreStackBlock(pc) out = append(out, block...) relocs = append(relocs, blReloc) pc += len(block) } return out, offsets, relocs, lines, spadj, nil } // loong64PCRelOffset reports the N of a branch operand spelled N(PC): the // displacement counted in source instructions from the branch itself. func loong64PCRelOffset(instr *ast.Instr) (int, bool) { mnem := strings.ToUpper(instr.Mnemonic.Text) branch := false switch mnem { case "JMP": branch = len(instr.Operands) == 1 case "JAL", "CALL", "BL": branch = len(instr.Operands) == 1 || len(instr.Operands) == 2 case "BFPT", "BFPF": branch = len(instr.Operands) == 1 case "BEQ", "BNE", "BLT", "BGE", "BLTU", "BGEU", "BEQZ", "BNEZ", "BLTZ", "BGEZ", "BLEZ", "BGTZ": branch = len(instr.Operands) >= 2 } if !branch { return 0, false } op := instr.Operands[len(instr.Operands)-1] if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "PC" { return int(op.Addr.Offset), true } return 0, false } // loong64Layout walks the function body once and returns the label offsets, // the loop-alignment padding due before each instruction (a pad of 0 needs // nothing), the pc each instruction starts at (its padding included) and the // first pc past the body. Explicit PCALIGN pads, the alignment pads for the // labels in aligns and those for the instructions in alignInstrs (backward // PC-relative targets) all contribute, mirroring the toolchain's layout // pass. func loong64Layout(t *ast.Text, start int, fi loong64FrameInfo, aligns map[string]bool, alignInstrs map[*ast.Instr]bool) (map[string]int, map[*ast.Instr]int, map[*ast.Instr]int, int) { offsets := map[string]int{} alignPad := map[*ast.Instr]int{} pcs := map[*ast.Instr]int{} pos := start pendingAlign := false var pendingNames []string explicit := false for _, stmt := range t.Body { switch s := stmt.(type) { case *ast.Label: if aligns[s.Name.Text] { pendingAlign = true } pendingNames = append(pendingNames, s.Name.Text) // Provisional: a branch to the label lands here unless a loop // alignment pad follows, in which case the label resolves to the // padded instruction (the toolchain's labels bind to the branch // target instruction, which the padding pass precedes). offsets[s.Name.Text] = pos case *ast.Instr: if strings.ToUpper(s.Mnemonic.Text) == "PCALIGN" { pos += loong64PCAlignPad(pos, s) explicit = true continue } if pendingAlign { pendingAlign = false if pos&15 != 0 { alignPad[s] = 16 - pos&15 } } if alignInstrs[s] && pos&15 != 0 { alignPad[s] = 16 - pos&15 } if !explicit { for _, n := range pendingNames { offsets[n] = pos + alignPad[s] } } pendingNames = nil explicit = false pcs[s] = pos + alignPad[s] pos += alignPad[s] + loong64InstrSize(s, fi) } } return offsets, alignPad, pcs, pos } // loong64BranchTarget reports the local label a branch-like instruction // transfers to, the loop-head signal the toolchain derives from backward // branch targets. func loong64BranchTarget(instr *ast.Instr) (string, bool) { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands var op *ast.Operand switch { case mnem == "JMP" || mnem == "JAL" || mnem == "BFPT" || mnem == "BFPF": if len(ops) != 1 { return "", false } op = ops[0] case mnem == "TEQ" || mnem == "TNE": return "", false case len(ops) >= 2: op = ops[len(ops)-1] default: return "", false } if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Sym.Name != "" { return op.Addr.Sym.Name, true } return "", false } // loong64JumpChain precomputes jump-to-jump folding, mirroring the linker's // branch-chasing pass: a label whose first instruction is an unconditional // local jump redirects its own jumpers to the ultimate target. The Go // toolchain chases these chains before it encodes branches, so matching its // bytes requires the same redirection. func loong64JumpChain(t *ast.Text) map[string]string { leadsTo := map[string]string{} for i, stmt := range t.Body { l, ok := stmt.(*ast.Label) if !ok { continue } j := i + 1 for j < len(t.Body) { if _, isLabel := t.Body[j].(*ast.Label); !isLabel { break } j++ } if j >= len(t.Body) { continue } in, ok := t.Body[j].(*ast.Instr) if !ok { continue } mnem := strings.ToUpper(in.Mnemonic.Text) if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 { continue } if name, ok := l64LabelOK(in.Operands[0]); ok { leadsTo[l.Name.Text] = name } } chain := map[string]string{} for name := range leadsTo { visited := map[string]bool{name: true} cur := name for { next, ok := leadsTo[cur] if !ok || visited[next] { break } visited[next] = true cur = next } if cur != name { chain[name] = cur } } return chain } // l64LabelOK returns the local label name of a jump operand. func l64LabelOK(op *ast.Operand) (string, bool) { if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Sym.Name != "" { return op.Addr.Sym.Name, true } return "", false } // l64SubToAdd rewrites the SUB family with an immediate first operand onto // its ADD counterpart with the negated immediate: LoongArch has no // subtract-immediate instructions, and the toolchain folds SUB $v into the // ADD immediate form through the same optab matching (the $0 fold into 3R // and the large-constant materialisations included). The negation is the // second result; the operand is left untouched because the size pass // normalises the same instruction. func l64SubToAdd(mnem string, ops []*ast.Operand) (string, bool) { if len(ops) >= 2 && isImmOperand(ops[0]) { switch mnem { case "SUB": return "ADD", true case "SUBW": return "ADDW", true case "SUBV", "SUBVU": return "ADDV", true } } return mnem, false } // loong64InstrSize returns the encoded size of an instruction: 4 bytes for // most, more for the multi-instruction expansions. func loong64InstrSize(instr *ast.Instr, fi loong64FrameInfo) int { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands var neg bool mnem, neg = l64SubToAdd(mnem, ops) if mnem == "RET" { return len(loong64Return(fi)) } switch mnem { case "END", "FUNCDATA", "PCDATA": return 0 // bookkeeping statements contribute no bytes case "GETCALLERPC": return 4 // or rd, r1, r0 case "TEQ", "TNE": return 8 // bne/beq over the BREAK, then BREAK case "PRELDX": return 20 // the four-instruction constant materialisation + preldx case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": return loong64MovSize(mnem, ops, fi) case "ADD", "ADDW", "ADDV", "ADDVU", "AND", "OR", "XOR", "SGT", "SGTU": if len(ops) >= 2 && isImmOperand(ops[0]) { v := l64Imm64(ops[0]) if neg { v = -v } if v == 0 { return 4 // folds into the 3R form (rk = R0) } switch mnem { case "ADD", "ADDW", "ADDV", "ADDVU", "SGT", "SGTU": // C_US12CON (−2048..0x7ff) encodes directly as addi/slti. if v >= -2048 && v <= 0x7ff { return 4 } // C_U12CON (0x800..0xfff) → ori r30, r0, v; op rd, rj, r30. if v >= 0x800 && v <= 0xfff { return 8 } default: // AND/OR/XOR // C_UU12CON (0..0x7ff) encodes directly as andi/ori/xori. if v >= 0 && v <= 0x7ff { return 4 } // C_S12CON (−2048..−1) → addi.d r30, r0, v; op rd, rj, r30. if v >= -2048 && v < 0 { return 8 } } // 0x800..0xfff for AND/OR/XOR and the 32/64-bit ranges go through // the lu12i.w materialisation. if v == int64(int32(v)) { if v&0xfff == 0 && (v < 0x800 || v > 0xfff) { return 8 // lu12i.w r30, v>>12; op rd, rj, r30 } return 12 // lu12i.w r30, v>>12; ori r30, r30, v; op rd, rj, r30 } return 4 * (len(l64DconMovWords(0, v)) + 1) // dcon materialisation + op } } return 4 } // loong64PCAlignPad returns the padding PCALIGN inserts before the next // instruction so that it starts at the requested boundary relative to the // function start. The boundary must be a power of two between 8 and 2048, as // the toolchain requires; anything else pads nothing. func loong64PCAlignPad(pos int, instr *ast.Instr) int { if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) { return 0 } align := int(immFromOperand(instr.Operands[0])) if align < 8 || align > 2048 || align&(align-1) != 0 { return 0 } return (align - pos%align) % align } // loong64PadBytes renders PCALIGN padding: the toolchain emits andi $0, $0, 0 // (the architecture's NOP) for every full 4 bytes of pad. func loong64PadBytes(pad int) []byte { nop := l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)) out := make([]byte, 0, pad/4*len(nop)) for i := 0; i < pad/4; i++ { out = append(out, nop...) } return out } // encodeLOONG64Instr encodes a single LoongArch instruction. func encodeLOONG64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi loong64FrameInfo, relocs *[]Reloc, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) { mnem := strings.ToUpper(instr.Mnemonic.Text) ops := instr.Operands // The SUB family with an immediate first operand folds onto the ADD // immediate form with the negated immediate; the negation happens on a // copy of the operand, never on the shared syntax tree. mnem, neg := l64SubToAdd(mnem, ops) if neg { c := *ops[0] c.Imm.Val = -c.Imm.Val ops2 := make([]*ast.Operand, len(ops)) ops2[0] = &c copy(ops2[1:], ops[1:]) ops = ops2 } // Pseudo-instructions and the branches first. switch mnem { case "RET": return loong64Return(fi), nil case "NOP", "NOOP": // andi r0, r0, 0 return l64wordLE(l64irr(l64DualTable["AND"].imm, 0, 0, 0)), nil case "UNDEF": // break 0 return l64wordLE(l64i15(l64InstrTable["BREAK"].op, 0)), nil case "WORD": if len(ops) != 1 { return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops)) } return l64wordLE(uint32(immFromOperand(ops[0]))), nil case "END", "FUNCDATA", "PCDATA", "GETCALLERPC": // The assembler's bookkeeping statements. END, FUNCDATA and PCDATA // contribute no bytes, the same shapes GOARCH=loong64 go tool asm // accepts and emits nothing for; GETCALLERPC reads the caller's // address out of R1 (RA) as or rd, r1, r0. switch mnem { case "END": if len(ops) != 0 { return nil, fmt.Errorf("END expects no operands, got %d", len(ops)) } return nil, nil case "FUNCDATA": if len(ops) != 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)") } return nil, nil case "PCDATA": if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) { return nil, fmt.Errorf("PCDATA expects $n, $n") } return nil, nil } if len(ops) != 1 || isMemOperand(ops[0]) || loong64RegClass(operandRegName(ops[0])) != l64ClsGR { return nil, fmt.Errorf("GETCALLERPC expects a general register") } rd := loong64RegNum(operandRegName(ops[0])) if rd < 0 { return nil, fmt.Errorf("GETCALLERPC: invalid register operand") } return l64wordLE(l64rrr(l64movRegTable["MOVV"].op, 0, 1, rd)), nil case "NEGW", "NEGV": // The integer negation pseudo is a subtract from zero: // NEGW src, dst → sub.w r0, src, dst. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } src, dst := l64Reg(ops[0]), l64Reg(ops[1]) if src < 0 || dst < 0 { return nil, fmt.Errorf("invalid register operand") } sub := l64InstrTable["SUBW"].op if mnem == "NEGV" { sub = l64InstrTable["SUBV"].op } return l64wordLE(l64rrr(sub, src, 0, dst)), nil case "TEQ", "TNE": // The trap pseudo expands to two instructions: bne/beq rj, rd over // the BREAK (offset 2 instruction units), then BREAK $code. if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } code := int(immFromOperand(ops[0])) rj, rd := 0, l64Reg(ops[len(ops)-1]) if len(ops) == 3 { rj = l64Reg(ops[1]) } if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } bop := l64branchTable["BNE"] if mnem == "TNE" { bop = l64branchTable["BEQ"] } return l64WordsLE( l64irr16(bop, 2, rj, rd), l64i15(l64InstrTable["BREAK"].op, code), ), nil case "PRELDX": // preldx offset(Rbase), $n, $hint: the 64-bit descriptor n packs // (addrSeq, blockSize, blockNums, stride); the constant v built from // it materialises in R30 across four instructions, then the preldx. if len(ops) != 3 || !isMemOperand(ops[0]) || !isImmOperand(ops[1]) || !isImmOperand(ops[2]) { return nil, fmt.Errorf("PRELDX expects offset(reg), $n, $hint") } rj := loong64RegNum(ops[0].Addr.Base) if rj < 0 { return nil, fmt.Errorf("invalid register operand") } n := uint64(l64Imm64(ops[1])) hint := int(l64Imm64(ops[2])) addrSeq := (n >> 0) & 0x1 blkSize := (n >> 1) & 0x7ff blkNums := (n >> 12) & 0x1ff stride := (n >> 21) & 0xffff v := uint64(ops[0].Addr.Offset)&0xffff + addrSeq<<16 + ((blkSize/16)-1)<<20 + (blkNums-1)<<32 + stride<<44 const ( lu12iw = 0x0a << 25 lu32id = 0x0b << 25 lu52id = 0x00c << 22 ori = 0x00e << 22 preldx = 0x7058 << 15 ) return l64WordsLE( l64ir(lu12iw, int(uint32(v>>12)), 30), l64irr(ori, int(uint32(v)), 30, 30), l64ir(lu32id, int(uint32(v>>32)), 30), l64irr(lu52id, int(uint32(v>>52)), 30, 30), l64rrr(preldx, 30, rj, hint), ), nil case "JMP", "B": return encodeLOONG64Branch(instr, mnem, pc, offsets, false, resolve, relocs, pcRelPcs) case "JAL", "CALL", "BL": return encodeLOONG64Branch(instr, mnem, pc, offsets, true, resolve, relocs, pcRelPcs) case "MOV", "MOVB", "MOVH", "MOVW", "MOVV", "MOVBU", "MOVHU", "MOVWU", "MOVF", "MOVD": return encodeLOONG64Mov(instr, mnem, fi, relocs) } // 16-bit branches (BEQ/BNE/BLT/BGE/BLTU/BGEU) and JIRL. if op, ok := l64branchTable[mnem]; ok { if mnem == "JIRL" { return encodeLOONG64Jirl(op, ops) } return encodeLOONG64Branch16(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs) } // Single-register branches with 21-bit offsets (BLTZ/BGEZ/BLEZ/BGTZ, // BFPT/BFPF; BEQZ/BNEZ are reached through BEQ/BNE with R0). if op, ok := l64branch21Table[mnem]; ok { return encodeLOONG64Branch21(instr, mnem, op, ops, pc, offsets, resolve, pcRelPcs) } // B/BL aliases reached only via JMP/JAL above. // The dual-form arithmetic mnemonics: register (3R) or immediate (2RI12). if de, ok := l64DualTable[mnem]; ok { if len(ops) >= 2 && isImmOperand(ops[0]) { if de.shift { // INSTR $shamt, rd or INSTR $shamt, rj, rd. if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } shamt := int(immFromOperand(ops[0])) rd := l64Reg(ops[len(ops)-1]) rj := rd if len(ops) == 3 { rj = l64Reg(ops[1]) } if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } // $0 folds into the register form (the toolchain matches the // zero constant against the 3R optab entry first). if shamt == 0 { return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil } // The .d variants take a 6-bit amount, the .w variants 5 bits. if isLoong64ShiftD(de.imm) { shamt &= 0x3f } else { shamt &= 0x1f } return l64wordLE(l64irr(de.imm, shamt, rj, rd)), nil } return encodeLOONG64ImmArith(mnem, de, ops) } // Register form: 3R. if len(ops) == 3 { rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]) if rk < 0 || rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64rrr(de.rrr, rk, rj, rd)), nil } if len(ops) == 2 { rk, rd := l64Reg(ops[0]), l64Reg(ops[1]) if rk < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64rrr(de.rrr, rk, rd, rd)), nil } return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } // The LSX/LASX vector slice and the VMOVQ/XVMOVQ move family, before // the integer/FP table (their mnemonics overlap the table's 2R format // but resolve vector-bank registers). if code, handled, err := encodeLOONG64Vector(instr, mnem, fi); handled { if err != nil { return nil, err } return code, nil } enc, ok := l64InstrTable[mnem] if !ok { return nil, fmt.Errorf("unsupported loong64 instruction %q", mnem) } switch enc.format { case l64Frrr: // INSTR rk, rj, rd (3 operands) or INSTR rk, rd (rj = rd). switch len(ops) { case 3: rk, rj, rd := l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]) if rk < 0 || rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil case 2: rk, rd := l64Reg(ops[0]), l64Reg(ops[1]) if rk < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64rrr(enc.op, rk, rd, rd)), nil } return nil, fmt.Errorf("%s expects 2 or 3 register operands, got %d", mnem, len(ops)) case l64Frr: // INSTR rj, rd. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rj, rd := l64Reg(ops[0]), l64Reg(ops[1]) if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64rr(enc.op, rj, rd)), nil case l64Firr: // LU52ID: INSTR $imm, rd or INSTR $imm, rj, rd. if len(ops) < 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects an immediate operand", mnem) } imm := int(immFromOperand(ops[0])) rd := l64Reg(ops[len(ops)-1]) rj := rd if len(ops) == 3 { rj = l64Reg(ops[1]) } if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64irr(enc.op, imm, rj, rd)), nil case l64Firr16: // ADDV16: INSTR $imm, rd or INSTR $imm, rj, rd; the immediate must be // a multiple of 65536 and is shifted right by 16. if len(ops) < 2 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects an immediate operand", mnem) } v := int(immFromOperand(ops[0])) if v&0xFFFF != 0 { return nil, fmt.Errorf("%s: the constant must be a multiple of 65536", mnem) } rd := l64Reg(ops[len(ops)-1]) rj := rd if len(ops) == 3 { rj = l64Reg(ops[1]) } if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64irr16(enc.op, v>>16, rj, rd)), nil case l64Firr14: // LL/SC/MOVWP: INSTR mem, rd (load) or INSTR rd, mem (store); the // 14-bit offset is scaled by 4 (byte offset >> 2). rd, rj, off, load, err := l64MemOperands(ops, fi) if err != nil { return nil, err } op := enc.op if load && (mnem == "MOVWP" || mnem == "MOVVP") { // ldptr.{w,d} = stptr.{w,d} minus the LSB of the opcode field. op -= 1 << 24 } return l64wordLE(l64irr14(op, int(off)>>2, rj, rd)), nil case l64Fir20: // LU12IW/LU32ID/PCALAU12I/PCADDU12I: INSTR rd, $imm. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rd := l64Reg(ops[0]) if rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64ir(enc.op, int(immFromOperand(ops[1])), rd)), nil case l64Frrrr: // FMADD/FMSUB/FNMADD/FNMSUB: INSTR fa, fk, fj, fd (4 operands) or // INSTR fa, fk, fd (fj = fd). fa, fk, fj, fd, err := l64FmaOperands(ops) if err != nil { return nil, err } return l64wordLE(l64rrrr(enc.op, fa, fk, fj, fd)), nil case l64Firir: // BSTRINS/BSTRPICK: INSTR $msb, rj, $lsb, rd (or $msb, rj, rd with // lsb = 0). if len(ops) != 4 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops)) } msb := int(immFromOperand(ops[0])) lsb := 0 rj := l64Reg(ops[1]) rd := l64Reg(ops[len(ops)-1]) if len(ops) == 4 { lsb = int(immFromOperand(ops[2])) } if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } // The toolchain validates the bit numbers ("illegal bit number"): // 0..31 for the .w forms, 0..63 for the .d forms, lsb <= msb. b := 64 if strings.HasSuffix(mnem, "W") { b = 32 } if msb < 0 || msb >= b || lsb < 0 || lsb >= b || lsb > msb { return nil, fmt.Errorf("%s: illegal bit number (msb %d, lsb %d)", mnem, msb, lsb) } return l64wordLE(l64irir(enc.op, msb, rj, lsb, rd)), nil case l64Firrr: // ALSL: INSTR $sa, rj, rk, rd (the toolchain's optab places rj in // the second register position); the source amount is 1-4, encoded // as sa-1. if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } sa := int(immFromOperand(ops[0])) - 1 rj, rk, rd := l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3]) if sa < 0 || sa > 3 { return nil, fmt.Errorf("shift amount out of range [1, 4]") } if rk < 0 || rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64irrr(enc.op, sa, rk, rj, rd)), nil case l64Fi15: // SYSCALL/BREAK/DBAR: no operands, or SYSCALL $code / BREAK $code. code := 0 if len(ops) == 1 { code = int(immFromOperand(ops[0])) } else if len(ops) > 1 { return nil, fmt.Errorf("%s expects at most 1 operand, got %d", mnem, len(ops)) } return l64wordLE(l64i15(enc.op, code)), nil case l64Fam: // AM* val, (addr), result. if len(ops) != 3 { return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) } rk := l64Reg(ops[0]) rj, _ := l64Mem(ops[1]) rd := l64Reg(ops[2]) if rk < 0 || rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return l64wordLE(l64rrr(enc.op, rk, rj, rd)), nil case l64Frdtime: // RDTIME* rd, rj (rd at bits [9:5], rj at bits [4:0]). if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rd, rj := l64Reg(ops[0]), l64Reg(ops[1]) if rd < 0 || rj < 0 { return nil, fmt.Errorf("invalid register operand") } return l64wordLE(l64rr(enc.op, rd, rj)), nil case l64Fpreld: // PRELD off(rj), $hint. if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } rj, off := l64Mem(ops[0]) hint := int(immFromOperand(ops[1])) if rj < 0 { return nil, fmt.Errorf("invalid operand in %s", mnem) } return l64wordLE(l64irr5i(enc.op, int(off), rj, hint)), nil } return nil, fmt.Errorf("cannot encode %s with %d operands", mnem, len(ops)) } // encodeLOONG64Branch encodes a label or indirect jump/call: // // JMP/B label → b label JMP/B (rj) → jirl r0, rj, 0 // JAL/CALL/BL label → bl label JAL/CALL/BL (rj) → jirl r1, rj, 0 func encodeLOONG64Branch(instr *ast.Instr, mnem string, pc int, offsets map[string]int, link bool, resolve func(string) string, relocs *[]Reloc, pcRelPcs map[*ast.Instr]int) ([]byte, error) { if len(instr.Operands) != 1 { return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(instr.Operands)) } op := instr.Operands[0] // PC-relative displacement: N(PC) resolves to the instruction N slots // away in source order (the toolchain's parse-time count), and the field // carries the final pc distance in instruction units. if op.Addr.Sym == nil && op.Addr.Base == "PC" { targetPc, ok := pcRelPcs[instr] if !ok { return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, op.Addr.Offset) } v := (targetPc - pc) >> 2 opc := l64jumpTable[mnem] return l64wordLE(l64bbl(opc, v)), nil } if isMemOperand(op) && op.Addr.Base != "" && op.Addr.Index == "" && op.Addr.Sym == nil { // Indirect: (rj) → jirl. rj := loong64RegNum(op.Addr.Base) if rj < 0 { return nil, fmt.Errorf("invalid register operand") } rd := 0 if link { rd = 1 // link register } return l64wordLE(l64irr16(l64branchTable["JIRL"], 0, rj, rd)), nil } // Direct symbol: sym+off(SB) → b/bl with an R_CALLLOONG64 relocation // (the linker fills the offset), as the toolchain does for CALL/BL/JAL // and for tail-calling JMP. if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" { opc := l64jumpTable["B"] if link { opc = l64jumpTable["BL"] } if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 4, Name: op.Addr.Sym.Name, Kind: RelLoong64Branch, Addend: op.Addr.Sym.Offset}) } return l64wordLE(l64bbl(opc, 0)), nil } // Direct: label → b/bl. target := resolve(l64Label(op)) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } v := (targetOff - pc) >> 2 if v < -1<<25 || v >= 1<<25 { return nil, fmt.Errorf("branch to %q too far (26-bit range)", target) } opc := l64jumpTable[mnem] return l64wordLE(l64bbl(opc, v)), nil } // encodeLOONG64Jirl encodes the raw JIRL spelling, JIRL rd, rj, offset, the // form the verify trampolines use. The (rj) indirect form without an offset // is handled by encodeLOONG64Branch. func encodeLOONG64Jirl(op uint32, ops []*ast.Operand) ([]byte, error) { if len(ops) != 3 { return nil, fmt.Errorf("JIRL expects 3 operands, got %d", len(ops)) } rd := l64Reg(ops[0]) rj := l64Reg(ops[1]) if rd < 0 || rj < 0 { return nil, fmt.Errorf("invalid register operand") } off, ok := l64offsetOperand(ops[2]) if !ok { return nil, fmt.Errorf("JIRL expects an immediate offset, got %q", ops[2].Raw) } if (int64(off)<<16)>>16 != int64(off) { return nil, fmt.Errorf("JIRL offset %d out of the 16-bit range", off) } return l64wordLE(l64irr16(op, int(off), rj, rd)), nil } // l64offsetOperand reads a bare numeric branch offset: an immediate ($n) or a // plain number, which parses as an empty address carrying the digits in Raw. func l64offsetOperand(op *ast.Operand) (int32, bool) { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return int32(v), true } if op.Kind == ast.OpAddr && op.Addr.Sym == nil && op.Addr.Base == "" && op.Addr.Index == "" { if v, err := strconv.ParseInt(op.Raw, 0, 64); err == nil { return int32(v), true } } return 0, false } // encodeLOONG64Branch16 encodes a 16-bit branch (BEQ/BNE/BLT/BGE/BLTU/BGEU): // INSTR rj, rd, label, or INSTR rj, label with rd = R0, which the toolchain // turns into the 21-bit BEQZ/BNEZ form when the register is the only operand. func encodeLOONG64Branch16(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } var target string var v int lastOp := ops[len(ops)-1] if lastOp.Kind == ast.OpAddr && lastOp.Addr.Sym == nil && lastOp.Addr.Base == "PC" { // N(PC) resolves to the instruction N slots away in source order. targetPc, ok := pcRelPcs[instr] if !ok { return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, lastOp.Addr.Offset) } v = (targetPc - pc) >> 2 } else { target = resolve(l64Label(lastOp)) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } v = (targetOff - pc) >> 2 } if len(ops) == 2 { // Single register: BEQ rj, label → beqz (21-bit), and the BLTZ/ // BGEZ-family aliases encoded with rj in the rj field. rj := l64Reg(ops[0]) if rj < 0 { return nil, fmt.Errorf("invalid register operand") } if mnem == "BLTU" || mnem == "BGEU" { // The unsigned compares have no single-register pseudo: the // toolchain keeps the register-register form with rd = R0 // (bltu rj, r0 is never taken), not a sometimes-taken beqz. if (v<<16)>>16 != v { return nil, fmt.Errorf("branch to %q too far (16-bit range)", target) } return l64wordLE(l64irr16(op, v, rj, 0)), nil } if (v<<11)>>11 != v { return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) } zop := l64branch21Table["BEQZ"] if mnem == "BNE" { zop = l64branch21Table["BNEZ"] } if mnem == "BLT" || mnem == "BLTZ" || mnem == "BGTZ" { zop = l64branch21Table["BLTZ"] } if mnem == "BGE" || mnem == "BGEZ" || mnem == "BLEZ" { zop = l64branch21Table["BGEZ"] } return l64wordLE(l64ir21(zop, v, rj)), nil } // Two registers: BEQ rj, rd, label. When one is R0 the toolchain // re-encodes as the 21-bit BEQZ/BNEZ form. rj, rd := l64Reg(ops[0]), l64Reg(ops[1]) if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } if rj == 0 { rj, rd = rd, 0 } if rd == 0 { if (v<<11)>>11 != v { return nil, fmt.Errorf("branch to %q too far (21-bit range)", target) } zop := l64branch21Table["BEQZ"] if mnem == "BNE" { zop = l64branch21Table["BNEZ"] } return l64wordLE(l64ir21(zop, v, rj)), nil } if (v<<16)>>16 != v { return nil, fmt.Errorf("branch to %q too far (16-bit range)", target) } return l64wordLE(l64irr16(op, v, rj, rd)), nil } // encodeLOONG64Branch21 encodes a single-register branch: BLTZ/BGEZ and // BFPT/BFPF use the 21-bit offset form (register in the rj field), while // BGTZ/BLEZ, which the toolchain encodes with the register in the rd field // and a 16-bit offset, are handled separately. func encodeLOONG64Branch21(instr *ast.Instr, mnem string, op uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string, pcRelPcs map[*ast.Instr]int) ([]byte, error) { isBF := mnem == "BFPT" || mnem == "BFPF" if len(ops) != 2 && !(isBF && (len(ops) == 1 || len(ops) == 2)) { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } var rj int tgtOp := ops[len(ops)-1] if isBF { // BFPT/BFPF test an FCC condition register, defaulting to FCC0 when // spelled without one. rj = 0 if len(ops) == 2 { rj = l64Reg(ops[0]) if rj < 0 { return nil, fmt.Errorf("invalid register operand") } } } else { rj = l64Reg(ops[0]) if rj < 0 { return nil, fmt.Errorf("invalid register operand") } } var v int if tgtOp.Kind == ast.OpAddr && tgtOp.Addr.Sym == nil && tgtOp.Addr.Base == "PC" { // N(PC) resolves to the instruction N slots away in source order. targetPc, ok := pcRelPcs[instr] if !ok { return nil, fmt.Errorf("%s: PC-relative target %d out of range", mnem, tgtOp.Addr.Offset) } v = (targetPc - pc) >> 2 } else { target := resolve(l64Label(tgtOp)) targetOff, ok := offsets[target] if !ok { return nil, fmt.Errorf("undefined label %q%s", target, suggestLabel(target, offsets)) } v = (targetOff - pc) >> 2 } if mnem == "BGTZ" || mnem == "BLEZ" { // The toolchain swaps the register into the rd field and keeps the // 16-bit offset form. if (v<<16)>>16 != v { return nil, fmt.Errorf("branch %d too far (16-bit range)", v) } return l64wordLE(l64irr16(op, v, 0, rj)), nil } if (v<<11)>>11 != v { return nil, fmt.Errorf("branch %d too far (21-bit range)", v) } return l64wordLE(l64ir21(op, v, rj)), nil } // encodeLOONG64ImmArith encodes an immediate arithmetic/logic instruction, // expanding the immediate exactly as the toolchain's aclass classifies it: // // ADD/SGT family: −2048..0x7ff → addi/slti directly (4 bytes) // 0x800..0xfff → ori r30, r0, v; op rd, rj, r30 (8) // AND/OR/XOR: 0..0x7ff → andi/ori/xori directly (4) // −2048..−1 → addi.d r30, r0, v; op rd, rj, r30 (8) // 32-bit: lu12i.w r30, v>>12 [; ori r30, r30, v]; op (8/12) // 64-bit: lu12i.w + ori + lu32i.d + lu52i.d + op (20) func encodeLOONG64ImmArith(mnem string, de l64DualEnc, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } v := l64Imm64(ops[0]) rd := l64Reg(ops[len(ops)-1]) rj := rd if len(ops) == 3 { rj = l64Reg(ops[1]) } if rj < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } // The two immediate families classify differently. additive := mnem == "ADD" || mnem == "ADDW" || mnem == "ADDV" || mnem == "ADDVU" || mnem == "SGT" || mnem == "SGTU" if additive { if v == 0 { // $0 folds into the 3R form (rk = R0), matching the toolchain's // optab matching of the zero constant against the register form. return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil } if v >= -2048 && v <= 0x7ff { return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil } if v >= 0x800 && v <= 0xfff { return l64WordsLE( l64irr(0x00e<<22, int(v), 0, 30), // ori r30, r0, v l64rrr(de.rrr, 30, rj, rd), ), nil } } else { if v == 0 { return l64wordLE(l64rrr(de.rrr, 0, rj, rd)), nil } if v >= 0 && v <= 0x7ff { return l64wordLE(l64irr(de.imm, int(v), rj, rd)), nil } if v >= -2048 && v < 0 { return l64WordsLE( l64irr(0x00b<<22, int(v), 0, 30), // addi.d r30, r0, v l64rrr(de.rrr, 30, rj, rd), ), nil } } // 32/64-bit constants are materialised in R30 (the assembler temp), // using the same dcon classification as the toolchain's case 24/60/70/ // 71/72 sequences. const ( lu12iw = 0x0a << 25 ori = 0x00e << 22 ) if v == int64(int32(v)) { if v&0xfff == 0 && (v < 0x800 || v > 0xfff) { return l64WordsLE( l64ir(lu12iw, int(int32(v)>>12), 30), l64rrr(de.rrr, 30, rj, rd), ), nil } return l64WordsLE( l64ir(lu12iw, int(int32(v)>>12), 30), l64irr(ori, int(v), 30, 30), l64rrr(de.rrr, 30, rj, rd), ), nil } words := l64DconMovWords(30, v) words = append(words, l64rrr(de.rrr, 30, rj, rd)) return l64WordsLE(words...), nil } // isLoong64ShiftD reports whether a shift-immediate opcode constant is one of // the 6-bit (.d) variants, the toolchain distinguishes them by the bit // position of the opcode field (bits [25:16]). func isLoong64ShiftD(op uint32) bool { return op&0x03ff0000 != 0 && op>>25 == 0 } // l64FmaOperands extracts the four fused-multiply-add operands: // INSTR fa, fk, fj, fd, or INSTR fa, fk, fd with fj = fd. func l64FmaOperands(ops []*ast.Operand) (fa, fk, fj, fd int, err error) { switch len(ops) { case 4: fa, fk, fj, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]), l64Reg(ops[3]) case 3: fa, fk, fd = l64Reg(ops[0]), l64Reg(ops[1]), l64Reg(ops[2]) fj = fd default: return 0, 0, 0, 0, fmt.Errorf("expected 3 or 4 operands, got %d", len(ops)) } if fa < 0 || fk < 0 || fj < 0 || fd < 0 { return 0, 0, 0, 0, fmt.Errorf("invalid register operand") } return fa, fk, fj, fd, nil } // l64MemOperands extracts (rd, rj, off, load) from a load/store instruction: // INSTR mem, rd is a load, INSTR rd, mem a store. func l64MemOperands(ops []*ast.Operand, fi loong64FrameInfo) (rd, rj int, off int32, load bool, err error) { if len(ops) != 2 { return 0, 0, 0, false, fmt.Errorf("expected 2 operands, got %d", len(ops)) } if isMemOperand(ops[0]) { rd = l64Reg(ops[1]) rj, off = l64MemWithFrame(ops[0], fi) load = true } else if isMemOperand(ops[1]) { rd = l64Reg(ops[0]) rj, off = l64MemWithFrame(ops[1], fi) } else { return 0, 0, 0, false, fmt.Errorf("expected a memory operand") } if rd < 0 || rj < 0 { return 0, 0, 0, false, fmt.Errorf("invalid operand") } return rd, rj, off, load, nil } // ---- the MOV pseudo-instruction ---- // encodeLOONG64Mov encodes the MOV family, the load/store/immediate // workhorse of Go's loong64 assembly. MOV is an alias of MOVV (the width // mnemonics MOVB/MOVH/MOVW/MOVV/MOVBU/MOVHU/MOVWU/MOVF/MOVD select the // access width). The forms, mirroring the toolchain: // // MOVx $imm, rd load immediate (addi/lu12i+ori/lu32i/lu52i) // MOVx mem, rd load from memory // MOVx rd, mem store to memory // MOVx rs, rd register move (incl. the FP-bank specials) // MOVx $sym(SB), rd address of a static symbol (pcalau12i+addi.d) // MOVx sym(SB), rd load from a static symbol (pcalau12i+ld) // MOVx rd, sym(SB) store to a static symbol (pcalau12i+st) func encodeLOONG64Mov(instr *ast.Instr, mnem string, fi loong64FrameInfo, relocs *[]Reloc) ([]byte, error) { ops := instr.Operands if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } if mnem == "MOV" { mnem = "MOVV" } src, dst := ops[0], ops[1] // Immediate → register. if isImmOperand(src) && !isMemOperand(src) { if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { rd := l64Reg(dst) if rd < 0 { return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem) } return encodeLOONG64SBAddr(src.Imm.Sym, rd, relocs), nil } rd := l64Reg(dst) if rd < 0 { return nil, fmt.Errorf("%s $imm: invalid destination register", mnem) } // MOVW $imm, Fd is the only immediate-to-F form the toolchain's optab // accepts (AMOVW's C_12CON against C_FREG): it materialises the // constant in R30 and moves it across with movgr2fr.w. MOVV/MOVF/ // MOVD are illegal combinations there, and are diagnosed here rather // than silently written into the GPR of the register's number. if loong64RegClass(operandRegName(dst)) == l64ClsFP { if mnem != "MOVW" { return nil, fmt.Errorf("%s $imm: illegal combination with an F register destination (only MOVW $c, Fd is supported)", mnem) } return encodeLOONG64ImmToFp(rd, l64Imm64(src)) } return encodeLOONG64LoadImm(rd, l64Imm64(src), mnem), nil } // Static symbol load/store via pcalau12i. if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) { rd := l64Reg(dst) if rd < 0 { return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem) } return encodeLOONG64SBLoad(src.Addr.Sym, rd, mnem, relocs), nil } if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) { rs := l64Reg(src) if rs < 0 { return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem) } return encodeLOONG64SBStore(dst.Addr.Sym, rs, mnem, relocs), nil } // Register-offset addressing: MOVx (rj)(rk), rd / MOVx rd, (rj)(rk). if src.Addr.Index != "" && !isMemOperand(dst) { rd := l64Reg(dst) rj, rk := loong64RegNum(src.Addr.Base), loong64RegNum(src.Addr.Index) if rd < 0 || rj < 0 || rk < 0 { return nil, fmt.Errorf("%s (rj)(rk): invalid register operand", mnem) } op, ok := l64IndexedTable[mnem] if !ok { return nil, fmt.Errorf("%s: no register-indexed form", mnem) } return l64wordLE(l64rrr(op.ld, rk, rj, rd)), nil } if dst.Addr.Index != "" && !isMemOperand(src) { rs := l64Reg(src) rj, rk := loong64RegNum(dst.Addr.Base), loong64RegNum(dst.Addr.Index) if rs < 0 || rj < 0 || rk < 0 { return nil, fmt.Errorf("%s rd, (rj)(rk): invalid register operand", mnem) } op, ok := l64IndexedTable[mnem] if !ok { return nil, fmt.Errorf("%s: no register-indexed form", mnem) } return l64wordLE(l64rrr(op.st, rk, rj, rs)), nil } // Memory load/store with a 12-bit (or larger, via expansion) offset. if isMemOperand(src) && !isMemOperand(dst) { rd := l64Reg(dst) if rd < 0 { return nil, fmt.Errorf("%s: invalid destination register", mnem) } return encodeLOONG64MemOp(mnem, ops[0], rd, true, fi) } if !isMemOperand(src) && isMemOperand(dst) { rs := l64Reg(src) if rs < 0 { return nil, fmt.Errorf("%s: invalid source register", mnem) } return encodeLOONG64MemOp(mnem, ops[1], rs, false, fi) } // Register → register. return encodeLOONG64RegMove(mnem, src, dst) } // loong64MovSize returns the encoded size of a MOV instruction. func loong64MovSize(mnem string, ops []*ast.Operand, fi loong64FrameInfo) int { if mnem == "MOV" { mnem = "MOVV" } if len(ops) != 2 { return 4 } src, dst := ops[0], ops[1] switch { case isImmOperand(src): if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" { return 8 // pcalau12i + addi.d } if loong64RegClass(operandRegName(dst)) == l64ClsFP { return 8 // ori/addi.w r30 + movgr2fr.w (an encode-time diagnostic when invalid) } v := l64Imm64(src) if v == 0 { return 4 } if v > 0 && v <= 0xfff { return 4 // ori rd, r0, v } if v >= -2048 && v < 0 { return 4 // addi.d rd, r0, v } if v == int64(int32(v)) { if v&0xfff == 0 { return 4 // lu12i.w } return 8 // lu12i.w + ori } return 4 * len(l64DconMovWords(0, v)) case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB": return 8 // pcalau12i + ld case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB": return 8 // pcalau12i + st case src.Addr.Index != "" || dst.Addr.Index != "": return 4 // ldx/stx case isMemOperand(src) || isMemOperand(dst): // A 12-bit offset fits in one instruction; larger offsets expand // to lu12i.w + add.d + the access. mem := src if !isMemOperand(src) { mem = dst } if l64MemOffset(mem, fi) >= -2048 && l64MemOffset(mem, fi) < 2048 { return 4 } return 12 default: return 4 // register move } } // encodeLOONG64ImmToFp materialises a 12-bit immediate in R30 and moves it // to an F register, the toolchain's expansion of MOVW $c, Fd: ori (which // zero-extends) for the positive span, addi.w for zero and the negative // span, then movgr2fr.w. The toolchain's optab accepts no wider constant on // this path (it never materialises one fully first), so values outside // [-2048, 4095] are diagnosed rather than masked into si12. func encodeLOONG64ImmToFp(fd int, v int64) ([]byte, error) { if v < -2048 || v > 4095 { return nil, fmt.Errorf("MOVW $%d: immediate out of the [-2048, 4095] range for an F register destination", v) } op := uint32(0x00a << 22) // addi.w r30, r0, v (sign-extends) if v > 0 { op = 0x00e << 22 // ori r30, r0, v (zero-extends) } return l64WordsLE( l64irr(op, int(v), 0, 30), l64rr(0x4529<<10, 30, fd), // movgr2fr.w fd, r30 ), nil } // ---- 64-bit immediate classification ---- // The dcon classes classify a 64-bit constant by which of the four // materialisation instructions (lu12i.w, ori, lu32i.d, lu52i.d) can be // dropped, mirroring the toolchain's dconClass: a field is ALL1/ALL0 when // it is all ones/zeros (fillable by sign/zero extension) or ST1/ST0 when it // starts with a 1/0 but is mixed. const ( l64All1 = iota l64All0 l64St1 l64St0 l64dcon120 l64dcon1220s l64dcon20s20 l64dcon1212s l64dcon20s12s l64dcon20s0 l64dcon1212u l64dcon20s12u l64dcon3212s l64dcon320 l64dcon3220 l64dcon1232s l64dcon20s32 l64dcon3212u l64Dcon ) // l64BitField classifies the bit field of v at [suf+len-1 : suf]. func l64BitField(v int64, suf, ln int8) int { var mask1, mask2 uint64 if ln == 12 { if suf == 0 { mask1, mask2 = 0xfff, 0x800 } else { mask1, mask2 = 0xfff0000000000000, 0x8000000000000000 } } else { if suf == 12 { mask1, mask2 = 0xfffff000, 0x80000000 } else { mask1, mask2 = 0xfffff00000000, 0x8000000000000 } } u := uint64(v) switch { case u&mask1 == mask1: return l64All1 case u&mask1 == 0: return l64All0 case u&mask2 == mask2: return l64St1 } return l64St0 } // l64DconClass returns the materialisation class of a 64-bit constant, // transcribed from cmd/internal/obj/loong64's dconClass. func l64DconClass(v int64) int { tzb := bits.TrailingZeros64(uint64(v)) hi12 := l64BitField(v, 52, 12) hi20 := l64BitField(v, 32, 20) lo20 := l64BitField(v, 12, 20) lo12 := l64BitField(v, 0, 12) if tzb >= 52 { return l64dcon120 } if tzb >= 32 { if ((hi20 == l64All1 || hi20 == l64St1) && hi12 == l64All1) || ((hi20 == l64All0 || hi20 == l64St0) && hi12 == l64All0) { return l64dcon20s0 } return l64dcon320 } if tzb >= 12 { if lo20 == l64St1 || lo20 == l64All1 { if hi20 == l64All1 { return l64dcon1220s } if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { return l64dcon20s20 } return l64dcon3220 } if hi20 == l64All0 { return l64dcon1220s } if (hi20 == l64St0 && hi12 == l64All0) || ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) { return l64dcon20s20 } return l64dcon3220 } if lo12 == l64St1 || lo12 == l64All1 { if lo20 == l64All1 { if hi20 == l64All1 { return l64dcon1212s } if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { return l64dcon20s12s } return l64dcon3212s } if lo20 == l64St1 { if hi20 == l64All1 { return l64dcon1232s } if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { return l64dcon20s32 } return l64Dcon } if lo20 == l64All0 { if hi20 == l64All0 { return l64dcon1212u } if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { return l64dcon20s12u } return l64dcon3212u } if hi20 == l64All0 { return l64dcon1232s } if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { return l64dcon20s32 } return l64Dcon } if lo20 == l64All0 { if hi20 == l64All0 { return l64dcon1212u } if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { return l64dcon20s12u } return l64dcon3212u } if lo20 == l64St1 || lo20 == l64All1 { if hi20 == l64All1 { return l64dcon1232s } if (hi20 == l64St1 && hi12 == l64All1) || ((hi20 == l64St0 || hi20 == l64All0) && hi12 == l64All0) { return l64dcon20s32 } return l64Dcon } if hi20 == l64All0 { return l64dcon1232s } if ((hi20 == l64St1 || hi20 == l64All1) && hi12 == l64All1) || (hi20 == l64St0 && hi12 == l64All0) { return l64dcon20s32 } return l64Dcon } // l64DconMovWords returns the materialisation words for a 64-bit constant // into rd, per the toolchain's case 67/68/69/59 sequences. func l64DconMovWords(rd int, v int64) []uint32 { const ( lu12iw = 0x0a << 25 lu32id = 0x0b << 25 lu52id = 0x00c << 22 addiw = 0x00a << 22 addid = 0x00b << 22 ori = 0x00e << 22 ) switch l64DconClass(v) { case l64dcon120: return []uint32{l64irr(lu52id, int(v>>52), 0, rd)} case l64dcon1220s: return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(lu52id, int(v>>52), rd, rd)} case l64dcon20s20: return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd)} case l64dcon1212s: return []uint32{l64irr(addid, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)} case l64dcon20s12s, l64dcon20s0: return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)} case l64dcon1212u: return []uint32{l64irr(ori, int(v), 0, rd), l64irr(lu52id, int(v>>52), rd, rd)} case l64dcon20s12u: return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd)} case l64dcon3212s, l64dcon320: return []uint32{l64irr(addiw, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} case l64dcon3220: return []uint32{l64ir(lu12iw, int(v>>12), rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} case l64dcon1232s: return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64irr(lu52id, int(v>>52), rd, rd)} case l64dcon20s32: return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd)} case l64dcon3212u: return []uint32{l64irr(ori, int(v), 0, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} default: return []uint32{l64ir(lu12iw, int(v>>12), rd), l64irr(ori, int(v), rd, rd), l64ir(lu32id, int(v>>32), rd), l64irr(lu52id, int(v>>52), rd, rd)} } } // encodeLOONG64LoadImm loads an immediate into a register, matching the // toolchain's MOVV/MOVW case 3/19/25/59 expansion: // // $0: or rd, r0, r0 (MOVW: sll.w rd, r0, r0) // 1..0xfff: ori rd, r0, imm // −2048..−1: addi.d rd, r0, imm // 32-bit (low 12 zero): lu12i.w rd, imm>>12 // 32-bit: lu12i.w rd, imm>>12; ori rd, rd, imm // 64-bit: lu12i.w rd, imm>>12; ori rd, rd, imm; // lu32i.d rd, imm>>32; lu52i.d rd, rd, imm>>52 func encodeLOONG64LoadImm(rd int, v int64, mnem string) []byte { if v == 0 { // The zero constant matches the register-form optab entry: MOVV → // or rd, r0, r0, MOVW → sll.w rd, r0, r0. op := l64movRegTable["MOVV"].op if mnem == "MOVW" { op = l64movRegTable["MOVW"].op } return l64wordLE(l64rrr(op, 0, 0, rd)) } if v > 0 && v <= 0xfff { return l64wordLE(l64irr(l64DualTable["OR"].imm, int(v), 0, rd)) } if v >= -2048 && v < 0 { // Both MOVV and MOVW use addi.d for negative constants. return l64wordLE(l64irr(l64DualTable["ADDV"].imm, int(v), 0, rd)) } if v == int64(int32(v)) { if v&0xfff == 0 { return l64wordLE(l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd)) } return l64WordsLE( l64ir(l64InstrTable["LU12IW"].op, int(int32(v)>>12), rd), l64irr(l64DualTable["OR"].imm, int(v), rd, rd), ) } // 64-bit constants use the shortest materialisation the bit pattern // admits (dcon classification). return l64WordsLE(l64DconMovWords(rd, v)...) } // encodeLOONG64MemOp encodes a memory load (load = true) or store with a // 12-bit offset, or the 3-instruction expansion for larger offsets: // lu12i.w r30, (off+0x800)>>12; add.d r30, rj, r30; ld/st rd, off(r30). func encodeLOONG64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi loong64FrameInfo) ([]byte, error) { rj, off := l64MemWithFrame(mem, fi) if rj < 0 { return nil, fmt.Errorf("invalid memory operand") } ls, ok := l64loadStoreTable[mnem] if !ok { return nil, fmt.Errorf("unsupported MOV width %q", mnem) } op := ls.st if load { op = ls.ld } if off >= -2048 && off < 2048 { return l64wordLE(l64irr(op, int(off), rj, reg)), nil } // Large offset: materialise the base in R30 (the assembler temp). return l64WordsLE( l64ir(l64InstrTable["LU12IW"].op, int((off+0x800)>>12), 30), l64rrr(l64DualTable["ADDV"].rrr, rj, 30, 30), l64irr(op, int(off), 30, reg), ), nil } // encodeLOONG64RegMove encodes a register-to-register move: the width // extensions (ext.w.b, ext.w.h, sll.w, or, andi, bstrpick.d) between GPRs, // fmov between F registers, and the special moves across the GPR/FP/FCC/FCSR // banks. func encodeLOONG64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) { rs, rd := l64Reg(src), l64Reg(dst) if rs < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand") } sc, dc := loong64RegClass(operandRegName(src)), loong64RegClass(operandRegName(dst)) // FP-bank specials (MOVV/MOVW between GPR/FCC/FCSR and F registers). if key, ok := l64FpMoveKey(mnem, sc, dc); ok { op, ok := l64FpMovTable[key] if !ok { return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst)) } return l64wordLE(l64rr(op, rs, rd)), nil } // GPR → GPR. if sc == l64ClsGR && dc == l64ClsGR { switch mnem { case "MOVHU": // bstrpick.d rd, rj, $15, $0 return l64wordLE(l64irir(0x3<<22, 15, rs, 0, rd)), nil case "MOVWU": // bstrpick.d rd, rj, $31, $0 return l64wordLE(l64irir(0x3<<22, 31, rs, 0, rd)), nil } if e, ok := l64movRegTable[mnem]; ok { if e.rr { return l64wordLE(l64rr(e.op, rs, rd)), nil } if e.imm != 0 { return l64wordLE(l64irr(e.op, e.imm, rs, rd)), nil } // 3R with rk = r0: or rd, rj, r0 / sll.w rd, rj, r0. return l64wordLE(l64rrr(e.op, 0, rs, rd)), nil } } // F → F. if sc == l64ClsFP && dc == l64ClsFP { if op, ok := l64movFpRegTable[mnem]; ok { return l64wordLE(l64rr(op, rs, rd)), nil } } return nil, fmt.Errorf("unsupported %s register move %s → %s", mnem, operandRegName(src), operandRegName(dst)) } // l64FpMoveKey builds the l64FpMovTable key for a cross-bank move, reporting // whether the move is a cross-bank special at all. func l64FpMoveKey(mnem string, sc, dc l64RegClass) (string, bool) { bank := func(c l64RegClass) string { switch c { case l64ClsFP: return "F" case l64ClsFCC: return "FCC" case l64ClsFCSR: return "FCSR" default: return "R" } } if sc == dc { return "", false } if mnem != "MOVV" && mnem != "MOVW" { return "", false } key := mnem + "." + bank(sc) + "." + bank(dc) _, ok := l64FpMovTable[key] return key, ok } // ---- static symbol references (pcalau12i + offset) ---- // encodeLOONG64SBAddr emits pcalau12i rd, 0; addi.d rd, rd, 0 with the // R_LOONG64_ADDR_HI/LO relocation pair, loading a symbol's address. func encodeLOONG64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte { if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset}, ) } return l64WordsLE( l64ir(l64InstrTable["PCALAU12I"].op, 0, rd), l64irr(l64DualTable["ADDV"].imm, 0, rd, rd), ) } // encodeLOONG64SBLoad emits pcalau12i r30, 0; ld rd, 0(r30) with the // R_LOONG64_ADDR_HI/LO pair, loading from a static symbol. func encodeLOONG64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) []byte { ls := l64loadStoreTable[mnem] if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset}, ) } return l64WordsLE( l64ir(l64InstrTable["PCALAU12I"].op, 0, 30), l64irr(ls.ld, 0, 30, rd), ) } // encodeLOONG64SBStore emits pcalau12i r30, 0; st rd, 0(r30) with the // R_LOONG64_ADDR_HI/LO pair, storing to a static symbol. func encodeLOONG64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) []byte { ls := l64loadStoreTable[mnem] if relocs != nil { *relocs = append(*relocs, Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelLoong64AddrHi, Addend: sym.Offset}, Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelLoong64AddrLo, Addend: sym.Offset}, ) } return l64WordsLE( l64ir(l64InstrTable["PCALAU12I"].op, 0, 30), l64irr(ls.st, 0, 30, rs), ) } // ---- operand helpers ---- // l64IndexedTable holds the register-indexed load/store (ldx/stx) opcodes. var l64IndexedTable = map[string]struct{ ld, st uint32 }{ "MOVB": {0x07000 << 15, 0x07020 << 15}, "MOVH": {0x07008 << 15, 0x07028 << 15}, "MOVW": {0x07010 << 15, 0x07030 << 15}, "MOVV": {0x07018 << 15, 0x07038 << 15}, "MOVBU": {0x07040 << 15, 0x07020 << 15}, "MOVHU": {0x07048 << 15, 0x07028 << 15}, "MOVWU": {0x07050 << 15, 0x07030 << 15}, "MOVF": {0x07060 << 15, 0x07070 << 15}, "MOVD": {0x07068 << 15, 0x07078 << 15}, } // operandRegName returns the register name of an operand, or "". func operandRegName(op *ast.Operand) string { if op.Addr.Base != "" { return op.Addr.Base } if op.Addr.Sym != nil && op.Addr.Sym.Name != "" { return op.Addr.Sym.Name } return "" } // l64Reg returns the register number of an operand, or -1. func l64Reg(op *ast.Operand) int { return loong64RegNum(operandRegName(op)) } // l64Imm64 returns the full 64-bit immediate value of an operand. func l64Imm64(op *ast.Operand) int64 { if op.Imm.HasVal { v := op.Imm.Val if op.Imm.Neg { v = -v } return v } return 0 } // l64Mem returns the base register and byte offset of a memory operand. func l64Mem(op *ast.Operand) (rj int, off int32) { rj = loong64RegNum(op.Addr.Base) off = int32(op.Addr.Offset) return } // l64MemWithFrame resolves a memory operand, translating FP/SP pseudo- // registers via the frame mapping. func l64MemWithFrame(op *ast.Operand, fi loong64FrameInfo) (rj int, off int32) { if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" { return loong64ResolvePseudo(op.Addr.Sym, fi) } return l64Mem(op) } // l64MemOffset returns the resolved byte offset of a memory operand. func l64MemOffset(op *ast.Operand, fi loong64FrameInfo) int32 { _, off := l64MemWithFrame(op, fi) return off } // l64Label returns the label name of an operand. func l64Label(op *ast.Operand) string { if op.Addr.Sym != nil { return op.Addr.Sym.Name } return op.Raw } // ---- LSX/LASX (V*/XV*) vector dispatch ---- // l64VecOperand describes a vector register operand: the 5-bit register // number, its bank and an optional width or element suffix (V0.B16, // V1.V[0], X3.WU[2]). The parser hands suffixed operands over verbatim // (the element index survives only in the raw text), so the suffix is // scanned from op.Raw. type l64VecOperand struct { num int // 5-bit register number lasx bool // X bank (LASX) rather than V (LSX) width byte // suffix width letter (B/H/W/V), 0 on a bare register lanes int // lane count of a width suffix (B16 → 16) elem int // element index of a .T[i] suffix hasEl bool // the suffix names an element (.T[i]) unsig bool // the suffix carries the U marker (.BU[0]) hasSuf bool // any suffix present } // l64ParseVecOperand parses a vector register operand with an optional // width or element suffix. ok reports whether the operand names a vector // register at all (V or X bank, with or without a suffix). func l64ParseVecOperand(op *ast.Operand) (v l64VecOperand, ok bool) { if op.Kind == ast.OpImmediate { return v, false } name := strings.ReplaceAll(op.Raw, " ", "") if name == "" || (name[0] != 'V' && name[0] != 'X') { return v, false } i := 1 num := 0 for i < len(name) && name[i] >= '0' && name[i] <= '9' { num = num*10 + int(name[i]-'0') if num > 31 { return v, false } i++ } if i == 1 { return v, false // no register digits } v.num, v.lasx = num, name[0] == 'X' if i == len(name) { return v, true } if name[i] != '.' || i+2 > len(name) { return v, false } i++ w := name[i] if w != 'B' && w != 'H' && w != 'W' && w != 'V' { return v, false } v.width, v.hasSuf = w, true i++ if i < len(name) && name[i] == 'U' { v.unsig = true i++ } if i < len(name) && name[i] == '[' { // Element form .T[i]: the closing bracket ends the operand. if name[len(name)-1] != ']' || i+2 > len(name)-1 { return v, false } idx := 0 for _, c := range name[i+1 : len(name)-1] { if c < '0' || c > '9' { return v, false } idx = idx*10 + int(c-'0') if idx > 31 { return v, false } } v.elem, v.hasEl = idx, true return v, true } // Width form .T: the trailing digits give the lane count. lanes := 0 if i >= len(name) { return v, false } for ; i < len(name); i++ { if name[i] < '0' || name[i] > '9' { return v, false } lanes = lanes*10 + int(name[i]-'0') if lanes > 64 { return v, false } } v.lanes = lanes return v, true } // l64VecSuffixWidth validates a width suffix against the bank (LSX: // B16/H8/W4/V2, LASX: B32/H16/W8/V4) and returns the encoded 2-bit width // selector of vreplgr2vr and vldrepl. func l64VecSuffixWidth(lasx bool, v l64VecOperand) (int, bool) { want := map[byte]int{'B': 16, 'H': 8, 'W': 4, 'V': 2} if lasx { want = map[byte]int{'B': 32, 'H': 16, 'W': 8, 'V': 4} } lanes, ok := want[v.width] if !ok || lanes != v.lanes { return 0, false } switch v.width { case 'B': return 0, true case 'H': return 1, true case 'W': return 2, true default: return 3, true } } // l64VecElementBase validates an element suffix against the bank and // returns the encoded index field: the index rides in the rk field above a // per-width base (vpickve2gr/vinsgr2vr give ui4 to .b, ui3 to .h, ui2 to .w // and ui1 to .d). The LASX bank has no .b/.h element forms: the toolchain // rejects `XVMOVQ R4, X2.B[0]` and `XVMOVQ X3.B[31], R5`. func l64VecElementBase(lasx bool, v l64VecOperand) (int, bool) { limit, base := 0, 0 switch v.width { case 'B': if lasx { return 0, false } limit, base = 15, 0 case 'H': if lasx { return 0, false } limit, base = 7, 16 case 'W': limit, base = 3, 24 if lasx { limit, base = 7, 16 } case 'V': limit, base = 1, 28 if lasx { limit, base = 3, 24 } default: return 0, false } if v.elem > limit { return 0, false } return base + v.elem, true } // encodeLOONG64Vector encodes the LSX/LASX mnemonics the table marks as // vector plus the VMOVQ/XVMOVQ move family. handled reports whether the // mnemonic belongs to the vector slice; the operand shapes and opcode // constants reproduce GOARCH=loong64 `go tool asm` exactly. func encodeLOONG64Vector(instr *ast.Instr, mnem string, fi loong64FrameInfo) ([]byte, bool, error) { if mnem == "VMOVQ" || mnem == "XVMOVQ" { code, err := encodeLOONG64Vmovq(mnem == "XVMOVQ", instr.Operands, fi) return code, true, err } lasx, ok := l64VecBank[mnem] if !ok { return nil, false, nil } ops := instr.Operands bank := "V" if lasx { bank = "X" } vec := func(op *ast.Operand) (int, error) { v, isVec := l64ParseVecOperand(op) if !isVec || v.lasx != lasx || v.hasSuf { return -1, fmt.Errorf("%s: expected a bare %s0-%s31 vector register, got %q", mnem, bank, bank, op.Raw) } return v.num, nil } // Two-operand forms (vpcnt.v): INSTR vj, vd. if l64Vec2R[mnem] { if len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } vj, err := vec(ops[0]) if err != nil { return nil, true, err } vd, err := vec(ops[1]) if err != nil { return nil, true, err } return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, vd)), true, nil } // Immediate forms: INSTR $imm, vd or INSTR $imm, vj, vd. if e, imm := l64VecImmInfo[mnem]; imm && len(ops) >= 2 && isImmOperand(ops[0]) { if len(ops) > 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } imm := int(immFromOperand(ops[0])) if imm < e.min || imm > e.max { return nil, true, fmt.Errorf("%s: immediate out of range [%d, %d]", mnem, e.min, e.max) } vd, err := vec(ops[len(ops)-1]) if err != nil { return nil, true, err } vj := vd if len(ops) == 3 { if vj, err = vec(ops[1]); err != nil { return nil, true, err } } return l64wordLE(l64irr(e.op, (imm+e.bias)&e.mask, vj, vd)), true, nil } // Vector-to-condition forms: INSTR vj, FCCn. if l64InstrTable[mnem].format == l64Fvcf { if len(ops) != 2 { return nil, true, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } vj, err := vec(ops[0]) if err != nil { return nil, true, err } if loong64RegClass(operandRegName(ops[1])) != l64ClsFCC { return nil, true, fmt.Errorf("%s: expected an FCC condition flag, got %q", mnem, ops[1].Raw) } fcc := loong64RegNum(operandRegName(ops[1])) return l64wordLE(l64rr(l64InstrTable[mnem].op, vj, fcc)), true, nil } // Four-register forms (vshuf.b): INSTR va, vk, vj, vd. if l64Vec4R[mnem] { if len(ops) != 4 { return nil, true, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops)) } va, err := vec(ops[0]) if err != nil { return nil, true, err } vk, err := vec(ops[1]) if err != nil { return nil, true, err } vj, err := vec(ops[2]) if err != nil { return nil, true, err } vd, err := vec(ops[3]) if err != nil { return nil, true, err } return l64wordLE(l64InstrTable[mnem].op | uint32(va&0x1f)<<15 | uint32(vk&0x1f)<<10 | uint32(vj&0x1f)<<5 | uint32(vd&0x1f)), true, nil } // Three-register forms: INSTR vk, vj, vd or INSTR vk, vd (vj = vd). if len(ops) != 2 && len(ops) != 3 { return nil, true, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } vk, err := vec(ops[0]) if err != nil { return nil, true, err } vd, err := vec(ops[len(ops)-1]) if err != nil { return nil, true, err } vj := vd if len(ops) == 3 { if vj, err = vec(ops[1]); err != nil { return nil, true, err } } return l64wordLE(l64rrr(l64InstrTable[mnem].op, vk, vj, vd)), true, nil } // encodeLOONG64Vmovq encodes the VMOVQ/XVMOVQ move family. One mnemonic // covers the whole LSX/LASX transfer surface, dispatched by operand shape // exactly as the toolchain's table does: // // VMOVQ vd, off(rj) vst VMOVQ off(rj), vd vld // VMOVQ vd, (rj)(rk) vstx VMOVQ (rj)(rk), vd vldx // VMOVQ off(rj), vd.T vldrepl (load and replicate one element) // VMOVQ vj, vd vori.b $0 (a register move) // VMOVQ rj, vd.T vreplgr2vr (duplicate a general register) // VMOVQ vj.T[i], rd vpickve2gr (extract one element) // VMOVQ rj, vd.T[i] vinsgr2vr (insert one element) func encodeLOONG64Vmovq(lasx bool, ops []*ast.Operand, fi loong64FrameInfo) ([]byte, error) { enc := l64VmovqTable[lasx] bank := "V" if lasx { bank = "X" } if len(ops) != 2 { return nil, fmt.Errorf("VMOVQ expects 2 operands, got %d", len(ops)) } src, srcVec := l64ParseVecOperand(ops[0]) dst, dstVec := l64ParseVecOperand(ops[1]) srcMem := isMemOperand(ops[0]) dstMem := isMemOperand(ops[1]) srcIdx := srcMem && ops[0].Addr.Index != "" dstIdx := dstMem && ops[1].Addr.Index != "" intReg := func(op *ast.Operand) (int, error) { if isMemOperand(op) { return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw) } name := operandRegName(op) if loong64RegClass(name) != l64ClsGR { return -1, fmt.Errorf("VMOVQ: expected a general register, got %q", op.Raw) } return loong64RegNum(name), nil } // Register move: VMOVQ vj, vd (vori.b/xvori.b with the zero constant), // both operands bare registers of the same bank. if srcVec && dstVec { if src.hasSuf || dst.hasSuf { return nil, fmt.Errorf("VMOVQ: a register move takes bare %s registers", bank) } if src.lasx != lasx || dst.lasx != lasx { return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) } return l64wordLE(l64rr(enc.move, src.num, dst.num)), nil } // Store: VMOVQ vd, off(rj) or VMOVQ vd, (rj)(rk). if srcVec && dstMem { if src.hasSuf || src.lasx != lasx { return nil, fmt.Errorf("VMOVQ: expected a bare %s0-%s31 register as the stored value", bank, bank) } if dstIdx { rj, rk := loong64RegNum(ops[1].Addr.Base), loong64RegNum(ops[1].Addr.Index) if rj < 0 || rk < 0 { return nil, fmt.Errorf("VMOVQ: invalid register operand") } return l64wordLE(l64rrr(enc.stx, rk, rj, src.num)), nil } rj, off := l64MemWithFrame(ops[1], fi) if rj < 0 || off < -2048 || off > 2047 { return nil, fmt.Errorf("VMOVQ: store offset out of range [-2048, 2047]") } return l64wordLE(l64irr(enc.st, int(off), rj, src.num)), nil } // Load: VMOVQ off(rj), vd, the indexed VMOVQ (rj)(rk), vd, and the // load-and-replicate form VMOVQ off(rj), vd.T. if srcMem && dstVec { if dst.lasx != lasx { return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) } if srcIdx { if dst.hasSuf { return nil, fmt.Errorf("VMOVQ: an indexed load takes a bare %s register", bank) } rj, rk := loong64RegNum(ops[0].Addr.Base), loong64RegNum(ops[0].Addr.Index) if rj < 0 || rk < 0 { return nil, fmt.Errorf("VMOVQ: invalid register operand") } return l64wordLE(l64rrr(enc.ldx, rk, rj, dst.num)), nil } rj, off := l64MemWithFrame(ops[0], fi) if rj < 0 || off < -2048 || off > 2047 { return nil, fmt.Errorf("VMOVQ: load offset out of range [-2048, 2047]") } op := enc.ld if dst.hasSuf { w, ok := l64VecSuffixWidth(lasx, dst) if !ok { return nil, fmt.Errorf("VMOVQ: invalid replicate width suffix %q", ops[1].Raw) } switch w { case 0: op = enc.replB case 1: op = enc.replH case 2: op = enc.replW default: op = enc.replD } } return l64wordLE(l64irr(op, int(off), rj, dst.num)), nil } // Element extract: VMOVQ vj.T[i], rd (vpickve2gr, signed or unsigned). if srcVec && src.hasEl && !dstVec && !dstMem { if src.lasx != lasx { return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) } idx, ok := l64VecElementBase(lasx, src) if !ok { return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[0].Raw) } rd, err := intReg(ops[1]) if err != nil { return nil, err } op := enc.pickS if src.unsig { op = enc.pickU } return l64wordLE(l64irr(op, idx, src.num, rd)), nil } // Insert and duplicate: VMOVQ rj, vd.T[i] (vinsgr2vr) and // VMOVQ rj, vd.T (vreplgr2vr). if !srcVec && !srcMem && dstVec && dst.hasSuf { if dst.lasx != lasx { return nil, fmt.Errorf("VMOVQ: expected %s-bank vector registers", bank) } rs, err := intReg(ops[0]) if err != nil { return nil, err } if dst.hasEl { idx, ok := l64VecElementBase(lasx, dst) if !ok { return nil, fmt.Errorf("VMOVQ: invalid element suffix %q", ops[1].Raw) } return l64wordLE(l64irr(enc.ins, idx, rs, dst.num)), nil } w, ok := l64VecSuffixWidth(lasx, dst) if !ok { return nil, fmt.Errorf("VMOVQ: invalid width suffix %q", ops[1].Raw) } return l64wordLE(l64irr(enc.dup, w, rs, dst.num)), nil } return nil, fmt.Errorf("VMOVQ: unsupported operand combination %q, %q", ops[0].Raw, ops[1].Raw) }