// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm // loong64 (LoongArch) instruction encoding. // // The encoder is data-driven: each mnemonic maps to an instruction format and // an opcode constant, and the format selects the bit layout. The opcode // constants and formats are transcribed from the Go toolchain's own loong64 // backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm` // exactly, the ground-truth oracle for the verify suite. // // All LoongArch instructions are 32 bits, little-endian. The formats used // here (per the LoongArch Volume I specification): // // 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0] // 2R opcode[31:15] | rj[4:0] | rd[4:0] // 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0] // 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0] // 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0] // 2RI20 opcode[31:25] | si20[19:0] | rd[4:0] // 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z) // B/BL opcode[31:26] | offs[25:0] // 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0] // IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0] // 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0] // // The opcode constants are pre-positioned (they include the zero bit ranges // of the immediate and register fields), mirroring the toolchain's OP_* // helpers, so each l64* function only ORs its fields in. import ( "maps" "strings" ) // loong64RegNum returns the 5-bit register number for a LoongArch register // name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition // flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's // assembly uses. Returns -1 for an unrecognised name. func loong64RegNum(name string) int { switch name { case "R0", "ZERO": return 0 case "R1", "RA", "LINK": return 1 case "R2", "TP": return 2 case "R3", "SP": return 3 case "R4", "A0": return 4 case "R5", "A1": return 5 case "R6", "A2": return 6 case "R7", "A3": return 7 case "R8", "A4": return 8 case "R9", "A5": return 9 case "R10", "A6": return 10 case "R11", "A7": return 11 case "R12", "T0": return 12 case "R13", "T1": return 13 case "R14", "T2": return 14 case "R15", "T3": return 15 case "R16", "T4": return 16 case "R17", "T5": return 17 case "R18", "T6": return 18 case "R19", "T7": return 19 case "R20", "T8": return 20 case "R21": return 21 case "R22", "G", "g", "FP": return 22 case "R23", "S0": return 23 case "R24", "S1": return 24 case "R25", "S2": return 25 case "R26", "S3": return 26 case "R27", "S4": return 27 case "R28", "S5": return 28 case "R29", "S6", "CTXT": return 29 case "R30", "S7", "TMP": return 30 case "R31", "S8": return 31 } // F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31, // X0-X31) are deliberately NOT accepted here: they are a separate // register class, and the toolchain rejects V/X names wherever an // integer or FP register is expected (GOARCH=loong64 go tool asm reports // "unrecognized instruction" for `BEQZ X0`). Vector operands are // resolved only through loong64VecRegNum. if len(name) >= 4 && name[:4] == "FCSR" { return loong64RegSpecial(name[4:], 31) } if len(name) >= 3 && name[:3] == "FCC" { return loong64RegSpecial(name[3:], 7) } if len(name) < 2 { return -1 } prefix, digits := name[:1], name[1:] if digits[0] < '0' || digits[0] > '9' { return -1 } n := 0 for i := 0; i < len(digits); i++ { if digits[i] < '0' || digits[i] > '9' { return -1 } n = n*10 + int(digits[i]-'0') } if prefix == "F" && n <= 31 { return n } return -1 } // loong64RegSpecial parses a numbered FCC/FCSR register. func loong64RegSpecial(digits string, max int) int { if digits == "" { return -1 } n := 0 for i := 0; i < len(digits); i++ { if digits[i] < '0' || digits[i] > '9' { return -1 } n = n*10 + int(digits[i]-'0') } if n <= max { return n } return -1 } // loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or // X0-X31) to its 5-bit number, or -1. The vector banks are a register class // of their own: the toolchain accepts them only in the vector operands of the // LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1, // V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X // spellings never reach the integer/FP resolver. func loong64VecRegNum(name string) int { if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') { return -1 } return loong64RegSpecial(name[1:], 31) } // ---- format helpers ---- // l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd. func l64rrr(op uint32, rk, rj, rd int) uint32 { return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64rr encodes a 2R instruction: op | rj<<5 | rd. func l64rr(op uint32, rj, rd int) uint32 { return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd. func l64irr(op uint32, imm, rj, rd int) uint32 { return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd. func l64irr14(op uint32, imm, rj, rd int) uint32 { return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd. func l64irr16(op uint32, imm, rj, rd int) uint32 { return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64ir encodes a 2RI20 instruction: op | si20<<5 | rd. func l64ir(op uint32, imm, rd int) uint32 { return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f) } // l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the // 4-byte-aligned word distance (the toolchain stores the shifted value). func l64bbl(op uint32, offs int) uint32 { return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF } // l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF): // op | si21[15:0]<<10 | rj<<5 | si21[20:16]. func l64ir21(op uint32, offs, rj int) uint32 { v := uint32(offs) return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F } // l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4. func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 { return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f) } // l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd. // The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller // must have validated them (0..31 for the .w forms, 0..63 for the .d forms, // lsb <= msb), the same rule the toolchain enforces as "illegal bit number". func l64irir(op uint32, msb, rj, lsb, rd int) uint32 { return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f) } // l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd. func l64irrr(op uint32, sa, rk, rj, rd int) uint32 { return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64i15 encodes a no-operand system instruction with a 15-bit code field // (SYSCALL, BREAK, DBAR): op | code[14:0]. func l64i15(op uint32, code int) uint32 { return op | uint32(code)&0x7FFF } // l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint. func l64irr5i(op uint32, offs, rj, hint int) uint32 { return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f) } // l64wordLE encodes a uint32 as 4 little-endian bytes. func l64wordLE(w uint32) []byte { return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} } // l64WordsLE concatenates one or more instruction words as little-endian bytes. func l64WordsLE(ws ...uint32) []byte { var out []byte for _, w := range ws { out = append(out, l64wordLE(w)...) } return out } // ---- instruction formats ---- type l64Format uint8 const ( l64Frrr l64Format = iota // 3R (integer and FP arithmetic) l64Frr // 2R l64Firr // 2RI12 (arithmetic with 12-bit immediate) l64Firr14 // 2RI14 (ldptr/stptr) l64Firr16 // 2RI16 (addu16i.d) l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i) l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel) l64Firir // bstrins/bstrpick l64Firrr // alsl l64Fi15 // syscall/break/dbar l64Fam // atomic (3R with the AM field order) l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0]) l64Fshift // 2RI12 with a 5/6-bit shift immediate l64Fpreld // preld (2RI12 + 5-bit hint) l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc ) // l64Enc is one instruction's encoding: its bit layout (format) and the // opcode constant, positioned at its exact bit range. type l64Enc struct { format l64Format op uint32 } // l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form // and the 2RI12 immediate form (which is a shift for the shift mnemonics). type l64DualEnc struct { rrr uint32 // 3R register form imm uint32 // 2RI12 immediate form shift bool // the immediate form is a 5/6-bit shift amount } // l64DualTable maps the dual-form arithmetic/logic mnemonics to both // encodings; the assembler picks by operand kind. var l64DualTable = map[string]l64DualEnc{} // l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them) // to their encoding. var l64InstrTable = map[string]l64Enc{} // l64Vec3Enc pairs a vector opcode with its register bank: false = LSX // (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per // spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and // `XVADDV X1, X2, X3`, and rejects the crossed spellings. type l64Vec3Enc struct { op uint32 lasx bool } // l64VecImmEnc carries the immediate-form encoding of a vector mnemonic: // the opcode, the bank, the accepted immediate range, the bias the toolchain // adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b // keeps a 5-bit two's-complement value, vseqi.d a 7-bit one). type l64VecImmEnc struct { op uint32 lasx bool min, max int bias int mask int } // l64VecBank marks the LSX/LASX mnemonics and records which register bank // each accepts; presence in the map routes the mnemonic through the vector // dispatcher rather than the integer/FP formats. var l64VecBank = map[string]bool{} // l64VecImmInfo mirrors l64VecImmTable for the dispatcher. var l64VecImmInfo = map[string]l64VecImmEnc{} // l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as // vpcnt.v). var l64Vec2R = map[string]bool{} // l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit // 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels. type l64VmovqEnc struct { ld, st, ldx, stx uint32 // plain and indexed load/store replB, replH, replW, replD uint32 // vldrepl: load and replicate element pickS, pickU uint32 // vpickve2gr.{,u} element extract ins uint32 // vinsgr2vr element insert dup uint32 // vreplgr2vr duplicate (width in [11:10]) move uint32 // vori.b/xvori.b $0 register move } var l64VmovqTable = map[bool]l64VmovqEnc{ false: { // VMOVQ, the LSX (V) bank ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15, replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15, pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15, ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15, }, true: { // XVMOVQ, the LASX (X) bank ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15, replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15, pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15, ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15, }, } func init() { // 3R, integer. rrr := map[string]uint32{ "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SGT": 0x24 << 15, "SGTU": 0x25 << 15, "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15, "NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15, "ORN": 0x2c << 15, "ANDN": 0x2d << 15, "SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15, "SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15, "ROTR": 0x36 << 15, "ROTRV": 0x37 << 15, "MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15, "MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15, "MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15, "DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15, "DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15, "DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15, "CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15, "CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15, } // 3R, floating point. rrr["MULF"] = 0x209 << 15 rrr["MULD"] = 0x20a << 15 rrr["DIVF"] = 0x20d << 15 rrr["DIVD"] = 0x20e << 15 rrr["SUBF"] = 0x205 << 15 rrr["SUBD"] = 0x206 << 15 rrr["ADDF"] = 0x201 << 15 rrr["ADDD"] = 0x202 << 15 rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15 rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15 rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15 rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15 rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15 rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15 rrr["FMINF"] = 0x215 << 15 rrr["FMIND"] = 0x216 << 15 rrr["FMAXF"] = 0x211 << 15 rrr["FMAXD"] = 0x212 << 15 rrr["FMAXAF"] = 0x219 << 15 rrr["FMAXAD"] = 0x21a << 15 rrr["FMINAF"] = 0x21d << 15 rrr["FMINAD"] = 0x21e << 15 rrr["FSCALEBF"] = 0x221 << 15 rrr["FSCALEBD"] = 0x222 << 15 rrr["FCOPYSGF"] = 0x225 << 15 rrr["FCOPYSGD"] = 0x226 << 15 for m, op := range rrr { l64InstrTable[m] = l64Enc{format: l64Frrr, op: op} } // 2R. rr := map[string]uint32{ "CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10, "CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10, "REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10, "REVH2W": 0x10 << 10, "REVHV": 0x11 << 10, "BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10, "EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10, "TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10, "MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10, "MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10, "FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10, "MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10, "ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10, "MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10, "NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10, "SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10, "FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10, "FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10, "FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10, "FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10, "FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10, "FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10, "FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10, "FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10, "FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10, "FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10, // LSX: convert a 64-bit integer lane to a double float. The operand // bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`), // so the entry stays on the 2R integer/FP format. "FFINTDV": 0x474a << 10, } for m, op := range rr { l64InstrTable[m] = l64Enc{format: l64Frr, op: op} } // RDTIME is a 2R instruction with rd and rj in swapped positions. l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10} l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10} l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10} // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), // selected by the operand kind; the shift mnemonics pair the 3R form // with a 5/6-bit shift immediate. maps.Copy(l64DualTable, map[string]l64DualEnc{ "ADD": {rrr: 0x20 << 15, imm: 0x00a << 22}, "ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22}, "ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22}, "ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22}, "AND": {rrr: 0x29 << 15, imm: 0x00d << 22}, "OR": {rrr: 0x2a << 15, imm: 0x00e << 22}, "XOR": {rrr: 0x2b << 15, imm: 0x00f << 22}, "SGT": {rrr: 0x24 << 15, imm: 0x008 << 22}, "SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22}, "SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true}, "SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true}, "SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true}, "ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true}, "SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true}, "SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true}, "SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true}, "ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true}, }) // 2RI12, pure immediate arithmetic (LU52ID has no register form). l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22} // ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16. l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26} // 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and // stores (ldptr/stptr), with the offset scaled by 4. l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w) l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w) l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d) // 2RI20. l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25} l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25} l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25} l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25} // LUI is the Plan 9 spelling of lu12i.w. l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} // 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC // condition flag, the layout matches the 4R shape). rrrr := map[string]uint32{ "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, "FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20, "FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20, "FSEL": 0x340 << 18, } for m, op := range rrrr { l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} } // IRIR, bit-field insert/extract. irir := map[string]uint32{ "BSTRINSW": 0x3<<21 | 0x0<<15, "BSTRINSV": 0x2 << 22, "BSTRPICKW": 0x3<<21 | 0x1<<15, "BSTRPICKV": 0x3 << 22, } for m, op := range irir { l64InstrTable[m] = l64Enc{format: l64Firir, op: op} } // 3RI2, ALSL. irrr := map[string]uint32{ "ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17, } for m, op := range irrr { l64InstrTable[m] = l64Enc{format: l64Firrr, op: op} } // 0-operand system instructions. l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15} l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15} l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15} // PRELD. l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} // Atomics, 3R with the AM field order (rk=value, rj=address, rd=result). // The toolchain's form is three operands, `AMADDW rk, (rj), rd` // (cmd/asm/internal/asm/testdata/loong64enc1.s and // internal/runtime/atomic/atomic_loong64.s); the two-register spelling // is rejected by the oracle. am := map[string]uint32{ "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, "AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15, "AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15, "AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15, "AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15, "AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15, "AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15, "AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15, "AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15, "AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15, "AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15, "AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15, "AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15, "AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15, "AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15, // The _dbar (acquire/release) add, and, or variants: opcodes read off // `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends. "AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15, "AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15, "AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15, } for m, op := range am { l64InstrTable[m] = l64Enc{format: l64Fam, op: op} } // ---- LSX/LASX (V*/XV*) ---- // Every opcode below was read off `go tool objdump` of a GOARCH=loong64 // `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks // most of them), not assumed from the LoongArch manual. // Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with // vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain // accepts: LSX spellings take V0-V31, LASX spellings X0-X31. vec3 := map[string]l64Vec3Enc{ "VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false}, "VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false}, "VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false}, "VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false}, "XVADDV": {0xE817 << 15, true}, "XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true}, "XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true}, } for m, e := range vec3 { l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op} l64VecBank[m] = e.lasx } // Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate // range, bias and field mask are the ones the toolchain encodes: vandi.b // stores the raw 8-bit constant, vsrai.b stores imm+8 (byte-lane bias), // vseqi.b and vseqi.d store 5-bit and 7-bit two's-complement values. // The mnemonics that also have a register form (VSEQB, VSEQV, VSRAB, // VROTRW) keep their three-register entry in l64InstrTable; the // dispatcher picks the immediate opcode from l64VecImmInfo by operand // kind, so the immediate entries must not overwrite the table. vecImm := map[string]l64VecImmEnc{ "VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF}, "XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF}, "VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F}, "XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F}, "VSEQV": {0xE503 << 15, false, -64, 63, 0, 0x7F}, "XVSEQV": {0xE903 << 15, true, -64, 63, 0, 0x7F}, "VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F}, "VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F}, } for m, e := range vecImm { l64VecImmInfo[m] = e l64VecBank[m] = e.lasx } // Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*, // vsetallnez.*): the sub-op rides in the rk field. vecCf := map[string]uint32{ "VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10, "VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10, "VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10, "VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10, } for m, op := range vecCf { l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op} l64VecBank[m] = strings.HasPrefix(m, "XV") } // Lane popcount: INSTR vj, vd (the 2R layout with the opcode extending // over the unused vk field). vec2r := map[string]l64Vec3Enc{ "VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true}, } for m, e := range vec2r { l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op} l64VecBank[m] = e.lasx l64Vec2R[m] = true } } // l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the // register move between the integer and floating-point register banks, the // MOVW/MOVV specials the Go assembler accepts. var l64FpMovTable = map[string]uint32{ "MOVV.R.F": 0x452a << 10, // movgr2fr.d "MOVV.R.FCC": 0x4536 << 10, // movgr2cf "MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr "MOVV.F.R": 0x452e << 10, // movfr2gr.d "MOVV.F.FCC": 0x4534 << 10, // movfr2cf "MOVV.FCC.R": 0x4537 << 10, // movcf2gr "MOVV.FCC.F": 0x4535 << 10, // movcf2fr "MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr "MOVW.R.F": 0x4529 << 10, // movgr2fr.w "MOVW.F.R": 0x452d << 10, // movfr2gr.s } // l64branchTable holds the 16-bit branch and jump encodings (2RI16). var l64branchTable = map[string]uint32{ "BEQ": 0x16 << 26, "BNE": 0x17 << 26, "BLT": 0x18 << 26, "BGE": 0x19 << 26, "BLTU": 0x1a << 26, "BGEU": 0x1b << 26, "JIRL": 0x13 << 26, } // l64branch21Table holds the single-register branches with 21-bit offsets: // the negative opcode constants the toolchain uses for the short forms. var l64branch21Table = map[string]uint32{ "BEQZ": 0x10 << 26, // beq r0, rj → beqz "BNEZ": 0x11 << 26, // bne r0, rj → bnez "BLTZ": 0x18 << 26, // blt rj, r0 → bltz "BGEZ": 0x19 << 26, // bge rj, r0 → bgez "BGTZ": 0x18 << 26, // blt r0, rj → bgtz "BLEZ": 0x19 << 26, // bge r0, rj → blez "BFPT": 0x12<<26 | 0x1<<8, "BFPF": 0x12<<26 | 0x0<<8, } // l64jumpTable maps the jump pseudo-instructions and their aliases to the // B/BL opcode constants. var l64jumpTable = map[string]uint32{ "JMP": 0x14 << 26, // b "B": 0x14 << 26, // b "JAL": 0x15 << 26, // bl "CALL": 0x15 << 26, // bl "BL": 0x15 << 26, // bl } // l64loadStoreTable maps the MOV width mnemonics to their load and store // 2RI12 opcodes. The load opcode is the negated store opcode, exactly as // the toolchain derives it. var l64loadStoreTable = map[string]struct{ ld, st uint32 }{ "MOVB": {0x0a0 << 22, 0x0a4 << 22}, "MOVH": {0x0a1 << 22, 0x0a5 << 22}, "MOVW": {0x0a2 << 22, 0x0a6 << 22}, "MOVV": {0x0a3 << 22, 0x0a7 << 22}, "MOVBU": {0x0a8 << 22, 0x0a4 << 22}, "MOVHU": {0x0a9 << 22, 0x0a5 << 22}, "MOVWU": {0x0aa << 22, 0x0a6 << 22}, "MOVF": {0x0ac << 22, 0x0ad << 22}, "MOVD": {0x0ae << 22, 0x0af << 22}, } // l64movRegTable maps a register-to-register MOV mnemonic to its expansion, // matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h, // MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d // and are handled separately in the assembler. type l64MovRegEnc struct { rr bool // 2R format (ext.w.b/ext.w.h) op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode imm int // 2RI12 immediate for MOVBU's andi } var l64movRegTable = map[string]l64MovRegEnc{ "MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj "MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj "MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0 "MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0 "MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff } // l64movFpRegTable maps a floating-point register move mnemonic to its 2R // opcode (fmov.s / fmov.d), used when both operands are F registers. var l64movFpRegTable = map[string]uint32{ "MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10, } // l64RegClass discriminates integer (R), floating-point (F) and condition // (FCC) registers for the MOV pseudo-instruction's register-move encoding. type l64RegClass int const ( l64ClsNone l64RegClass = iota l64ClsGR l64ClsFP l64ClsFCC l64ClsFCSR ) // loong64RegClass reports the register class of a register operand name. func loong64RegClass(name string) l64RegClass { switch { case name == "": return l64ClsNone case len(name) >= 3 && name[:3] == "FCC": return l64ClsFCC case len(name) >= 4 && name[:4] == "FCSR": return l64ClsFCSR case name[0] == 'F': return l64ClsFP default: return l64ClsGR } }