// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm // loong64 (LoongArch) instruction encoding. // // The encoder is data-driven: each mnemonic maps to an instruction format and // an opcode constant, and the format selects the bit layout. The opcode // constants and formats are transcribed from the Go toolchain's own loong64 // backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm` // exactly, the ground-truth oracle for the verify suite. // // All LoongArch instructions are 32 bits, little-endian. The formats used // here (per the LoongArch Volume I specification): // // 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0] // 2R opcode[31:15] | rj[4:0] | rd[4:0] // 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0] // 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0] // 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0] // 2RI20 opcode[31:25] | si20[19:0] | rd[4:0] // 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z) // B/BL opcode[31:26] | offs[25:0] // 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0] // IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0] // 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0] // // The opcode constants are pre-positioned (they include the zero bit ranges // of the immediate and register fields), mirroring the toolchain's OP_* // helpers, so each l64* function only ORs its fields in. import ( "maps" "strings" ) // loong64RegNum returns the 5-bit register number for a LoongArch register // name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition // flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's // assembly uses. Returns -1 for an unrecognised name. func loong64RegNum(name string) int { switch name { case "R0", "ZERO": return 0 case "R1", "RA", "LINK": return 1 case "R2", "TP": return 2 case "R3", "SP": return 3 case "R4", "A0": return 4 case "R5", "A1": return 5 case "R6", "A2": return 6 case "R7", "A3": return 7 case "R8", "A4": return 8 case "R9", "A5": return 9 case "R10", "A6": return 10 case "R11", "A7": return 11 case "R12", "T0": return 12 case "R13", "T1": return 13 case "R14", "T2": return 14 case "R15", "T3": return 15 case "R16", "T4": return 16 case "R17", "T5": return 17 case "R18", "T6": return 18 case "R19", "T7": return 19 case "R20", "T8": return 20 case "R21": return 21 case "R22", "G", "g", "FP": return 22 case "R23", "S0": return 23 case "R24", "S1": return 24 case "R25", "S2": return 25 case "R26", "S3": return 26 case "R27", "S4": return 27 case "R28", "S5": return 28 case "R29", "S6", "CTXT": return 29 case "R30", "S7", "TMP": return 30 case "R31", "S8": return 31 } // F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31, // X0-X31) are deliberately NOT accepted here: they are a separate // register class, and the toolchain rejects V/X names wherever an // integer or FP register is expected (GOARCH=loong64 go tool asm reports // "unrecognized instruction" for `BEQZ X0`). Vector operands are // resolved only through loong64VecRegNum. if len(name) >= 4 && name[:4] == "FCSR" { return loong64RegSpecial(name[4:], 31) } if len(name) >= 3 && name[:3] == "FCC" { return loong64RegSpecial(name[3:], 7) } if len(name) < 2 { return -1 } prefix, digits := name[:1], name[1:] if digits[0] < '0' || digits[0] > '9' { return -1 } n := 0 for i := 0; i < len(digits); i++ { if digits[i] < '0' || digits[i] > '9' { return -1 } n = n*10 + int(digits[i]-'0') } if prefix == "F" && n <= 31 { return n } return -1 } // loong64RegSpecial parses a numbered FCC/FCSR register. func loong64RegSpecial(digits string, max int) int { if digits == "" { return -1 } n := 0 for i := 0; i < len(digits); i++ { if digits[i] < '0' || digits[i] > '9' { return -1 } n = n*10 + int(digits[i]-'0') } if n <= max { return n } return -1 } // loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or // X0-X31) to its 5-bit number, or -1. The vector banks are a register class // of their own: the toolchain accepts them only in the vector operands of the // LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1, // V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X // spellings never reach the integer/FP resolver. func loong64VecRegNum(name string) int { if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') { return -1 } return loong64RegSpecial(name[1:], 31) } // ---- format helpers ---- // l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd. func l64rrr(op uint32, rk, rj, rd int) uint32 { return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64rr encodes a 2R instruction: op | rj<<5 | rd. func l64rr(op uint32, rj, rd int) uint32 { return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd. func l64irr(op uint32, imm, rj, rd int) uint32 { return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd. func l64irr14(op uint32, imm, rj, rd int) uint32 { return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd. func l64irr16(op uint32, imm, rj, rd int) uint32 { return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64ir encodes a 2RI20 instruction: op | si20<<5 | rd. func l64ir(op uint32, imm, rd int) uint32 { return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f) } // l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the // 4-byte-aligned word distance (the toolchain stores the shifted value). func l64bbl(op uint32, offs int) uint32 { return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF } // l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF): // op | si21[15:0]<<10 | rj<<5 | si21[20:16]. func l64ir21(op uint32, offs, rj int) uint32 { v := uint32(offs) return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F } // l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4. func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 { return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f) } // l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd. // The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller // must have validated them (0..31 for the .w forms, 0..63 for the .d forms, // lsb <= msb), the same rule the toolchain enforces as "illegal bit number". func l64irir(op uint32, msb, rj, lsb, rd int) uint32 { return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f) } // l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd. func l64irrr(op uint32, sa, rk, rj, rd int) uint32 { return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f) } // l64i15 encodes a no-operand system instruction with a 15-bit code field // (SYSCALL, BREAK, DBAR): op | code[14:0]. func l64i15(op uint32, code int) uint32 { return op | uint32(code)&0x7FFF } // l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint. func l64irr5i(op uint32, offs, rj, hint int) uint32 { return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f) } // l64wordLE encodes a uint32 as 4 little-endian bytes. func l64wordLE(w uint32) []byte { return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} } // l64WordsLE concatenates one or more instruction words as little-endian bytes. func l64WordsLE(ws ...uint32) []byte { var out []byte for _, w := range ws { out = append(out, l64wordLE(w)...) } return out } // ---- instruction formats ---- type l64Format uint8 const ( l64Frrr l64Format = iota // 3R (integer and FP arithmetic) l64Frr // 2R l64Firr // 2RI12 (arithmetic with 12-bit immediate) l64Firr14 // 2RI14 (ldptr/stptr) l64Firr16 // 2RI16 (addu16i.d) l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i) l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel) l64Firir // bstrins/bstrpick l64Firrr // alsl l64Fi15 // syscall/break/dbar l64Fam // atomic (3R with the AM field order) l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0]) l64Fshift // 2RI12 with a 5/6-bit shift immediate l64Fpreld // preld (2RI12 + 5-bit hint) l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd ) // l64Enc is one instruction's encoding: its bit layout (format) and the // opcode constant, positioned at its exact bit range. type l64Enc struct { format l64Format op uint32 } // l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form // and the 2RI12 immediate form (which is a shift for the shift mnemonics). type l64DualEnc struct { rrr uint32 // 3R register form imm uint32 // 2RI12 immediate form shift bool // the immediate form is a 5/6-bit shift amount } // l64DualTable maps the dual-form arithmetic/logic mnemonics to both // encodings; the assembler picks by operand kind. var l64DualTable = map[string]l64DualEnc{} // l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them) // to their encoding. var l64InstrTable = map[string]l64Enc{} // l64Vec3Enc pairs a vector opcode with its register bank: false = LSX // (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per // spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and // `XVADDV X1, X2, X3`, and rejects the crossed spellings. type l64Vec3Enc struct { op uint32 lasx bool } // l64VecImmEnc carries the immediate-form encoding of a vector mnemonic: // the opcode, the bank, the accepted immediate range, the bias the toolchain // adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b // keeps a 5-bit two's-complement value, vseqi.d a 7-bit one). type l64VecImmEnc struct { op uint32 lasx bool min, max int bias int mask int } // l64VecBank marks the LSX/LASX mnemonics and records which register bank // each accepts; presence in the map routes the mnemonic through the vector // dispatcher rather than the integer/FP formats. var l64VecBank = map[string]bool{} // l64VecImmInfo mirrors l64VecImmTable for the dispatcher. var l64VecImmInfo = map[string]l64VecImmEnc{} // l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as // vpcnt.v). var l64Vec2R = map[string]bool{} // l64Vec4R marks the four-operand vector mnemonics (INSTR va, vk, vj, vd, // such as vshuf.b). var l64Vec4R = map[string]bool{} // l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit // 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels. type l64VmovqEnc struct { ld, st, ldx, stx uint32 // plain and indexed load/store replB, replH, replW, replD uint32 // vldrepl: load and replicate element pickS, pickU uint32 // vpickve2gr.{,u} element extract ins uint32 // vinsgr2vr element insert dup uint32 // vreplgr2vr duplicate (width in [11:10]) move uint32 // vori.b/xvori.b $0 register move } var l64VmovqTable = map[bool]l64VmovqEnc{ false: { // VMOVQ, the LSX (V) bank ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15, replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15, pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15, ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15, }, true: { // XVMOVQ, the LASX (X) bank ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15, replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15, pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15, ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15, }, } func init() { // 3R, integer. rrr := map[string]uint32{ "ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15, "SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15, "SGT": 0x24 << 15, "SGTU": 0x25 << 15, "MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15, "NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15, "ORN": 0x2c << 15, "ANDN": 0x2d << 15, "SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15, "SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15, "ROTR": 0x36 << 15, "ROTRV": 0x37 << 15, "MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15, "MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15, "MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15, "DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15, "DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15, "DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15, "CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15, "CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15, } // 3R, floating point. rrr["MULF"] = 0x209 << 15 rrr["MULD"] = 0x20a << 15 rrr["DIVF"] = 0x20d << 15 rrr["DIVD"] = 0x20e << 15 rrr["SUBF"] = 0x205 << 15 rrr["SUBD"] = 0x206 << 15 rrr["ADDF"] = 0x201 << 15 rrr["ADDD"] = 0x202 << 15 rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15 rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15 rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15 rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15 rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15 rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15 rrr["FMINF"] = 0x215 << 15 rrr["FMIND"] = 0x216 << 15 rrr["FMAXF"] = 0x211 << 15 rrr["FMAXD"] = 0x212 << 15 rrr["FMAXAF"] = 0x219 << 15 rrr["FMAXAD"] = 0x21a << 15 rrr["FMINAF"] = 0x21d << 15 rrr["FMINAD"] = 0x21e << 15 rrr["FSCALEBF"] = 0x221 << 15 rrr["FSCALEBD"] = 0x222 << 15 rrr["FCOPYSGF"] = 0x225 << 15 rrr["FCOPYSGD"] = 0x226 << 15 for m, op := range rrr { l64InstrTable[m] = l64Enc{format: l64Frrr, op: op} } // 2R. rr := map[string]uint32{ "CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10, "CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10, "REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10, "REVH2W": 0x10 << 10, "REVHV": 0x11 << 10, "BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10, "EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10, "TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10, "MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10, "MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10, "FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10, "MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10, "ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10, "MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10, "NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10, "SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10, "FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10, "FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10, "FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10, "FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10, "FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10, "FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10, "FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10, "FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10, "FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10, "FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10, // LSX: convert a 64-bit integer lane to a double float. The operand // bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`), // so the entry stays on the 2R integer/FP format. "FFINTDV": 0x474a << 10, // The rest of the scalar conversions (all F-bank, 2R). "FFINTFW": 0x4744 << 10, // ffint.s.w "FFINTFV": 0x4746 << 10, // ffint.s.l "FFINTDW": 0x4748 << 10, // ffint.d.w "FTINTWF": 0x46c1 << 10, // ftint.w.s "FTINTWD": 0x46c2 << 10, // ftint.w.d "FTINTVF": 0x46c9 << 10, // ftint.l.s "FTINTVD": 0x46ca << 10, // ftint.l.d } for m, op := range rr { l64InstrTable[m] = l64Enc{format: l64Frr, op: op} } // RDTIME is a 2R instruction with rd and rj in swapped positions. l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10} l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10} l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10} // The dual-form arithmetic mnemonics (register 3R + immediate 2RI12), // selected by the operand kind; the shift mnemonics pair the 3R form // with a 5/6-bit shift immediate. maps.Copy(l64DualTable, map[string]l64DualEnc{ "ADD": {rrr: 0x20 << 15, imm: 0x00a << 22}, "ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22}, "ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22}, "ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22}, "AND": {rrr: 0x29 << 15, imm: 0x00d << 22}, "OR": {rrr: 0x2a << 15, imm: 0x00e << 22}, "XOR": {rrr: 0x2b << 15, imm: 0x00f << 22}, "SGT": {rrr: 0x24 << 15, imm: 0x008 << 22}, "SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22}, "SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true}, "SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true}, "SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true}, "ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true}, "SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true}, "SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true}, "SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true}, "ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true}, }) // 2RI12, pure immediate arithmetic (LU52ID has no register form). l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22} // ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16. l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26} // 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and // stores (ldptr/stptr), with the offset scaled by 4. l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w) l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w) l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d) // 2RI20. l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25} l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25} l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25} l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25} // LUI is the Plan 9 spelling of lu12i.w. l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25} // 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC // condition flag, the layout matches the 4R shape). rrrr := map[string]uint32{ "FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20, "FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20, "FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20, "FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20, "FSEL": 0x340 << 18, } for m, op := range rrrr { l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op} } // IRIR, bit-field insert/extract. irir := map[string]uint32{ "BSTRINSW": 0x3<<21 | 0x0<<15, "BSTRINSV": 0x2 << 22, "BSTRPICKW": 0x3<<21 | 0x1<<15, "BSTRPICKV": 0x3 << 22, } for m, op := range irir { l64InstrTable[m] = l64Enc{format: l64Firir, op: op} } // 3RI2, ALSL. irrr := map[string]uint32{ "ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17, } for m, op := range irrr { l64InstrTable[m] = l64Enc{format: l64Firrr, op: op} } // 0-operand system instructions. l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15} l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15} l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15} // PRELD. l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22} // Atomics, 3R with the AM field order (rk=value, rj=address, rd=result). // The toolchain's form is three operands, `AMADDW rk, (rj), rd` // (cmd/asm/internal/asm/testdata/loong64enc1.s and // internal/runtime/atomic/atomic_loong64.s); the two-register spelling // is rejected by the oracle. am := map[string]uint32{ "AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15, "AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15, "AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15, "AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15, "AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15, "AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15, "AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15, "AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15, "AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15, "AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15, "AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15, "AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15, "AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15, "AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15, "AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15, "AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15, // The _dbar (acquire/release) add, and, or variants: opcodes read off // `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends. "AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15, "AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15, "AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15, // The remaining _dbar exchange variants (loong64enc1.s). "AMXORDBW": 0x070DA << 15, "AMXORDBV": 0x070DB << 15, "AMMAXDBW": 0x070DC << 15, "AMMAXDBV": 0x070DD << 15, "AMMINDBW": 0x070DE << 15, "AMMINDBV": 0x070DF << 15, "AMMAXDBWU": 0x070E0 << 15, "AMMAXDBVU": 0x070E1 << 15, "AMMINDBWU": 0x070E2 << 15, "AMMINDBVU": 0x070E3 << 15, } for m, op := range am { l64InstrTable[m] = l64Enc{format: l64Fam, op: op} } // ---- LSX/LASX (V*/XV*) ---- // Every opcode below was read off `go tool objdump` of a GOARCH=loong64 // `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks // most of them), not assumed from the LoongArch manual. // Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with // vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain // accepts: LSX spellings take V0-V31, LASX spellings X0-X31. vec3 := map[string]l64Vec3Enc{ "VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false}, "VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false}, "VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false}, "VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false}, "XVADDV": {0xE817 << 15, true}, "XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true}, "XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true}, } // The integer and FP add/subtract families: [X]VADD and [X]VSUB by lane // width, plus the [X]VSADD/[X]VSSUB saturating pairs. // Opcodes transcribed from the toolchain's loong64enc1.s. addsub := map[string]l64Vec3Enc{ "VADDB": {0xE014 << 15, false}, "VADDH": {0xE015 << 15, false}, "VADDD": {0xE262 << 15, false}, "VADDF": {0xE261 << 15, false}, "VADDQ": {0xE25A << 15, false}, "VSUBB": {0xE018 << 15, false}, "VSUBH": {0xE019 << 15, false}, "VSUBW": {0xE01A << 15, false}, "VSUBV": {0xE01B << 15, false}, "VSUBQ": {0xE25B << 15, false}, "VSUBF": {0xE265 << 15, false}, "VSUBD": {0xE266 << 15, false}, "VSADDB": {0xE08C << 15, false}, "VSADDH": {0xE08D << 15, false}, "VSADDW": {0xE08E << 15, false}, "VSADDV": {0xE08F << 15, false}, "VSADDBU": {0xE094 << 15, false}, "VSADDHU": {0xE095 << 15, false}, "VSADDWU": {0xE096 << 15, false}, "VSADDVU": {0xE097 << 15, false}, "VSSUBB": {0xE090 << 15, false}, "VSSUBH": {0xE091 << 15, false}, "VSSUBW": {0xE092 << 15, false}, "VSSUBV": {0xE093 << 15, false}, "VSSUBBU": {0xE098 << 15, false}, "VSSUBHU": {0xE099 << 15, false}, "VSSUBWU": {0xE09A << 15, false}, "VSSUBVU": {0xE09B << 15, false}, "XVADDB": {0xE814 << 15, true}, "XVADDH": {0xE815 << 15, true}, "XVADDW": {0xE816 << 15, true}, "XVADDD": {0xEA62 << 15, true}, "XVADDF": {0xEA61 << 15, true}, "XVADDQ": {0xEA5A << 15, true}, "XVSUBB": {0xE818 << 15, true}, "XVSUBH": {0xE819 << 15, true}, "XVSUBW": {0xE81A << 15, true}, "XVSUBV": {0xE81B << 15, true}, "XVSUBQ": {0xEA5B << 15, true}, "XVSUBF": {0xEA65 << 15, true}, "XVSUBD": {0xEA66 << 15, true}, "XVSADDB": {0xE88C << 15, true}, "XVSADDH": {0xE88D << 15, true}, "XVSADDW": {0xE88E << 15, true}, "XVSADDV": {0xE88F << 15, true}, "XVSADDBU": {0xE894 << 15, true}, "XVSADDHU": {0xE895 << 15, true}, "XVSADDWU": {0xE896 << 15, true}, "XVSADDVU": {0xE897 << 15, true}, "XVSSUBB": {0xE890 << 15, true}, "XVSSUBH": {0xE891 << 15, true}, "XVSSUBW": {0xE892 << 15, true}, "XVSSUBV": {0xE893 << 15, true}, "XVSSUBBU": {0xE898 << 15, true}, "XVSSUBHU": {0xE899 << 15, true}, "XVSSUBWU": {0xE89A << 15, true}, "XVSSUBVU": {0xE89B << 15, true}, } // The multiply families: plain and high-half [X]VMUL/[X]VMUH, the // widening [X]VMULW{EV,OD} ladder and its accumulating [X]VMADDW twins, // plus the [X]VMADD/[X]VMSUB fused multiply-add and the [X]VDIV/[X]VMOD // divide and modulo pairs. muldiv := map[string]l64Vec3Enc{ "VMULB": {0xE108 << 15, false}, "VMULH": {0xE109 << 15, false}, "VMULW": {0xE10A << 15, false}, "VMULV": {0xE10B << 15, false}, "VMUHB": {0xE10C << 15, false}, "VMUHH": {0xE10D << 15, false}, "VMUHW": {0xE10E << 15, false}, "VMUHV": {0xE10F << 15, false}, "VMUHBU": {0xE110 << 15, false}, "VMUHHU": {0xE111 << 15, false}, "VMUHWU": {0xE112 << 15, false}, "VMUHVU": {0xE113 << 15, false}, "VMULWEVHB": {0xE120 << 15, false}, "VMULWEVWH": {0xE121 << 15, false}, "VMULWEVVW": {0xE122 << 15, false}, "VMULWEVQV": {0xE123 << 15, false}, "VMULWODHB": {0xE124 << 15, false}, "VMULWODWH": {0xE125 << 15, false}, "VMULWODVW": {0xE126 << 15, false}, "VMULWODQV": {0xE127 << 15, false}, "VMULWEVHBU": {0xE130 << 15, false}, "VMULWEVWHU": {0xE131 << 15, false}, "VMULWEVVWU": {0xE132 << 15, false}, "VMULWEVQVU": {0xE133 << 15, false}, "VMULWODHBU": {0xE134 << 15, false}, "VMULWODWHU": {0xE135 << 15, false}, "VMULWODVWU": {0xE136 << 15, false}, "VMULWODQVU": {0xE137 << 15, false}, "VMULWEVHBUB": {0xE140 << 15, false}, "VMULWEVWHUH": {0xE141 << 15, false}, "VMULWEVVWUW": {0xE142 << 15, false}, "VMULWEVQVUV": {0xE143 << 15, false}, "VMULWODHBUB": {0xE144 << 15, false}, "VMULWODWHUH": {0xE145 << 15, false}, "VMULWODVWUW": {0xE146 << 15, false}, "VMULWODQVUV": {0xE147 << 15, false}, "VMADDB": {0xE150 << 15, false}, "VMADDH": {0xE151 << 15, false}, "VMADDW": {0xE152 << 15, false}, "VMADDV": {0xE153 << 15, false}, "VMSUBB": {0xE154 << 15, false}, "VMSUBH": {0xE155 << 15, false}, "VMSUBW": {0xE156 << 15, false}, "VMSUBV": {0xE157 << 15, false}, "VMADDWEVHB": {0xE158 << 15, false}, "VMADDWEVWH": {0xE159 << 15, false}, "VMADDWEVVW": {0xE15A << 15, false}, "VMADDWEVQV": {0xE15B << 15, false}, "VMADDWODHB": {0xE15C << 15, false}, "VMADDWODWH": {0xE15D << 15, false}, "VMADDWODVW": {0xE15E << 15, false}, "VMADDWODQV": {0xE15F << 15, false}, "VMADDWEVHBU": {0xE168 << 15, false}, "VMADDWEVWHU": {0xE169 << 15, false}, "VMADDWEVVWU": {0xE16A << 15, false}, "VMADDWEVQVU": {0xE16B << 15, false}, "VMADDWODHBU": {0xE16C << 15, false}, "VMADDWODWHU": {0xE16D << 15, false}, "VMADDWODVWU": {0xE16E << 15, false}, "VMADDWODQVU": {0xE16F << 15, false}, "VMADDWEVHBUB": {0xE178 << 15, false}, "VMADDWEVWHUH": {0xE179 << 15, false}, "VMADDWEVVWUW": {0xE17A << 15, false}, "VMADDWEVQVUV": {0xE17B << 15, false}, "VMADDWODHBUB": {0xE17C << 15, false}, "VMADDWODWHUH": {0xE17D << 15, false}, "VMADDWODVWUW": {0xE17E << 15, false}, "VMADDWODQVUV": {0xE17F << 15, false}, "VDIVB": {0xE1C0 << 15, false}, "VDIVH": {0xE1C1 << 15, false}, "VDIVW": {0xE1C2 << 15, false}, "VDIVV": {0xE1C3 << 15, false}, "VMODB": {0xE1C4 << 15, false}, "VMODH": {0xE1C5 << 15, false}, "VMODW": {0xE1C6 << 15, false}, "VMODV": {0xE1C7 << 15, false}, "VDIVBU": {0xE1C8 << 15, false}, "VDIVHU": {0xE1C9 << 15, false}, "VDIVWU": {0xE1CA << 15, false}, "VDIVVU": {0xE1CB << 15, false}, "VMODBU": {0xE1CC << 15, false}, "VMODHU": {0xE1CD << 15, false}, "VMODWU": {0xE1CE << 15, false}, "VMODVU": {0xE1CF << 15, false}, "VMULF": {0xE271 << 15, false}, "VMULD": {0xE272 << 15, false}, "VDIVF": {0xE275 << 15, false}, "VDIVD": {0xE276 << 15, false}, "XVMULB": {0xE908 << 15, true}, "XVMULH": {0xE909 << 15, true}, "XVMULW": {0xE90A << 15, true}, "XVMULV": {0xE90B << 15, true}, "XVMUHB": {0xE90C << 15, true}, "XVMUHH": {0xE90D << 15, true}, "XVMUHW": {0xE90E << 15, true}, "XVMUHV": {0xE90F << 15, true}, "XVMUHBU": {0xE910 << 15, true}, "XVMUHHU": {0xE911 << 15, true}, "XVMUHWU": {0xE912 << 15, true}, "XVMUHVU": {0xE913 << 15, true}, "XVMULWEVHB": {0xE920 << 15, true}, "XVMULWEVWH": {0xE921 << 15, true}, "XVMULWEVVW": {0xE922 << 15, true}, "XVMULWEVQV": {0xE923 << 15, true}, "XVMULWODHB": {0xE924 << 15, true}, "XVMULWODWH": {0xE925 << 15, true}, "XVMULWODVW": {0xE926 << 15, true}, "XVMULWODQV": {0xE927 << 15, true}, "XVMULWEVHBU": {0xE930 << 15, true}, "XVMULWEVWHU": {0xE931 << 15, true}, "XVMULWEVVWU": {0xE932 << 15, true}, "XVMULWEVQVU": {0xE933 << 15, true}, "XVMULWODHBU": {0xE934 << 15, true}, "XVMULWODWHU": {0xE935 << 15, true}, "XVMULWODVWU": {0xE936 << 15, true}, "XVMULWODQVU": {0xE937 << 15, true}, "XVMULWEVHBUB": {0xE940 << 15, true}, "XVMULWEVWHUH": {0xE941 << 15, true}, "XVMULWEVVWUW": {0xE942 << 15, true}, "XVMULWEVQVUV": {0xE943 << 15, true}, "XVMULWODHBUB": {0xE944 << 15, true}, "XVMULWODWHUH": {0xE945 << 15, true}, "XVMULWODVWUW": {0xE946 << 15, true}, "XVMULWODQVUV": {0xE947 << 15, true}, "XVMADDB": {0xE950 << 15, true}, "XVMADDH": {0xE951 << 15, true}, "XVMADDW": {0xE952 << 15, true}, "XVMADDV": {0xE953 << 15, true}, "XVMSUBB": {0xE954 << 15, true}, "XVMSUBH": {0xE955 << 15, true}, "XVMSUBW": {0xE956 << 15, true}, "XVMSUBV": {0xE957 << 15, true}, "XVMADDWEVHB": {0xE958 << 15, true}, "XVMADDWEVWH": {0xE959 << 15, true}, "XVMADDWEVVW": {0xE95A << 15, true}, "XVMADDWEVQV": {0xE95B << 15, true}, "XVMADDWODHB": {0xE95C << 15, true}, "XVMADDWODWH": {0xE95D << 15, true}, "XVMADDWODVW": {0xE95E << 15, true}, "XVMADDWODQV": {0xE95F << 15, true}, "XVMADDWEVHBU": {0xE968 << 15, true}, "XVMADDWEVWHU": {0xE969 << 15, true}, "XVMADDWEVVWU": {0xE96A << 15, true}, "XVMADDWEVQVU": {0xE96B << 15, true}, "XVMADDWODHBU": {0xE96C << 15, true}, "XVMADDWODWHU": {0xE96D << 15, true}, "XVMADDWODVWU": {0xE96E << 15, true}, "XVMADDWODQVU": {0xE96F << 15, true}, "XVMADDWEVHBUB": {0xE978 << 15, true}, "XVMADDWEVWHUH": {0xE979 << 15, true}, "XVMADDWEVVWUW": {0xE97A << 15, true}, "XVMADDWEVQVUV": {0xE97B << 15, true}, "XVMADDWODHBUB": {0xE97C << 15, true}, "XVMADDWODWHUH": {0xE97D << 15, true}, "XVMADDWODVWUW": {0xE97E << 15, true}, "XVMADDWODQVUV": {0xE97F << 15, true}, "XVDIVB": {0xE9C0 << 15, true}, "XVDIVH": {0xE9C1 << 15, true}, "XVDIVW": {0xE9C2 << 15, true}, "XVDIVV": {0xE9C3 << 15, true}, "XVMODB": {0xE9C4 << 15, true}, "XVMODH": {0xE9C5 << 15, true}, "XVMODW": {0xE9C6 << 15, true}, "XVMODV": {0xE9C7 << 15, true}, "XVDIVBU": {0xE9C8 << 15, true}, "XVDIVHU": {0xE9C9 << 15, true}, "XVDIVWU": {0xE9CA << 15, true}, "XVDIVVU": {0xE9CB << 15, true}, "XVMODBU": {0xE9CC << 15, true}, "XVMODHU": {0xE9CD << 15, true}, "XVMODWU": {0xE9CE << 15, true}, "XVMODVU": {0xE9CF << 15, true}, "XVMULF": {0xEA71 << 15, true}, "XVMULD": {0xEA72 << 15, true}, "XVDIVF": {0xEA75 << 15, true}, "XVDIVD": {0xEA76 << 15, true}, } // The lane-wise shifts and rotates (three-register forms; the immediate // forms live in l64VecImmInfo), the interleave families, the bit // clear/set/rev register forms, the remaining logic and compare // spellings, the widening add/subtract ladder and the vector FP // arithmetic. vecmisc := map[string]l64Vec3Enc{ "VSLLB": {0xE1D0 << 15, false}, "VSLLH": {0xE1D1 << 15, false}, "VSLLW": {0xE1D2 << 15, false}, "VSLLV": {0xE1D3 << 15, false}, "VSRLB": {0xE1D4 << 15, false}, "VSRLH": {0xE1D5 << 15, false}, "VSRLW": {0xE1D6 << 15, false}, "VSRLV": {0xE1D7 << 15, false}, "VSRAH": {0xE1D9 << 15, false}, "VSRAW": {0xE1DA << 15, false}, "VSRAV": {0xE1DB << 15, false}, "VROTRB": {0xE1DC << 15, false}, "VROTRH": {0xE1DD << 15, false}, "VROTRV": {0xE1DF << 15, false}, "VILVLB": {0xE234 << 15, false}, "VILVLH": {0xE235 << 15, false}, "VILVLW": {0xE236 << 15, false}, "VILVLV": {0xE237 << 15, false}, "VILVHB": {0xE238 << 15, false}, "VILVHH": {0xE239 << 15, false}, "VILVHW": {0xE23A << 15, false}, "VILVHV": {0xE23B << 15, false}, "VBITCLRB": {0xE218 << 15, false}, "VBITCLRH": {0xE219 << 15, false}, "VBITCLRW": {0xE21A << 15, false}, "VBITCLRV": {0xE21B << 15, false}, "VBITSETB": {0xE21C << 15, false}, "VBITSETH": {0xE21D << 15, false}, "VBITSETW": {0xE21E << 15, false}, "VBITSETV": {0xE21F << 15, false}, "VBITREVB": {0xE220 << 15, false}, "VBITREVH": {0xE221 << 15, false}, "VBITREVW": {0xE222 << 15, false}, "VBITREVV": {0xE223 << 15, false}, "VORV": {0xE24D << 15, false}, "VNORV": {0xE24F << 15, false}, "VANDNV": {0xE250 << 15, false}, "VORNV": {0xE251 << 15, false}, "VSEQH": {0xE001 << 15, false}, "VSEQW": {0xE002 << 15, false}, "VSLTB": {0xE00C << 15, false}, "VSLTH": {0xE00D << 15, false}, "VSLTW": {0xE00E << 15, false}, "VSLTV": {0xE00F << 15, false}, "VSLTBU": {0xE010 << 15, false}, "VSLTHU": {0xE011 << 15, false}, "VSLTWU": {0xE012 << 15, false}, "VSLTVU": {0xE013 << 15, false}, "VADDWEVHB": {0xE03C << 15, false}, "VADDWEVWH": {0xE03D << 15, false}, "VADDWEVVW": {0xE03E << 15, false}, "VADDWEVQV": {0xE03F << 15, false}, "VSUBWEVHB": {0xE040 << 15, false}, "VSUBWEVWH": {0xE041 << 15, false}, "VSUBWEVVW": {0xE042 << 15, false}, "VSUBWEVQV": {0xE043 << 15, false}, "VADDWODHB": {0xE044 << 15, false}, "VADDWODWH": {0xE045 << 15, false}, "VADDWODVW": {0xE046 << 15, false}, "VADDWODQV": {0xE047 << 15, false}, "VSUBWODHB": {0xE048 << 15, false}, "VSUBWODWH": {0xE049 << 15, false}, "VSUBWODVW": {0xE04A << 15, false}, "VSUBWODQV": {0xE04B << 15, false}, "VSUBWEVHBU": {0xE060 << 15, false}, "VSUBWEVWHU": {0xE061 << 15, false}, "VSUBWEVVWU": {0xE062 << 15, false}, "VSUBWEVQVU": {0xE063 << 15, false}, "VADDWEVHBU": {0xE05C << 15, false}, "VADDWEVWHU": {0xE05D << 15, false}, "VADDWEVVWU": {0xE05E << 15, false}, "VADDWEVQVU": {0xE05F << 15, false}, "VADDWODHBU": {0xE064 << 15, false}, "VADDWODWHU": {0xE065 << 15, false}, "VADDWODVWU": {0xE066 << 15, false}, "VADDWODQVU": {0xE067 << 15, false}, "VSUBWODHBU": {0xE068 << 15, false}, "VSUBWODWHU": {0xE069 << 15, false}, "VSUBWODVWU": {0xE06A << 15, false}, "VSUBWODQVU": {0xE06B << 15, false}, "VSHUFH": {0xE2F5 << 15, false}, "VSHUFW": {0xE2F6 << 15, false}, "VSHUFV": {0xE2F7 << 15, false}, "XVSLLB": {0xE9D0 << 15, true}, "XVSLLH": {0xE9D1 << 15, true}, "XVSLLW": {0xE9D2 << 15, true}, "XVSLLV": {0xE9D3 << 15, true}, "XVSRLB": {0xE9D4 << 15, true}, "XVSRLH": {0xE9D5 << 15, true}, "XVSRLW": {0xE9D6 << 15, true}, "XVSRLV": {0xE9D7 << 15, true}, "XVSRAB": {0xE9D8 << 15, true}, "XVSRAH": {0xE9D9 << 15, true}, "XVSRAW": {0xE9DA << 15, true}, "XVSRAV": {0xE9DB << 15, true}, "XVROTRB": {0xE9DC << 15, true}, "XVROTRH": {0xE9DD << 15, true}, "XVROTRW": {0xE9DE << 15, true}, "XVROTRV": {0xE9DF << 15, true}, "XVILVLB": {0xEA34 << 15, true}, "XVILVLH": {0xEA35 << 15, true}, "XVILVLW": {0xEA36 << 15, true}, "XVILVLV": {0xEA37 << 15, true}, "XVILVHB": {0xEA38 << 15, true}, "XVILVHH": {0xEA39 << 15, true}, "XVILVHW": {0xEA3A << 15, true}, "XVILVHV": {0xEA3B << 15, true}, "XVBITCLRB": {0xEA18 << 15, true}, "XVBITCLRH": {0xEA19 << 15, true}, "XVBITCLRW": {0xEA1A << 15, true}, "XVBITCLRV": {0xEA1B << 15, true}, "XVBITSETB": {0xEA1C << 15, true}, "XVBITSETH": {0xEA1D << 15, true}, "XVBITSETW": {0xEA1E << 15, true}, "XVBITSETV": {0xEA1F << 15, true}, "XVBITREVB": {0xEA20 << 15, true}, "XVBITREVH": {0xEA21 << 15, true}, "XVBITREVW": {0xEA22 << 15, true}, "XVBITREVV": {0xEA23 << 15, true}, "XVORV": {0xEA4D << 15, true}, "XVNORV": {0xEA4F << 15, true}, "XVANDNV": {0xEA50 << 15, true}, "XVORNV": {0xEA51 << 15, true}, "XVSEQH": {0xE801 << 15, true}, "XVSEQW": {0xE802 << 15, true}, "XVSLTB": {0xE80C << 15, true}, "XVSLTH": {0xE80D << 15, true}, "XVSLTW": {0xE80E << 15, true}, "XVSLTV": {0xE80F << 15, true}, "XVSLTBU": {0xE810 << 15, true}, "XVSLTHU": {0xE811 << 15, true}, "XVSLTWU": {0xE812 << 15, true}, "XVSLTVU": {0xE813 << 15, true}, "XVADDWEVHB": {0xE83C << 15, true}, "XVADDWEVWH": {0xE83D << 15, true}, "XVADDWEVVW": {0xE83E << 15, true}, "XVADDWEVQV": {0xE83F << 15, true}, "XVSUBWEVHB": {0xE840 << 15, true}, "XVSUBWEVWH": {0xE841 << 15, true}, "XVSUBWEVVW": {0xE842 << 15, true}, "XVSUBWEVQV": {0xE843 << 15, true}, "XVADDWODHB": {0xE844 << 15, true}, "XVADDWODWH": {0xE845 << 15, true}, "XVADDWODVW": {0xE846 << 15, true}, "XVADDWODQV": {0xE847 << 15, true}, "XVSUBWODHB": {0xE848 << 15, true}, "XVSUBWODWH": {0xE849 << 15, true}, "XVSUBWODVW": {0xE84A << 15, true}, "XVSUBWODQV": {0xE84B << 15, true}, "XVADDWEVHBU": {0xE85C << 15, true}, "XVADDWEVWHU": {0xE85D << 15, true}, "XVADDWEVVWU": {0xE85E << 15, true}, "XVADDWEVQVU": {0xE85F << 15, true}, "XVSUBWEVHBU": {0xE860 << 15, true}, "XVSUBWEVWHU": {0xE861 << 15, true}, "XVSUBWEVVWU": {0xE862 << 15, true}, "XVSUBWEVQVU": {0xE863 << 15, true}, "XVADDWODHBU": {0xE864 << 15, true}, "XVADDWODWHU": {0xE865 << 15, true}, "XVADDWODVWU": {0xE866 << 15, true}, "XVADDWODQVU": {0xE867 << 15, true}, "XVSUBWODHBU": {0xE868 << 15, true}, "XVSUBWODWHU": {0xE869 << 15, true}, "XVSUBWODVWU": {0xE86A << 15, true}, "XVSUBWODQVU": {0xE86B << 15, true}, "XVSHUFH": {0xEAF5 << 15, true}, "XVSHUFW": {0xEAF6 << 15, true}, "XVSHUFV": {0xEAF7 << 15, true}, } for _, tab := range []map[string]l64Vec3Enc{addsub, muldiv, vecmisc} { for m, e := range tab { if _, dup := vec3[m]; dup { panic("loong64: duplicate vector mnemonic " + m) } vec3[m] = e } } for m, e := range vec3 { l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op} l64VecBank[m] = e.lasx } // Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate // range, bias and field mask are the ones the toolchain encodes: vandi.b // stores the raw 8-bit constant, vsrari.b stores imm+8 (lane-width // bias), the si5 compares store 5-bit two's-complement values and vseqi.d // a 7-bit field the toolchain range-checks down to si5. // The mnemonics that also have a register form (the shifts, the bit // clear/set/rev families, VSEQ and the logic immediates) keep their // three-register entry in l64InstrTable; the dispatcher picks the // immediate opcode from l64VecImmInfo by operand kind, so the immediate // entries must not overwrite the table. vecImm := map[string]l64VecImmEnc{ "VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF}, "XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF}, "VORB": {0xE7A8 << 15, false, 0, 255, 0, 0xFF}, "XVORB": {0xEFA8 << 15, true, 0, 255, 0, 0xFF}, "VXORB": {0xE7B0 << 15, false, 0, 255, 0, 0xFF}, "XVXORB": {0xEFB0 << 15, true, 0, 255, 0, 0xFF}, "VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF}, "XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF}, "VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F}, "XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F}, // vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a // 7-bit field, but the toolchain range-checks it down to si5 as well // (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64). "VSEQH": {0xE501 << 15, false, -16, 15, 0, 0x1F}, "XVSEQH": {0xED01 << 15, true, -16, 15, 0, 0x1F}, "VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F}, "XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F}, "VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F}, "XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F}, // vslti compares against a signed (or, in the U spellings, unsigned) // si5/ui5 constant. "VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F}, "XVSLTB": {0xED0C << 15, true, -16, 15, 0, 0x1F}, "VSLTH": {0xE50D << 15, false, -16, 15, 0, 0x1F}, "XVSLTH": {0xED0D << 15, true, -16, 15, 0, 0x1F}, "VSLTW": {0xE50E << 15, false, -16, 15, 0, 0x1F}, "XVSLTW": {0xED0E << 15, true, -16, 15, 0, 0x1F}, "VSLTV": {0xE50F << 15, false, -16, 15, 0, 0x1F}, "XVSLTV": {0xED0F << 15, true, -16, 15, 0, 0x1F}, "VSLTBU": {0xE510 << 15, false, 0, 31, 0, 0x1F}, "XVSLTBU": {0xED10 << 15, true, 0, 31, 0, 0x1F}, "VSLTHU": {0xE511 << 15, false, 0, 31, 0, 0x1F}, "XVSLTHU": {0xED11 << 15, true, 0, 31, 0, 0x1F}, "VSLTWU": {0xE512 << 15, false, 0, 31, 0, 0x1F}, "XVSLTWU": {0xED12 << 15, true, 0, 31, 0, 0x1F}, "VSLTVU": {0xE513 << 15, false, 0, 31, 0, 0x1F}, "XVSLTVU": {0xED13 << 15, true, 0, 31, 0, 0x1F}, // vaddi/vsubi take ui5 constants for every width on this toolchain // (VADDVU $32 is rejected by the oracle although the field is ui8). "VADDBU": {0xE514 << 15, false, 0, 31, 0, 0x1F}, "XVADDBU": {0xED14 << 15, true, 0, 31, 0, 0x1F}, "VADDHU": {0xE515 << 15, false, 0, 31, 0, 0x1F}, "XVADDHU": {0xED15 << 15, true, 0, 31, 0, 0x1F}, "VADDWU": {0xE516 << 15, false, 0, 31, 0, 0x1F}, "XVADDWU": {0xED16 << 15, true, 0, 31, 0, 0x1F}, "VADDVU": {0xE517 << 15, false, 0, 31, 0, 0x1F}, "XVADDVU": {0xED17 << 15, true, 0, 31, 0, 0x1F}, "VSUBBU": {0xE518 << 15, false, 0, 31, 0, 0x1F}, "XVSUBBU": {0xED18 << 15, true, 0, 31, 0, 0x1F}, "VSUBHU": {0xE519 << 15, false, 0, 31, 0, 0x1F}, "XVSUBHU": {0xED19 << 15, true, 0, 31, 0, 0x1F}, "VSUBWU": {0xE51A << 15, false, 0, 31, 0, 0x1F}, "XVSUBWU": {0xED1A << 15, true, 0, 31, 0, 0x1F}, "VSUBVU": {0xE51B << 15, false, 0, 31, 0, 0x1F}, "XVSUBVU": {0xED1B << 15, true, 0, 31, 0, 0x1F}, // The shift/rotate immediates ride in a width-sized field whose upper // bits carry the lane-width code: vslli.b stores ui3 at [12:0] with // bits [14:13] inside the opcode, vslli.h ui4 under a 4 bit mask, and // the .w/.d spellings a raw ui5/ui6. "VSLLB": {0x732C2000, false, 0, 7, 0, 0x7}, "XVSLLB": {0x772C2000, true, 0, 7, 0, 0x7}, "VSLLH": {0x732C4000, false, 0, 15, 0, 0xF}, "XVSLLH": {0x772C4000, true, 0, 15, 0, 0xF}, "VSLLW": {0xE659 << 15, false, 0, 31, 0, 0x1F}, "XVSLLW": {0xEE59 << 15, true, 0, 31, 0, 0x1F}, "VSLLV": {0xE65A << 15, false, 0, 63, 0, 0x3F}, "XVSLLV": {0xEE5A << 15, true, 0, 63, 0, 0x3F}, "VSRLB": {0x73302000, false, 0, 7, 0, 0x7}, "XVSRLB": {0x77302000, true, 0, 7, 0, 0x7}, "VSRLH": {0x73304000, false, 0, 15, 0, 0xF}, "XVSRLH": {0x77304000, true, 0, 15, 0, 0xF}, "VSRLW": {0xE661 << 15, false, 0, 31, 0, 0x1F}, "XVSRLW": {0xEE61 << 15, true, 0, 31, 0, 0x1F}, "VSRLV": {0xE662 << 15, false, 0, 63, 0, 0x3F}, "XVSRLV": {0xEE62 << 15, true, 0, 63, 0, 0x3F}, // vsrari/vrotri bias the field so the lane-width code rides above the // shift amount (.b adds 8, .h 16, .w 32; .d is a raw ui6). "VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F}, "XVSRAB": {0xEE68 << 15, true, 0, 7, 8, 0x1F}, "VSRAH": {0x73344000, false, 0, 15, 0, 0xF}, "XVSRAH": {0x77344000, true, 0, 15, 0, 0xF}, "VSRAW": {0xE669 << 15, false, 0, 31, 0, 0x1F}, "XVSRAW": {0xEE69 << 15, true, 0, 31, 0, 0x1F}, "VSRAV": {0xE66A << 15, false, 0, 63, 0, 0x3F}, "XVSRAV": {0xEE6A << 15, true, 0, 63, 0, 0x3F}, "VROTRB": {0x72A02000, false, 0, 7, 0, 0x7}, "XVROTRB": {0x76A02000, true, 0, 7, 0, 0x7}, "VROTRH": {0x72A04000, false, 0, 15, 0, 0xF}, "XVROTRH": {0x76A04000, true, 0, 15, 0, 0xF}, "VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F}, "XVROTRW": {0xED41 << 15, true, 0, 31, 0, 0x1F}, "VROTRV": {0xE542 << 15, false, 0, 63, 0, 0x3F}, "XVROTRV": {0xED42 << 15, true, 0, 63, 0, 0x3F}, // vbitclri/vbitseti/vbitrevi follow the same width-coded layout. "VBITCLRB": {0x73102000, false, 0, 7, 0, 0x7}, "XVBITCLRB": {0x77102000, true, 0, 7, 0, 0x7}, "VBITCLRH": {0x73104000, false, 0, 15, 0, 0xF}, "XVBITCLRH": {0x77104000, true, 0, 15, 0, 0xF}, "VBITCLRW": {0xE621 << 15, false, 0, 31, 0, 0x1F}, "XVBITCLRW": {0xEE21 << 15, true, 0, 31, 0, 0x1F}, "VBITCLRV": {0xE622 << 15, false, 0, 63, 0, 0x3F}, "XVBITCLRV": {0xEE22 << 15, true, 0, 63, 0, 0x3F}, "VBITSETB": {0x73142000, false, 0, 7, 0, 0x7}, "XVBITSETB": {0x77142000, true, 0, 7, 0, 0x7}, "VBITSETH": {0x73144000, false, 0, 15, 0, 0xF}, "XVBITSETH": {0x77144000, true, 0, 15, 0, 0xF}, "VBITSETW": {0xE629 << 15, false, 0, 31, 0, 0x1F}, "XVBITSETW": {0xEE29 << 15, true, 0, 31, 0, 0x1F}, "VBITSETV": {0xE62A << 15, false, 0, 63, 0, 0x3F}, "XVBITSETV": {0xEE2A << 15, true, 0, 63, 0, 0x3F}, "VBITREVB": {0x73182000, false, 0, 7, 0, 0x7}, "XVBITREVB": {0x77182000, true, 0, 7, 0, 0x7}, "VBITREVH": {0x73184000, false, 0, 15, 0, 0xF}, "XVBITREVH": {0x77184000, true, 0, 15, 0, 0xF}, "VBITREVW": {0xE631 << 15, false, 0, 31, 0, 0x1F}, "XVBITREVW": {0xEE31 << 15, true, 0, 31, 0, 0x1F}, "VBITREVV": {0xE632 << 15, false, 0, 63, 0, 0x3F}, "XVBITREVV": {0xEE32 << 15, true, 0, 63, 0, 0x3F}, // The 4-bit-select shuffles and the byte-extract/insert permutations // take ui8 (the .d shuffle ui4 range-checked to 0..15 by the // toolchain) packing both position nibbles. "VSHUF4IB": {0xE720 << 15, false, 0, 255, 0, 0xFF}, "XVSHUF4IB": {0xEF20 << 15, true, 0, 255, 0, 0xFF}, "VSHUF4IH": {0xE728 << 15, false, 0, 255, 0, 0xFF}, "XVSHUF4IH": {0xEF28 << 15, true, 0, 255, 0, 0xFF}, "VSHUF4IW": {0xE730 << 15, false, 0, 255, 0, 0xFF}, "XVSHUF4IW": {0xEF30 << 15, true, 0, 255, 0, 0xFF}, "VSHUF4IV": {0xE738 << 15, false, 0, 15, 0, 0xFF}, "XVSHUF4IV": {0xEF38 << 15, true, 0, 15, 0, 0xFF}, "VPERMIW": {0xE7C8 << 15, false, 0, 255, 0, 0xFF}, "XVPERMIW": {0xEFC8 << 15, true, 0, 255, 0, 0xFF}, "XVPERMIV": {0xEFD0 << 15, true, 0, 255, 0, 0xFF}, "XVPERMIQ": {0xEFD8 << 15, true, 0, 255, 0, 0xFF}, "VEXTRINSB": {0xE718 << 15, false, 0, 255, 0, 0xFF}, "XVEXTRINSB": {0xEF18 << 15, true, 0, 255, 0, 0xFF}, "VEXTRINSH": {0xE710 << 15, false, 0, 255, 0, 0xFF}, "XVEXTRINSH": {0xEF10 << 15, true, 0, 255, 0, 0xFF}, "VEXTRINSW": {0xE708 << 15, false, 0, 255, 0, 0xFF}, "XVEXTRINSW": {0xEF08 << 15, true, 0, 255, 0, 0xFF}, "VEXTRINSV": {0xE700 << 15, false, 0, 255, 0, 0xFF}, "XVEXTRINSV": {0xEF00 << 15, true, 0, 255, 0, 0xFF}, } for m, e := range vecImm { l64VecImmInfo[m] = e l64VecBank[m] = e.lasx } // Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*, // vsetallnez.*): the sub-op rides in the rk field. vecCf := map[string]uint32{ "VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10, "VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10, "VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10, "VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10, "VSETEQV": 0xE539<<15 | 6<<10, "XVSETEQV": 0xED39<<15 | 6<<10, "VSETANYEQH": 0xE539<<15 | 9<<10, "XVSETANYEQH": 0xED39<<15 | 9<<10, "VSETANYEQW": 0xE539<<15 | 10<<10, "XVSETANYEQW": 0xED39<<15 | 10<<10, "VSETALLNEB": 0xE539<<15 | 12<<10, "XVSETALLNEB": 0xED39<<15 | 12<<10, "VSETALLNEH": 0xE539<<15 | 13<<10, "XVSETALLNEH": 0xED39<<15 | 13<<10, "VSETALLNEW": 0xE539<<15 | 14<<10, "XVSETALLNEW": 0xED39<<15 | 14<<10, } for m, op := range vecCf { l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op} l64VecBank[m] = strings.HasPrefix(m, "XV") } // Lane popcount and the two-operand vector FP/unary spellings: INSTR vj, // vd (the 2R layout with the opcode extending over the unused vk field; // the low byte of each constant is the instruction's own sub-op). vec2r := map[string]l64Vec3Enc{ "VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true}, } // The rest of the lane popcounts, the vector negations and the vector FP // unary conversions (loong64enc1.s). vec2rMore := map[string]l64Vec3Enc{ "VPCNTB": {0x1CA708 << 10, false}, "VPCNTH": {0x1CA709 << 10, false}, "VPCNTW": {0x1CA70A << 10, false}, "VNEGB": {0x1CA70C << 10, false}, "VNEGH": {0x1CA70D << 10, false}, "VNEGW": {0x1CA70E << 10, false}, "VNEGV": {0x1CA70F << 10, false}, "VFCLASSF": {0x1CA735 << 10, false}, "VFCLASSD": {0x1CA736 << 10, false}, "VFSQRTF": {0x1CA739 << 10, false}, "VFSQRTD": {0x1CA73A << 10, false}, "VFRECIPF": {0x1CA73D << 10, false}, "VFRECIPD": {0x1CA73E << 10, false}, "VFRSQRTF": {0x1CA741 << 10, false}, "VFRSQRTD": {0x1CA742 << 10, false}, "VFRINTF": {0x1CA74D << 10, false}, "VFRINTD": {0x1CA74E << 10, false}, "VFRINTRMF": {0x1CA751 << 10, false}, "VFRINTRMD": {0x1CA752 << 10, false}, "VFRINTRPF": {0x1CA755 << 10, false}, "VFRINTRPD": {0x1CA756 << 10, false}, "VFRINTRZF": {0x1CA759 << 10, false}, "VFRINTRZD": {0x1CA75A << 10, false}, "VFRINTRNEF": {0x1CA75D << 10, false}, "VFRINTRNED": {0x1CA75E << 10, false}, "XVPCNTB": {0x1DA708 << 10, true}, "XVPCNTH": {0x1DA709 << 10, true}, "XVPCNTW": {0x1DA70A << 10, true}, "XVNEGB": {0x1DA70C << 10, true}, "XVNEGH": {0x1DA70D << 10, true}, "XVNEGW": {0x1DA70E << 10, true}, "XVNEGV": {0x1DA70F << 10, true}, "XVFCLASSF": {0x1DA735 << 10, true}, "XVFCLASSD": {0x1DA736 << 10, true}, "XVFSQRTF": {0x1DA739 << 10, true}, "XVFSQRTD": {0x1DA73A << 10, true}, "XVFRECIPF": {0x1DA73D << 10, true}, "XVFRECIPD": {0x1DA73E << 10, true}, "XVFRSQRTF": {0x1DA741 << 10, true}, "XVFRSQRTD": {0x1DA742 << 10, true}, "XVFRINTF": {0x1DA74D << 10, true}, "XVFRINTD": {0x1DA74E << 10, true}, "XVFRINTRMF": {0x1DA751 << 10, true}, "XVFRINTRMD": {0x1DA752 << 10, true}, "XVFRINTRPF": {0x1DA755 << 10, true}, "XVFRINTRPD": {0x1DA756 << 10, true}, "XVFRINTRZF": {0x1DA759 << 10, true}, "XVFRINTRZD": {0x1DA75A << 10, true}, "XVFRINTRNEF": {0x1DA75D << 10, true}, "XVFRINTRNED": {0x1DA75E << 10, true}, } maps.Copy(vec2r, vec2rMore) for m, e := range vec2r { l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op} l64VecBank[m] = e.lasx l64Vec2R[m] = true } // The four-register byte shuffle: INSTR va, vk, vj, vd (the operand the // table reads in each field position, va at bits [19:15]). vec4r := map[string]l64Vec3Enc{ "VSHUFB": {0x0D50 << 16, false}, "XVSHUFB": {0x0D60 << 16, true}, } for m, e := range vec4r { l64InstrTable[m] = l64Enc{format: l64Fvvvv, op: e.op} l64VecBank[m] = e.lasx l64Vec4R[m] = true } } // l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the // register move between the integer and floating-point register banks, the // MOVW/MOVV specials the Go assembler accepts. var l64FpMovTable = map[string]uint32{ "MOVV.R.F": 0x452a << 10, // movgr2fr.d "MOVV.R.FCC": 0x4536 << 10, // movgr2cf "MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr "MOVV.F.R": 0x452e << 10, // movfr2gr.d "MOVV.F.FCC": 0x4534 << 10, // movfr2cf "MOVV.FCC.R": 0x4537 << 10, // movcf2gr "MOVV.FCC.F": 0x4535 << 10, // movcf2fr "MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr "MOVW.R.F": 0x4529 << 10, // movgr2fr.w "MOVW.F.R": 0x452d << 10, // movfr2gr.s } // l64branchTable holds the 16-bit branch and jump encodings (2RI16). var l64branchTable = map[string]uint32{ "BEQ": 0x16 << 26, "BNE": 0x17 << 26, "BLT": 0x18 << 26, "BGE": 0x19 << 26, "BLTU": 0x1a << 26, "BGEU": 0x1b << 26, "JIRL": 0x13 << 26, } // l64branch21Table holds the single-register branches with 21-bit offsets: // the negative opcode constants the toolchain uses for the short forms. var l64branch21Table = map[string]uint32{ "BEQZ": 0x10 << 26, // beq r0, rj → beqz "BNEZ": 0x11 << 26, // bne r0, rj → bnez "BLTZ": 0x18 << 26, // blt rj, r0 → bltz "BGEZ": 0x19 << 26, // bge rj, r0 → bgez "BGTZ": 0x18 << 26, // blt r0, rj → bgtz "BLEZ": 0x19 << 26, // bge r0, rj → blez "BFPT": 0x12<<26 | 0x1<<8, "BFPF": 0x12<<26 | 0x0<<8, } // l64jumpTable maps the jump pseudo-instructions and their aliases to the // B/BL opcode constants. var l64jumpTable = map[string]uint32{ "JMP": 0x14 << 26, // b "B": 0x14 << 26, // b "JAL": 0x15 << 26, // bl "CALL": 0x15 << 26, // bl "BL": 0x15 << 26, // bl } // l64loadStoreTable maps the MOV width mnemonics to their load and store // 2RI12 opcodes. The load opcode is the negated store opcode, exactly as // the toolchain derives it. var l64loadStoreTable = map[string]struct{ ld, st uint32 }{ "MOVB": {0x0a0 << 22, 0x0a4 << 22}, "MOVH": {0x0a1 << 22, 0x0a5 << 22}, "MOVW": {0x0a2 << 22, 0x0a6 << 22}, "MOVV": {0x0a3 << 22, 0x0a7 << 22}, "MOVBU": {0x0a8 << 22, 0x0a4 << 22}, "MOVHU": {0x0a9 << 22, 0x0a5 << 22}, "MOVWU": {0x0aa << 22, 0x0a6 << 22}, "MOVF": {0x0ac << 22, 0x0ad << 22}, "MOVD": {0x0ae << 22, 0x0af << 22}, } // l64movRegTable maps a register-to-register MOV mnemonic to its expansion, // matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h, // MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d // and are handled separately in the assembler. type l64MovRegEnc struct { rr bool // 2R format (ext.w.b/ext.w.h) op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode imm int // 2RI12 immediate for MOVBU's andi } var l64movRegTable = map[string]l64MovRegEnc{ "MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj "MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj "MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0 "MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0 "MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff } // l64movFpRegTable maps a floating-point register move mnemonic to its 2R // opcode (fmov.s / fmov.d), used when both operands are F registers. var l64movFpRegTable = map[string]uint32{ "MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10, } // l64RegClass discriminates integer (R), floating-point (F) and condition // (FCC) registers for the MOV pseudo-instruction's register-move encoding. type l64RegClass int const ( l64ClsNone l64RegClass = iota l64ClsGR l64ClsFP l64ClsFCC l64ClsFCSR ) // loong64RegClass reports the register class of a register operand name. func loong64RegClass(name string) l64RegClass { switch { case name == "": return l64ClsNone case len(name) >= 3 && name[:3] == "FCC": return l64ClsFCC case len(name) >= 4 && name[:4] == "FCSR": return l64ClsFCSR case name[0] == 'F': return l64ClsFP default: return l64ClsGR } }