1200 lines
54 KiB
Go
1200 lines
54 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
// loong64 (LoongArch) instruction encoding.
|
|
//
|
|
// The encoder is data-driven: each mnemonic maps to an instruction format and
|
|
// an opcode constant, and the format selects the bit layout. The opcode
|
|
// constants and formats are transcribed from the Go toolchain's own loong64
|
|
// backend (cmd/internal/obj/loong64), so the emitted bytes match `go tool asm`
|
|
// exactly, the ground-truth oracle for the verify suite.
|
|
//
|
|
// All LoongArch instructions are 32 bits, little-endian. The formats used
|
|
// here (per the LoongArch Volume I specification):
|
|
//
|
|
// 3R opcode[31:15] | rk[4:0] | rj[4:0] | rd[4:0]
|
|
// 2R opcode[31:15] | rj[4:0] | rd[4:0]
|
|
// 2RI12 opcode[31:22] | si12[11:0] | rj[4:0] | rd[4:0]
|
|
// 2RI14 opcode[31:18] | si14[13:0] | rj[4:0] | rd[4:0]
|
|
// 2RI16 opcode[31:22] | si16[15:0] | rj[4:0] | rd[4:0]
|
|
// 2RI20 opcode[31:25] | si20[19:0] | rd[4:0]
|
|
// 1RI21 opcode[31:26] | si21[20:0] | rj[4:0] (BEQZ/BNEZ, B*Z, BC*Z)
|
|
// B/BL opcode[31:26] | offs[25:0]
|
|
// 4R opcode[31:20] | r1[4:0] | r2[4:0] | r3[4:0] | r4[4:0]
|
|
// IRIR opcode[31:22] | msb[4:0] | rj[4:0] | lsb[4:0] | rd[4:0]
|
|
// 3RI2 opcode[31:17] | sa2[1:0] | rk[4:0] | rj[4:0] | rd[4:0]
|
|
//
|
|
// The opcode constants are pre-positioned (they include the zero bit ranges
|
|
// of the immediate and register fields), mirroring the toolchain's OP_*
|
|
// helpers, so each l64* function only ORs its fields in.
|
|
|
|
import (
|
|
"maps"
|
|
"strings"
|
|
)
|
|
|
|
// loong64RegNum returns the 5-bit register number for a LoongArch register
|
|
// name: R0-R31 (integer), F0-F31 (floating point), FCC0-FCC7 (condition
|
|
// flags), FCSR0-FCSR31 (control/status) and the ABI aliases the runtime's
|
|
// assembly uses. Returns -1 for an unrecognised name.
|
|
func loong64RegNum(name string) int {
|
|
switch name {
|
|
case "R0", "ZERO":
|
|
return 0
|
|
case "R1", "RA", "LINK":
|
|
return 1
|
|
case "R2", "TP":
|
|
return 2
|
|
case "R3", "SP":
|
|
return 3
|
|
case "R4", "A0":
|
|
return 4
|
|
case "R5", "A1":
|
|
return 5
|
|
case "R6", "A2":
|
|
return 6
|
|
case "R7", "A3":
|
|
return 7
|
|
case "R8", "A4":
|
|
return 8
|
|
case "R9", "A5":
|
|
return 9
|
|
case "R10", "A6":
|
|
return 10
|
|
case "R11", "A7":
|
|
return 11
|
|
case "R12", "T0":
|
|
return 12
|
|
case "R13", "T1":
|
|
return 13
|
|
case "R14", "T2":
|
|
return 14
|
|
case "R15", "T3":
|
|
return 15
|
|
case "R16", "T4":
|
|
return 16
|
|
case "R17", "T5":
|
|
return 17
|
|
case "R18", "T6":
|
|
return 18
|
|
case "R19", "T7":
|
|
return 19
|
|
case "R20", "T8":
|
|
return 20
|
|
case "R21":
|
|
return 21
|
|
case "R22", "G", "g", "FP":
|
|
return 22
|
|
case "R23", "S0":
|
|
return 23
|
|
case "R24", "S1":
|
|
return 24
|
|
case "R25", "S2":
|
|
return 25
|
|
case "R26", "S3":
|
|
return 26
|
|
case "R27", "S4":
|
|
return 27
|
|
case "R28", "S5":
|
|
return 28
|
|
case "R29", "S6", "CTXT":
|
|
return 29
|
|
case "R30", "S7", "TMP":
|
|
return 30
|
|
case "R31", "S8":
|
|
return 31
|
|
}
|
|
// F0-F31, FCC0-FCC7, FCSR0-FCSR31. The LSX/LASX vector banks (V0-V31,
|
|
// X0-X31) are deliberately NOT accepted here: they are a separate
|
|
// register class, and the toolchain rejects V/X names wherever an
|
|
// integer or FP register is expected (GOARCH=loong64 go tool asm reports
|
|
// "unrecognized instruction" for `BEQZ X0`). Vector operands are
|
|
// resolved only through loong64VecRegNum.
|
|
if len(name) >= 4 && name[:4] == "FCSR" {
|
|
return loong64RegSpecial(name[4:], 31)
|
|
}
|
|
if len(name) >= 3 && name[:3] == "FCC" {
|
|
return loong64RegSpecial(name[3:], 7)
|
|
}
|
|
if len(name) < 2 {
|
|
return -1
|
|
}
|
|
prefix, digits := name[:1], name[1:]
|
|
if digits[0] < '0' || digits[0] > '9' {
|
|
return -1
|
|
}
|
|
n := 0
|
|
for i := 0; i < len(digits); i++ {
|
|
if digits[i] < '0' || digits[i] > '9' {
|
|
return -1
|
|
}
|
|
n = n*10 + int(digits[i]-'0')
|
|
}
|
|
if prefix == "F" && n <= 31 {
|
|
return n
|
|
}
|
|
return -1
|
|
}
|
|
|
|
// loong64RegSpecial parses a numbered FCC/FCSR register.
|
|
func loong64RegSpecial(digits string, max int) int {
|
|
if digits == "" {
|
|
return -1
|
|
}
|
|
n := 0
|
|
for i := 0; i < len(digits); i++ {
|
|
if digits[i] < '0' || digits[i] > '9' {
|
|
return -1
|
|
}
|
|
n = n*10 + int(digits[i]-'0')
|
|
}
|
|
if n <= max {
|
|
return n
|
|
}
|
|
return -1
|
|
}
|
|
|
|
// loong64VecRegNum resolves an LSX/LASX vector register name (V0-V31 or
|
|
// X0-X31) to its 5-bit number, or -1. The vector banks are a register class
|
|
// of their own: the toolchain accepts them only in the vector operands of the
|
|
// LSX/LASX instructions (GOARCH=loong64 go tool asm assembles `VADDV V0, V1,
|
|
// V2` and `XVADDV X0, X1, X2`, and rejects `VADDV R4, R5, R6`), so the V/X
|
|
// spellings never reach the integer/FP resolver.
|
|
func loong64VecRegNum(name string) int {
|
|
if len(name) < 2 || (name[0] != 'V' && name[0] != 'X') {
|
|
return -1
|
|
}
|
|
return loong64RegSpecial(name[1:], 31)
|
|
}
|
|
|
|
// ---- format helpers ----
|
|
|
|
// l64rrr encodes a 3R instruction: op | rk<<10 | rj<<5 | rd.
|
|
func l64rrr(op uint32, rk, rj, rd int) uint32 {
|
|
return op | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64rr encodes a 2R instruction: op | rj<<5 | rd.
|
|
func l64rr(op uint32, rj, rd int) uint32 {
|
|
return op | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64irr encodes a 2RI12 instruction: op | si12<<10 | rj<<5 | rd.
|
|
func l64irr(op uint32, imm, rj, rd int) uint32 {
|
|
return op | (uint32(imm)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64irr14 encodes a 2RI14 instruction: op | si14<<10 | rj<<5 | rd.
|
|
func l64irr14(op uint32, imm, rj, rd int) uint32 {
|
|
return op | (uint32(imm)&0x3FFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64irr16 encodes a 2RI16 instruction: op | si16<<10 | rj<<5 | rd.
|
|
func l64irr16(op uint32, imm, rj, rd int) uint32 {
|
|
return op | (uint32(imm)&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64ir encodes a 2RI20 instruction: op | si20<<5 | rd.
|
|
func l64ir(op uint32, imm, rd int) uint32 {
|
|
return op | (uint32(imm)&0xFFFFF)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64bbl encodes a B/BL instruction: op | offs[25:0], where offs is the
|
|
// 4-byte-aligned word distance (the toolchain stores the shifted value).
|
|
func l64bbl(op uint32, offs int) uint32 {
|
|
return op | (uint32(offs)&0xFFFF)<<10 | (uint32(offs)>>16)&0x3FF
|
|
}
|
|
|
|
// l64ir21 encodes a 1RI21 branch (BEQZ/BNEZ, BLTZ/BGEZ/BLEZ/BGTZ, BFPT/BFPF):
|
|
// op | si21[15:0]<<10 | rj<<5 | si21[20:16].
|
|
func l64ir21(op uint32, offs, rj int) uint32 {
|
|
v := uint32(offs)
|
|
return op | (v&0xFFFF)<<10 | uint32(rj&0x1f)<<5 | (v>>16)&0x1F
|
|
}
|
|
|
|
// l64rrrr encodes a 4R instruction: op | r1<<15 | r2<<10 | r3<<5 | r4.
|
|
func l64rrrr(op uint32, r1, r2, r3, r4 int) uint32 {
|
|
return op | uint32(r1&0x1f)<<15 | uint32(r2&0x1f)<<10 | uint32(r3&0x1f)<<5 | uint32(r4&0x1f)
|
|
}
|
|
|
|
// l64irir encodes a BSTRINS/BSTRPICK instruction: op | msb<<16 | rj<<5 | lsb<<10 | rd.
|
|
// The msb/lsb fields are 6 bits wide and are inserted unmasked: the caller
|
|
// must have validated them (0..31 for the .w forms, 0..63 for the .d forms,
|
|
// lsb <= msb), the same rule the toolchain enforces as "illegal bit number".
|
|
func l64irir(op uint32, msb, rj, lsb, rd int) uint32 {
|
|
return op | uint32(msb)<<16 | uint32(rj&0x1f)<<5 | uint32(lsb)<<10 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64irrr encodes a 3RI2 instruction (ALSL): op | sa<<15 | rk<<10 | rj<<5 | rd.
|
|
func l64irrr(op uint32, sa, rk, rj, rd int) uint32 {
|
|
return op | uint32(sa&0x3)<<15 | uint32(rk&0x1f)<<10 | uint32(rj&0x1f)<<5 | uint32(rd&0x1f)
|
|
}
|
|
|
|
// l64i15 encodes a no-operand system instruction with a 15-bit code field
|
|
// (SYSCALL, BREAK, DBAR): op | code[14:0].
|
|
func l64i15(op uint32, code int) uint32 {
|
|
return op | uint32(code)&0x7FFF
|
|
}
|
|
|
|
// l64irr5i encodes PRELD: op | offs<<10 | rj<<5 | hint.
|
|
func l64irr5i(op uint32, offs, rj, hint int) uint32 {
|
|
return op | (uint32(offs)&0xFFF)<<10 | uint32(rj&0x1f)<<5 | uint32(hint&0x1f)
|
|
}
|
|
|
|
// l64wordLE encodes a uint32 as 4 little-endian bytes.
|
|
func l64wordLE(w uint32) []byte {
|
|
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
|
}
|
|
|
|
// l64WordsLE concatenates one or more instruction words as little-endian bytes.
|
|
func l64WordsLE(ws ...uint32) []byte {
|
|
var out []byte
|
|
for _, w := range ws {
|
|
out = append(out, l64wordLE(w)...)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// ---- instruction formats ----
|
|
|
|
type l64Format uint8
|
|
|
|
const (
|
|
l64Frrr l64Format = iota // 3R (integer and FP arithmetic)
|
|
l64Frr // 2R
|
|
l64Firr // 2RI12 (arithmetic with 12-bit immediate)
|
|
l64Firr14 // 2RI14 (ldptr/stptr)
|
|
l64Firr16 // 2RI16 (addu16i.d)
|
|
l64Fir20 // 2RI20 (lu12i.w, lu32i.d, pcalau12i, pcaddu12i)
|
|
l64Frrrr // 4R (fmadd/fmsub/fnmadd/fnmsub, fsel)
|
|
l64Firir // bstrins/bstrpick
|
|
l64Firrr // alsl
|
|
l64Fi15 // syscall/break/dbar
|
|
l64Fam // atomic (3R with the AM field order)
|
|
l64Frdtime // rdtime (rd at bits [9:5], rj at bits [4:0])
|
|
l64Fshift // 2RI12 with a 5/6-bit shift immediate
|
|
l64Fpreld // preld (2RI12 + 5-bit hint)
|
|
l64Fvvv // 3R vector (LSX/LASX): op | vk<<10 | vj<<5 | vd
|
|
l64Fvcf // vector-to-condition: op | subop<<10 | vj<<5 | fcc
|
|
l64Fvvvv // 4R vector shuffle: op | va<<15 | vk<<10 | vj<<5 | vd
|
|
)
|
|
|
|
// l64Enc is one instruction's encoding: its bit layout (format) and the
|
|
// opcode constant, positioned at its exact bit range.
|
|
type l64Enc struct {
|
|
format l64Format
|
|
op uint32
|
|
}
|
|
|
|
// l64DualEnc holds both forms of a dual-form mnemonic: the 3R register form
|
|
// and the 2RI12 immediate form (which is a shift for the shift mnemonics).
|
|
type l64DualEnc struct {
|
|
rrr uint32 // 3R register form
|
|
imm uint32 // 2RI12 immediate form
|
|
shift bool // the immediate form is a 5/6-bit shift amount
|
|
}
|
|
|
|
// l64DualTable maps the dual-form arithmetic/logic mnemonics to both
|
|
// encodings; the assembler picks by operand kind.
|
|
var l64DualTable = map[string]l64DualEnc{}
|
|
|
|
// l64InstrTable maps LoongArch mnemonics (as the Go assembler spells them)
|
|
// to their encoding.
|
|
var l64InstrTable = map[string]l64Enc{}
|
|
|
|
// l64Vec3Enc pairs a vector opcode with its register bank: false = LSX
|
|
// (V0-V31), true = LASX (X0-X31). The toolchain accepts one bank per
|
|
// spelling: GOARCH=loong64 go tool asm assembles `VADDV V1, V2, V3` and
|
|
// `XVADDV X1, X2, X3`, and rejects the crossed spellings.
|
|
type l64Vec3Enc struct {
|
|
op uint32
|
|
lasx bool
|
|
}
|
|
|
|
// l64VecImmEnc carries the immediate-form encoding of a vector mnemonic:
|
|
// the opcode, the bank, the accepted immediate range, the bias the toolchain
|
|
// adds (vsrai.b encodes imm+8) and the mask of the encoded field (vseqi.b
|
|
// keeps a 5-bit two's-complement value, vseqi.d a 7-bit one).
|
|
type l64VecImmEnc struct {
|
|
op uint32
|
|
lasx bool
|
|
min, max int
|
|
bias int
|
|
mask int
|
|
}
|
|
|
|
// l64VecBank marks the LSX/LASX mnemonics and records which register bank
|
|
// each accepts; presence in the map routes the mnemonic through the vector
|
|
// dispatcher rather than the integer/FP formats.
|
|
var l64VecBank = map[string]bool{}
|
|
|
|
// l64VecImmInfo mirrors l64VecImmTable for the dispatcher.
|
|
var l64VecImmInfo = map[string]l64VecImmEnc{}
|
|
|
|
// l64Vec2R marks the two-operand vector mnemonics (INSTR vj, vd, such as
|
|
// vpcnt.v).
|
|
var l64Vec2R = map[string]bool{}
|
|
|
|
// l64Vec4R marks the four-operand vector mnemonics (INSTR va, vk, vj, vd,
|
|
// such as vshuf.b).
|
|
var l64Vec4R = map[string]bool{}
|
|
|
|
// l64VmovqOps holds the VMOVQ/XVMOVQ opcode constants (pre-shifted to bit
|
|
// 15), read off `go tool objdump` of GOARCH=loong64 `go tool asm` kernels.
|
|
type l64VmovqEnc struct {
|
|
ld, st, ldx, stx uint32 // plain and indexed load/store
|
|
replB, replH, replW, replD uint32 // vldrepl: load and replicate element
|
|
pickS, pickU uint32 // vpickve2gr.{,u} element extract
|
|
ins uint32 // vinsgr2vr element insert
|
|
dup uint32 // vreplgr2vr duplicate (width in [11:10])
|
|
move uint32 // vori.b/xvori.b $0 register move
|
|
}
|
|
|
|
var l64VmovqTable = map[bool]l64VmovqEnc{
|
|
false: { // VMOVQ, the LSX (V) bank
|
|
ld: 0x5800 << 15, st: 0x5880 << 15, ldx: 0x7080 << 15, stx: 0x7088 << 15,
|
|
replB: 0x6100 << 15, replH: 0x6080 << 15, replW: 0x6040 << 15, replD: 0x6020 << 15,
|
|
pickS: 0xE5DF << 15, pickU: 0xE5E7 << 15,
|
|
ins: 0xE5D7 << 15, dup: 0xE53E << 15, move: 0xE65A << 15,
|
|
},
|
|
true: { // XVMOVQ, the LASX (X) bank
|
|
ld: 0x5900 << 15, st: 0x5980 << 15, ldx: 0x7090 << 15, stx: 0x7098 << 15,
|
|
replB: 0x6500 << 15, replH: 0x6480 << 15, replW: 0x6440 << 15, replD: 0x6420 << 15,
|
|
pickS: 0xEDDF << 15, pickU: 0xEDE7 << 15,
|
|
ins: 0xEDD7 << 15, dup: 0xED3E << 15, move: 0xEE5A << 15,
|
|
},
|
|
}
|
|
|
|
func init() {
|
|
// 3R, integer.
|
|
rrr := map[string]uint32{
|
|
"ADD": 0x20 << 15, "ADDW": 0x20 << 15, "ADDV": 0x21 << 15, "ADDVU": 0x21 << 15,
|
|
"SUB": 0x22 << 15, "SUBW": 0x22 << 15, "SUBV": 0x23 << 15, "SUBVU": 0x23 << 15,
|
|
"SGT": 0x24 << 15, "SGTU": 0x25 << 15,
|
|
"MASKEQZ": 0x26 << 15, "MASKNEZ": 0x27 << 15, "SCQ": 0x070AE << 15,
|
|
"NOR": 0x28 << 15, "AND": 0x29 << 15, "OR": 0x2a << 15, "XOR": 0x2b << 15,
|
|
"ORN": 0x2c << 15, "ANDN": 0x2d << 15,
|
|
"SLL": 0x2e << 15, "SRL": 0x2f << 15, "SRA": 0x30 << 15,
|
|
"SLLV": 0x31 << 15, "SRLV": 0x32 << 15, "SRAV": 0x33 << 15,
|
|
"ROTR": 0x36 << 15, "ROTRV": 0x37 << 15,
|
|
"MUL": 0x38 << 15, "MULW": 0x38 << 15, "MULH": 0x39 << 15, "MULHU": 0x3a << 15,
|
|
"MULV": 0x3b << 15, "MULVU": 0x3b << 15, "MULHV": 0x3c << 15, "MULHVU": 0x3d << 15,
|
|
"MULWVW": 0x3e << 15, "MULWVWU": 0x3f << 15,
|
|
"DIV": 0x40 << 15, "DIVW": 0x40 << 15, "REM": 0x41 << 15, "REMW": 0x41 << 15,
|
|
"DIVU": 0x42 << 15, "DIVWU": 0x42 << 15, "REMU": 0x43 << 15, "REMWU": 0x43 << 15,
|
|
"DIVV": 0x44 << 15, "REMV": 0x45 << 15, "DIVVU": 0x46 << 15, "REMVU": 0x47 << 15,
|
|
"CRCWBW": 0x48 << 15, "CRCWHW": 0x49 << 15, "CRCWWW": 0x4a << 15, "CRCWVW": 0x4b << 15,
|
|
"CRCCWBW": 0x4c << 15, "CRCCWHW": 0x4d << 15, "CRCCWWW": 0x4e << 15, "CRCCWVW": 0x4f << 15,
|
|
}
|
|
// 3R, floating point.
|
|
rrr["MULF"] = 0x209 << 15
|
|
rrr["MULD"] = 0x20a << 15
|
|
rrr["DIVF"] = 0x20d << 15
|
|
rrr["DIVD"] = 0x20e << 15
|
|
rrr["SUBF"] = 0x205 << 15
|
|
rrr["SUBD"] = 0x206 << 15
|
|
rrr["ADDF"] = 0x201 << 15
|
|
rrr["ADDD"] = 0x202 << 15
|
|
rrr["CMPEQF"] = 0x0c1<<20 | 0x4<<15
|
|
rrr["CMPEQD"] = 0x0c2<<20 | 0x4<<15
|
|
rrr["CMPGED"] = 0x0c2<<20 | 0x7<<15
|
|
rrr["CMPGEF"] = 0x0c1<<20 | 0x7<<15
|
|
rrr["CMPGTD"] = 0x0c2<<20 | 0x3<<15
|
|
rrr["CMPGTF"] = 0x0c1<<20 | 0x3<<15
|
|
rrr["FMINF"] = 0x215 << 15
|
|
rrr["FMIND"] = 0x216 << 15
|
|
rrr["FMAXF"] = 0x211 << 15
|
|
rrr["FMAXD"] = 0x212 << 15
|
|
rrr["FMAXAF"] = 0x219 << 15
|
|
rrr["FMAXAD"] = 0x21a << 15
|
|
rrr["FMINAF"] = 0x21d << 15
|
|
rrr["FMINAD"] = 0x21e << 15
|
|
rrr["FSCALEBF"] = 0x221 << 15
|
|
rrr["FSCALEBD"] = 0x222 << 15
|
|
rrr["FCOPYSGF"] = 0x225 << 15
|
|
rrr["FCOPYSGD"] = 0x226 << 15
|
|
for m, op := range rrr {
|
|
l64InstrTable[m] = l64Enc{format: l64Frrr, op: op}
|
|
}
|
|
|
|
// 2R.
|
|
rr := map[string]uint32{
|
|
"CLOW": 0x4 << 10, "CLZW": 0x5 << 10, "CTOW": 0x6 << 10, "CTZW": 0x7 << 10,
|
|
"CLOV": 0x8 << 10, "CLZV": 0x9 << 10, "CTOV": 0xa << 10, "CTZV": 0xb << 10,
|
|
"REVB2H": 0xc << 10, "REVB4H": 0xd << 10, "REVB2W": 0xe << 10, "REVBV": 0xf << 10,
|
|
"REVH2W": 0x10 << 10, "REVHV": 0x11 << 10,
|
|
"BITREV4B": 0x12 << 10, "BITREV8B": 0x13 << 10, "BITREVW": 0x14 << 10, "BITREVV": 0x15 << 10,
|
|
"EXTWH": 0x16 << 10, "EXTWB": 0x17 << 10, "CPUCFG": 0x1b << 10,
|
|
"TRUNCFV": 0x46a9 << 10, "TRUNCDV": 0x46aa << 10, "TRUNCFW": 0x46a1 << 10, "TRUNCDW": 0x46a2 << 10,
|
|
"MOVWF": 0x4744 << 10, "MOVVF": 0x4746 << 10, "MOVWD": 0x4748 << 10, "MOVVD": 0x474a << 10,
|
|
"MOVFW": 0x46c1 << 10, "MOVDW": 0x46c2 << 10, "MOVFV": 0x46c9 << 10, "MOVDV": 0x46ca << 10,
|
|
"FRINTF": 0x4791 << 10, "FRINTD": 0x4792 << 10,
|
|
"MOVDF": 0x4646 << 10, "MOVFD": 0x4649 << 10,
|
|
"ABSF": 0x4501 << 10, "ABSD": 0x4502 << 10,
|
|
"MOVF": 0x4525 << 10, "MOVD": 0x4526 << 10,
|
|
"NEGF": 0x4505 << 10, "NEGD": 0x4506 << 10,
|
|
"SQRTF": 0x4511 << 10, "SQRTD": 0x4512 << 10,
|
|
"FLOGBF": 0x4509 << 10, "FLOGBD": 0x450a << 10,
|
|
"FCLASSF": 0x450d << 10, "FCLASSD": 0x450e << 10,
|
|
"FTINTRMWF": 0x4681 << 10, "FTINTRMWD": 0x4682 << 10,
|
|
"FTINTRMVF": 0x4689 << 10, "FTINTRMVD": 0x468a << 10,
|
|
"FTINTRPWF": 0x4691 << 10, "FTINTRPWD": 0x4692 << 10,
|
|
"FTINTRPVF": 0x4699 << 10, "FTINTRPVD": 0x469a << 10,
|
|
"FTINTRZWF": 0x46a1 << 10, "FTINTRZWD": 0x46a2 << 10,
|
|
"FTINTRZVF": 0x46a9 << 10, "FTINTRZVD": 0x46aa << 10,
|
|
"FTINTRNEWF": 0x46b1 << 10, "FTINTRNEWD": 0x46b2 << 10,
|
|
"FTINTRNEVF": 0x46b9 << 10, "FTINTRNEVD": 0x46ba << 10,
|
|
// LSX: convert a 64-bit integer lane to a double float. The operand
|
|
// bank is the FP registers (the toolchain spells it `FFINTDV F0, F1`),
|
|
// so the entry stays on the 2R integer/FP format.
|
|
"FFINTDV": 0x474a << 10,
|
|
// The rest of the scalar conversions (all F-bank, 2R).
|
|
"FFINTFW": 0x4744 << 10, // ffint.s.w
|
|
"FFINTFV": 0x4746 << 10, // ffint.s.l
|
|
"FFINTDW": 0x4748 << 10, // ffint.d.w
|
|
"FTINTWF": 0x46c1 << 10, // ftint.w.s
|
|
"FTINTWD": 0x46c2 << 10, // ftint.w.d
|
|
"FTINTVF": 0x46c9 << 10, // ftint.l.s
|
|
"FTINTVD": 0x46ca << 10, // ftint.l.d
|
|
}
|
|
for m, op := range rr {
|
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: op}
|
|
}
|
|
// RDTIME is a 2R instruction with rd and rj in swapped positions.
|
|
l64InstrTable["RDTIMELW"] = l64Enc{format: l64Frdtime, op: 0x18 << 10}
|
|
l64InstrTable["RDTIMEHW"] = l64Enc{format: l64Frdtime, op: 0x19 << 10}
|
|
l64InstrTable["RDTIMED"] = l64Enc{format: l64Frdtime, op: 0x1a << 10}
|
|
|
|
// The dual-form arithmetic mnemonics (register 3R + immediate 2RI12),
|
|
// selected by the operand kind; the shift mnemonics pair the 3R form
|
|
// with a 5/6-bit shift immediate.
|
|
maps.Copy(l64DualTable, map[string]l64DualEnc{
|
|
"ADD": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
|
"ADDW": {rrr: 0x20 << 15, imm: 0x00a << 22},
|
|
"ADDV": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
|
"ADDVU": {rrr: 0x21 << 15, imm: 0x00b << 22},
|
|
"AND": {rrr: 0x29 << 15, imm: 0x00d << 22},
|
|
"OR": {rrr: 0x2a << 15, imm: 0x00e << 22},
|
|
"XOR": {rrr: 0x2b << 15, imm: 0x00f << 22},
|
|
"SGT": {rrr: 0x24 << 15, imm: 0x008 << 22},
|
|
"SGTU": {rrr: 0x25 << 15, imm: 0x009 << 22},
|
|
"SLL": {rrr: 0x2e << 15, imm: 0x00081 << 15, shift: true},
|
|
"SRL": {rrr: 0x2f << 15, imm: 0x00089 << 15, shift: true},
|
|
"SRA": {rrr: 0x30 << 15, imm: 0x00091 << 15, shift: true},
|
|
"ROTR": {rrr: 0x36 << 15, imm: 0x00099 << 15, shift: true},
|
|
"SLLV": {rrr: 0x31 << 15, imm: 0x0041 << 16, shift: true},
|
|
"SRLV": {rrr: 0x32 << 15, imm: 0x0045 << 16, shift: true},
|
|
"SRAV": {rrr: 0x33 << 15, imm: 0x0049 << 16, shift: true},
|
|
"ROTRV": {rrr: 0x37 << 15, imm: 0x004d << 16, shift: true},
|
|
})
|
|
|
|
// 2RI12, pure immediate arithmetic (LU52ID has no register form).
|
|
l64InstrTable["LU52ID"] = l64Enc{format: l64Firr, op: 0x00c << 22}
|
|
// ADDV16 (addu16i.d): 2RI16 with the immediate shifted right by 16.
|
|
l64InstrTable["ADDV16"] = l64Enc{format: l64Firr16, op: 0x4 << 26}
|
|
|
|
// 2RI14, LL/SC are aliased by the Go assembler to the pointer loads and
|
|
// stores (ldptr/stptr), with the offset scaled by 4.
|
|
l64InstrTable["MOVWP"] = l64Enc{format: l64Firr14, op: 0x25 << 24} // stptr.w
|
|
l64InstrTable["MOVVP"] = l64Enc{format: l64Firr14, op: 0x27 << 24} // stptr.d
|
|
l64InstrTable["SC"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
|
|
l64InstrTable["SCW"] = l64Enc{format: l64Firr14, op: 0x21 << 24} // sc.w
|
|
l64InstrTable["SCV"] = l64Enc{format: l64Firr14, op: 0x23 << 24} // sc.d
|
|
l64InstrTable["LL"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
|
|
l64InstrTable["LLW"] = l64Enc{format: l64Firr14, op: 0x20 << 24} // ldptr.w (ll.w)
|
|
l64InstrTable["LLV"] = l64Enc{format: l64Firr14, op: 0x22 << 24} // ldptr.d (ll.d)
|
|
|
|
// 2RI20.
|
|
l64InstrTable["LU12IW"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
|
l64InstrTable["LU32ID"] = l64Enc{format: l64Fir20, op: 0x0b << 25}
|
|
l64InstrTable["PCALAU12I"] = l64Enc{format: l64Fir20, op: 0x0d << 25}
|
|
l64InstrTable["PCADDU12I"] = l64Enc{format: l64Fir20, op: 0x0e << 25}
|
|
// LUI is the Plan 9 spelling of lu12i.w.
|
|
l64InstrTable["LUI"] = l64Enc{format: l64Fir20, op: 0x0a << 25}
|
|
|
|
// 4R, fused multiply-add, and FSEL (fsel.d: the first operand is a FCC
|
|
// condition flag, the layout matches the 4R shape).
|
|
rrrr := map[string]uint32{
|
|
"FMADDF": 0x81 << 20, "FMADDD": 0x82 << 20,
|
|
"FMSUBF": 0x85 << 20, "FMSUBD": 0x86 << 20,
|
|
"FNMADDF": 0x89 << 20, "FNMADDD": 0x8a << 20,
|
|
"FNMSUBF": 0x8d << 20, "FNMSUBD": 0x8e << 20,
|
|
"FSEL": 0x340 << 18,
|
|
}
|
|
for m, op := range rrrr {
|
|
l64InstrTable[m] = l64Enc{format: l64Frrrr, op: op}
|
|
}
|
|
|
|
// IRIR, bit-field insert/extract.
|
|
irir := map[string]uint32{
|
|
"BSTRINSW": 0x3<<21 | 0x0<<15,
|
|
"BSTRINSV": 0x2 << 22,
|
|
"BSTRPICKW": 0x3<<21 | 0x1<<15,
|
|
"BSTRPICKV": 0x3 << 22,
|
|
}
|
|
for m, op := range irir {
|
|
l64InstrTable[m] = l64Enc{format: l64Firir, op: op}
|
|
}
|
|
|
|
// 3RI2, ALSL.
|
|
irrr := map[string]uint32{
|
|
"ALSLW": 0x2 << 17, "ALSLWU": 0x3 << 17, "ALSLV": 0x16 << 17,
|
|
}
|
|
for m, op := range irrr {
|
|
l64InstrTable[m] = l64Enc{format: l64Firrr, op: op}
|
|
}
|
|
|
|
// 0-operand system instructions.
|
|
l64InstrTable["SYSCALL"] = l64Enc{format: l64Fi15, op: 0x56 << 15}
|
|
l64InstrTable["BREAK"] = l64Enc{format: l64Fi15, op: 0x54 << 15}
|
|
l64InstrTable["DBAR"] = l64Enc{format: l64Fi15, op: 0x70e4 << 15}
|
|
|
|
// PRELD.
|
|
l64InstrTable["PRELD"] = l64Enc{format: l64Fpreld, op: 0x0ab << 22}
|
|
|
|
// Atomics, 3R with the AM field order (rk=value, rj=address, rd=result).
|
|
// The toolchain's form is three operands, `AMADDW rk, (rj), rd`
|
|
// (cmd/asm/internal/asm/testdata/loong64enc1.s and
|
|
// internal/runtime/atomic/atomic_loong64.s); the two-register spelling
|
|
// is rejected by the oracle.
|
|
am := map[string]uint32{
|
|
"AMSWAPB": 0x070B8 << 15, "AMSWAPH": 0x070B9 << 15,
|
|
"AMSWAPW": 0x070C0 << 15, "AMSWAPV": 0x070C1 << 15,
|
|
"AMCASB": 0x070B0 << 15, "AMCASH": 0x070B1 << 15,
|
|
"AMCASW": 0x070B2 << 15, "AMCASV": 0x070B3 << 15,
|
|
"AMADDW": 0x070C2 << 15, "AMADDV": 0x070C3 << 15,
|
|
"AMANDW": 0x070C4 << 15, "AMANDV": 0x070C5 << 15,
|
|
"AMORW": 0x070C6 << 15, "AMORV": 0x070C7 << 15,
|
|
"AMXORW": 0x070C8 << 15, "AMXORV": 0x070C9 << 15,
|
|
"AMMAXW": 0x070CA << 15, "AMMAXV": 0x070CB << 15,
|
|
"AMMINW": 0x070CC << 15, "AMMINV": 0x070CD << 15,
|
|
"AMMAXWU": 0x070CE << 15, "AMMAXVU": 0x070CF << 15,
|
|
"AMMINWU": 0x070D0 << 15, "AMMINVU": 0x070D1 << 15,
|
|
"AMSWAPDBB": 0x070BC << 15, "AMSWAPDBH": 0x070BD << 15,
|
|
"AMSWAPDBW": 0x070D2 << 15, "AMSWAPDBV": 0x070D3 << 15,
|
|
"AMCASDBB": 0x070B4 << 15, "AMCASDBH": 0x070B5 << 15,
|
|
"AMCASDBW": 0x070B6 << 15, "AMCASDBV": 0x070B7 << 15,
|
|
// The _dbar (acquire/release) add, and, or variants: opcodes read off
|
|
// `go tool objdump` of `AMADDDBW R14, (R13), R12` and friends.
|
|
"AMADDDBW": 0x070D4 << 15, "AMADDDBV": 0x070D5 << 15,
|
|
"AMANDDBW": 0x070D6 << 15, "AMANDDBV": 0x070D7 << 15,
|
|
"AMORDBW": 0x070D8 << 15, "AMORDBV": 0x070D9 << 15,
|
|
// The remaining _dbar exchange variants (loong64enc1.s).
|
|
"AMXORDBW": 0x070DA << 15, "AMXORDBV": 0x070DB << 15,
|
|
"AMMAXDBW": 0x070DC << 15, "AMMAXDBV": 0x070DD << 15,
|
|
"AMMINDBW": 0x070DE << 15, "AMMINDBV": 0x070DF << 15,
|
|
"AMMAXDBWU": 0x070E0 << 15, "AMMAXDBVU": 0x070E1 << 15,
|
|
"AMMINDBWU": 0x070E2 << 15, "AMMINDBVU": 0x070E3 << 15,
|
|
}
|
|
for m, op := range am {
|
|
l64InstrTable[m] = l64Enc{format: l64Fam, op: op}
|
|
}
|
|
|
|
// ---- LSX/LASX (V*/XV*) ----
|
|
// Every opcode below was read off `go tool objdump` of a GOARCH=loong64
|
|
// `go tool asm` kernel (the toolchain's own loong64enc1.s cross-checks
|
|
// most of them), not assumed from the LoongArch manual.
|
|
|
|
// Three vector registers: INSTR vk, vj, vd (or INSTR vk, vd with
|
|
// vj = vd). l64Vec3Enc.lasx selects the register bank the toolchain
|
|
// accepts: LSX spellings take V0-V31, LASX spellings X0-X31.
|
|
vec3 := map[string]l64Vec3Enc{
|
|
"VADDW": {0xE016 << 15, false}, "VADDV": {0xE017 << 15, false},
|
|
"VANDV": {0xE24C << 15, false}, "VXORV": {0xE24E << 15, false},
|
|
"VSEQB": {0xE000 << 15, false}, "VSEQV": {0xE003 << 15, false},
|
|
"VSRAB": {0xE1D8 << 15, false}, "VROTRW": {0xE1DE << 15, false},
|
|
"XVADDV": {0xE817 << 15, true},
|
|
"XVANDV": {0xEA4C << 15, true}, "XVXORV": {0xEA4E << 15, true},
|
|
"XVSEQB": {0xE800 << 15, true}, "XVSEQV": {0xE803 << 15, true},
|
|
}
|
|
|
|
// The integer and FP add/subtract families: [X]VADD and [X]VSUB by lane
|
|
// width, plus the [X]VSADD/[X]VSSUB saturating pairs.
|
|
// Opcodes transcribed from the toolchain's loong64enc1.s.
|
|
addsub := map[string]l64Vec3Enc{
|
|
"VADDB": {0xE014 << 15, false}, "VADDH": {0xE015 << 15, false},
|
|
"VADDD": {0xE262 << 15, false}, "VADDF": {0xE261 << 15, false},
|
|
"VADDQ": {0xE25A << 15, false},
|
|
"VSUBB": {0xE018 << 15, false}, "VSUBH": {0xE019 << 15, false},
|
|
"VSUBW": {0xE01A << 15, false}, "VSUBV": {0xE01B << 15, false},
|
|
"VSUBQ": {0xE25B << 15, false},
|
|
"VSUBF": {0xE265 << 15, false}, "VSUBD": {0xE266 << 15, false},
|
|
"VSADDB": {0xE08C << 15, false}, "VSADDH": {0xE08D << 15, false},
|
|
"VSADDW": {0xE08E << 15, false}, "VSADDV": {0xE08F << 15, false},
|
|
"VSADDBU": {0xE094 << 15, false}, "VSADDHU": {0xE095 << 15, false},
|
|
"VSADDWU": {0xE096 << 15, false}, "VSADDVU": {0xE097 << 15, false},
|
|
"VSSUBB": {0xE090 << 15, false}, "VSSUBH": {0xE091 << 15, false},
|
|
"VSSUBW": {0xE092 << 15, false}, "VSSUBV": {0xE093 << 15, false},
|
|
"VSSUBBU": {0xE098 << 15, false}, "VSSUBHU": {0xE099 << 15, false},
|
|
"VSSUBWU": {0xE09A << 15, false}, "VSSUBVU": {0xE09B << 15, false},
|
|
"XVADDB": {0xE814 << 15, true}, "XVADDH": {0xE815 << 15, true},
|
|
"XVADDW": {0xE816 << 15, true},
|
|
"XVADDD": {0xEA62 << 15, true}, "XVADDF": {0xEA61 << 15, true},
|
|
"XVADDQ": {0xEA5A << 15, true},
|
|
"XVSUBB": {0xE818 << 15, true}, "XVSUBH": {0xE819 << 15, true},
|
|
"XVSUBW": {0xE81A << 15, true}, "XVSUBV": {0xE81B << 15, true},
|
|
"XVSUBQ": {0xEA5B << 15, true},
|
|
"XVSUBF": {0xEA65 << 15, true}, "XVSUBD": {0xEA66 << 15, true},
|
|
"XVSADDB": {0xE88C << 15, true}, "XVSADDH": {0xE88D << 15, true},
|
|
"XVSADDW": {0xE88E << 15, true}, "XVSADDV": {0xE88F << 15, true},
|
|
"XVSADDBU": {0xE894 << 15, true}, "XVSADDHU": {0xE895 << 15, true},
|
|
"XVSADDWU": {0xE896 << 15, true}, "XVSADDVU": {0xE897 << 15, true},
|
|
"XVSSUBB": {0xE890 << 15, true}, "XVSSUBH": {0xE891 << 15, true},
|
|
"XVSSUBW": {0xE892 << 15, true}, "XVSSUBV": {0xE893 << 15, true},
|
|
"XVSSUBBU": {0xE898 << 15, true}, "XVSSUBHU": {0xE899 << 15, true},
|
|
"XVSSUBWU": {0xE89A << 15, true}, "XVSSUBVU": {0xE89B << 15, true},
|
|
}
|
|
|
|
// The multiply families: plain and high-half [X]VMUL/[X]VMUH, the
|
|
// widening [X]VMULW{EV,OD} ladder and its accumulating [X]VMADDW twins,
|
|
// plus the [X]VMADD/[X]VMSUB fused multiply-add and the [X]VDIV/[X]VMOD
|
|
// divide and modulo pairs.
|
|
muldiv := map[string]l64Vec3Enc{
|
|
"VMULB": {0xE108 << 15, false}, "VMULH": {0xE109 << 15, false},
|
|
"VMULW": {0xE10A << 15, false}, "VMULV": {0xE10B << 15, false},
|
|
"VMUHB": {0xE10C << 15, false}, "VMUHH": {0xE10D << 15, false},
|
|
"VMUHW": {0xE10E << 15, false}, "VMUHV": {0xE10F << 15, false},
|
|
"VMUHBU": {0xE110 << 15, false}, "VMUHHU": {0xE111 << 15, false},
|
|
"VMUHWU": {0xE112 << 15, false}, "VMUHVU": {0xE113 << 15, false},
|
|
"VMULWEVHB": {0xE120 << 15, false}, "VMULWEVWH": {0xE121 << 15, false},
|
|
"VMULWEVVW": {0xE122 << 15, false}, "VMULWEVQV": {0xE123 << 15, false},
|
|
"VMULWODHB": {0xE124 << 15, false}, "VMULWODWH": {0xE125 << 15, false},
|
|
"VMULWODVW": {0xE126 << 15, false}, "VMULWODQV": {0xE127 << 15, false},
|
|
"VMULWEVHBU": {0xE130 << 15, false}, "VMULWEVWHU": {0xE131 << 15, false},
|
|
"VMULWEVVWU": {0xE132 << 15, false}, "VMULWEVQVU": {0xE133 << 15, false},
|
|
"VMULWODHBU": {0xE134 << 15, false}, "VMULWODWHU": {0xE135 << 15, false},
|
|
"VMULWODVWU": {0xE136 << 15, false}, "VMULWODQVU": {0xE137 << 15, false},
|
|
"VMULWEVHBUB": {0xE140 << 15, false}, "VMULWEVWHUH": {0xE141 << 15, false},
|
|
"VMULWEVVWUW": {0xE142 << 15, false}, "VMULWEVQVUV": {0xE143 << 15, false},
|
|
"VMULWODHBUB": {0xE144 << 15, false}, "VMULWODWHUH": {0xE145 << 15, false},
|
|
"VMULWODVWUW": {0xE146 << 15, false}, "VMULWODQVUV": {0xE147 << 15, false},
|
|
"VMADDB": {0xE150 << 15, false}, "VMADDH": {0xE151 << 15, false},
|
|
"VMADDW": {0xE152 << 15, false}, "VMADDV": {0xE153 << 15, false},
|
|
"VMSUBB": {0xE154 << 15, false}, "VMSUBH": {0xE155 << 15, false},
|
|
"VMSUBW": {0xE156 << 15, false}, "VMSUBV": {0xE157 << 15, false},
|
|
"VMADDWEVHB": {0xE158 << 15, false}, "VMADDWEVWH": {0xE159 << 15, false},
|
|
"VMADDWEVVW": {0xE15A << 15, false}, "VMADDWEVQV": {0xE15B << 15, false},
|
|
"VMADDWODHB": {0xE15C << 15, false}, "VMADDWODWH": {0xE15D << 15, false},
|
|
"VMADDWODVW": {0xE15E << 15, false}, "VMADDWODQV": {0xE15F << 15, false},
|
|
"VMADDWEVHBU": {0xE168 << 15, false}, "VMADDWEVWHU": {0xE169 << 15, false},
|
|
"VMADDWEVVWU": {0xE16A << 15, false}, "VMADDWEVQVU": {0xE16B << 15, false},
|
|
"VMADDWODHBU": {0xE16C << 15, false}, "VMADDWODWHU": {0xE16D << 15, false},
|
|
"VMADDWODVWU": {0xE16E << 15, false}, "VMADDWODQVU": {0xE16F << 15, false},
|
|
"VMADDWEVHBUB": {0xE178 << 15, false}, "VMADDWEVWHUH": {0xE179 << 15, false},
|
|
"VMADDWEVVWUW": {0xE17A << 15, false}, "VMADDWEVQVUV": {0xE17B << 15, false},
|
|
"VMADDWODHBUB": {0xE17C << 15, false}, "VMADDWODWHUH": {0xE17D << 15, false},
|
|
"VMADDWODVWUW": {0xE17E << 15, false}, "VMADDWODQVUV": {0xE17F << 15, false},
|
|
"VDIVB": {0xE1C0 << 15, false}, "VDIVH": {0xE1C1 << 15, false},
|
|
"VDIVW": {0xE1C2 << 15, false}, "VDIVV": {0xE1C3 << 15, false},
|
|
"VMODB": {0xE1C4 << 15, false}, "VMODH": {0xE1C5 << 15, false},
|
|
"VMODW": {0xE1C6 << 15, false}, "VMODV": {0xE1C7 << 15, false},
|
|
"VDIVBU": {0xE1C8 << 15, false}, "VDIVHU": {0xE1C9 << 15, false},
|
|
"VDIVWU": {0xE1CA << 15, false}, "VDIVVU": {0xE1CB << 15, false},
|
|
"VMODBU": {0xE1CC << 15, false}, "VMODHU": {0xE1CD << 15, false},
|
|
"VMODWU": {0xE1CE << 15, false}, "VMODVU": {0xE1CF << 15, false},
|
|
"VMULF": {0xE271 << 15, false}, "VMULD": {0xE272 << 15, false},
|
|
"VDIVF": {0xE275 << 15, false}, "VDIVD": {0xE276 << 15, false},
|
|
"XVMULB": {0xE908 << 15, true}, "XVMULH": {0xE909 << 15, true},
|
|
"XVMULW": {0xE90A << 15, true}, "XVMULV": {0xE90B << 15, true},
|
|
"XVMUHB": {0xE90C << 15, true}, "XVMUHH": {0xE90D << 15, true},
|
|
"XVMUHW": {0xE90E << 15, true}, "XVMUHV": {0xE90F << 15, true},
|
|
"XVMUHBU": {0xE910 << 15, true}, "XVMUHHU": {0xE911 << 15, true},
|
|
"XVMUHWU": {0xE912 << 15, true}, "XVMUHVU": {0xE913 << 15, true},
|
|
"XVMULWEVHB": {0xE920 << 15, true}, "XVMULWEVWH": {0xE921 << 15, true},
|
|
"XVMULWEVVW": {0xE922 << 15, true}, "XVMULWEVQV": {0xE923 << 15, true},
|
|
"XVMULWODHB": {0xE924 << 15, true}, "XVMULWODWH": {0xE925 << 15, true},
|
|
"XVMULWODVW": {0xE926 << 15, true}, "XVMULWODQV": {0xE927 << 15, true},
|
|
"XVMULWEVHBU": {0xE930 << 15, true}, "XVMULWEVWHU": {0xE931 << 15, true},
|
|
"XVMULWEVVWU": {0xE932 << 15, true}, "XVMULWEVQVU": {0xE933 << 15, true},
|
|
"XVMULWODHBU": {0xE934 << 15, true}, "XVMULWODWHU": {0xE935 << 15, true},
|
|
"XVMULWODVWU": {0xE936 << 15, true}, "XVMULWODQVU": {0xE937 << 15, true},
|
|
"XVMULWEVHBUB": {0xE940 << 15, true}, "XVMULWEVWHUH": {0xE941 << 15, true},
|
|
"XVMULWEVVWUW": {0xE942 << 15, true}, "XVMULWEVQVUV": {0xE943 << 15, true},
|
|
"XVMULWODHBUB": {0xE944 << 15, true}, "XVMULWODWHUH": {0xE945 << 15, true},
|
|
"XVMULWODVWUW": {0xE946 << 15, true}, "XVMULWODQVUV": {0xE947 << 15, true},
|
|
"XVMADDB": {0xE950 << 15, true}, "XVMADDH": {0xE951 << 15, true},
|
|
"XVMADDW": {0xE952 << 15, true}, "XVMADDV": {0xE953 << 15, true},
|
|
"XVMSUBB": {0xE954 << 15, true}, "XVMSUBH": {0xE955 << 15, true},
|
|
"XVMSUBW": {0xE956 << 15, true}, "XVMSUBV": {0xE957 << 15, true},
|
|
"XVMADDWEVHB": {0xE958 << 15, true}, "XVMADDWEVWH": {0xE959 << 15, true},
|
|
"XVMADDWEVVW": {0xE95A << 15, true}, "XVMADDWEVQV": {0xE95B << 15, true},
|
|
"XVMADDWODHB": {0xE95C << 15, true}, "XVMADDWODWH": {0xE95D << 15, true},
|
|
"XVMADDWODVW": {0xE95E << 15, true}, "XVMADDWODQV": {0xE95F << 15, true},
|
|
"XVMADDWEVHBU": {0xE968 << 15, true}, "XVMADDWEVWHU": {0xE969 << 15, true},
|
|
"XVMADDWEVVWU": {0xE96A << 15, true}, "XVMADDWEVQVU": {0xE96B << 15, true},
|
|
"XVMADDWODHBU": {0xE96C << 15, true}, "XVMADDWODWHU": {0xE96D << 15, true},
|
|
"XVMADDWODVWU": {0xE96E << 15, true}, "XVMADDWODQVU": {0xE96F << 15, true},
|
|
"XVMADDWEVHBUB": {0xE978 << 15, true}, "XVMADDWEVWHUH": {0xE979 << 15, true},
|
|
"XVMADDWEVVWUW": {0xE97A << 15, true}, "XVMADDWEVQVUV": {0xE97B << 15, true},
|
|
"XVMADDWODHBUB": {0xE97C << 15, true}, "XVMADDWODWHUH": {0xE97D << 15, true},
|
|
"XVMADDWODVWUW": {0xE97E << 15, true}, "XVMADDWODQVUV": {0xE97F << 15, true},
|
|
"XVDIVB": {0xE9C0 << 15, true}, "XVDIVH": {0xE9C1 << 15, true},
|
|
"XVDIVW": {0xE9C2 << 15, true}, "XVDIVV": {0xE9C3 << 15, true},
|
|
"XVMODB": {0xE9C4 << 15, true}, "XVMODH": {0xE9C5 << 15, true},
|
|
"XVMODW": {0xE9C6 << 15, true}, "XVMODV": {0xE9C7 << 15, true},
|
|
"XVDIVBU": {0xE9C8 << 15, true}, "XVDIVHU": {0xE9C9 << 15, true},
|
|
"XVDIVWU": {0xE9CA << 15, true}, "XVDIVVU": {0xE9CB << 15, true},
|
|
"XVMODBU": {0xE9CC << 15, true}, "XVMODHU": {0xE9CD << 15, true},
|
|
"XVMODWU": {0xE9CE << 15, true}, "XVMODVU": {0xE9CF << 15, true},
|
|
"XVMULF": {0xEA71 << 15, true}, "XVMULD": {0xEA72 << 15, true},
|
|
"XVDIVF": {0xEA75 << 15, true}, "XVDIVD": {0xEA76 << 15, true},
|
|
}
|
|
|
|
// The lane-wise shifts and rotates (three-register forms; the immediate
|
|
// forms live in l64VecImmInfo), the interleave families, the bit
|
|
// clear/set/rev register forms, the remaining logic and compare
|
|
// spellings, the widening add/subtract ladder and the vector FP
|
|
// arithmetic.
|
|
vecmisc := map[string]l64Vec3Enc{
|
|
"VSLLB": {0xE1D0 << 15, false}, "VSLLH": {0xE1D1 << 15, false},
|
|
"VSLLW": {0xE1D2 << 15, false}, "VSLLV": {0xE1D3 << 15, false},
|
|
"VSRLB": {0xE1D4 << 15, false}, "VSRLH": {0xE1D5 << 15, false},
|
|
"VSRLW": {0xE1D6 << 15, false}, "VSRLV": {0xE1D7 << 15, false},
|
|
"VSRAH": {0xE1D9 << 15, false}, "VSRAW": {0xE1DA << 15, false},
|
|
"VSRAV": {0xE1DB << 15, false},
|
|
"VROTRB": {0xE1DC << 15, false}, "VROTRH": {0xE1DD << 15, false},
|
|
"VROTRV": {0xE1DF << 15, false},
|
|
"VILVLB": {0xE234 << 15, false}, "VILVLH": {0xE235 << 15, false},
|
|
"VILVLW": {0xE236 << 15, false}, "VILVLV": {0xE237 << 15, false},
|
|
"VILVHB": {0xE238 << 15, false}, "VILVHH": {0xE239 << 15, false},
|
|
"VILVHW": {0xE23A << 15, false}, "VILVHV": {0xE23B << 15, false},
|
|
"VBITCLRB": {0xE218 << 15, false}, "VBITCLRH": {0xE219 << 15, false},
|
|
"VBITCLRW": {0xE21A << 15, false}, "VBITCLRV": {0xE21B << 15, false},
|
|
"VBITSETB": {0xE21C << 15, false}, "VBITSETH": {0xE21D << 15, false},
|
|
"VBITSETW": {0xE21E << 15, false}, "VBITSETV": {0xE21F << 15, false},
|
|
"VBITREVB": {0xE220 << 15, false}, "VBITREVH": {0xE221 << 15, false},
|
|
"VBITREVW": {0xE222 << 15, false}, "VBITREVV": {0xE223 << 15, false},
|
|
"VORV": {0xE24D << 15, false}, "VNORV": {0xE24F << 15, false},
|
|
"VANDNV": {0xE250 << 15, false}, "VORNV": {0xE251 << 15, false},
|
|
"VSEQH": {0xE001 << 15, false}, "VSEQW": {0xE002 << 15, false},
|
|
"VSLTB": {0xE00C << 15, false}, "VSLTH": {0xE00D << 15, false},
|
|
"VSLTW": {0xE00E << 15, false}, "VSLTV": {0xE00F << 15, false},
|
|
"VSLTBU": {0xE010 << 15, false}, "VSLTHU": {0xE011 << 15, false},
|
|
"VSLTWU": {0xE012 << 15, false}, "VSLTVU": {0xE013 << 15, false},
|
|
"VADDWEVHB": {0xE03C << 15, false}, "VADDWEVWH": {0xE03D << 15, false},
|
|
"VADDWEVVW": {0xE03E << 15, false}, "VADDWEVQV": {0xE03F << 15, false},
|
|
"VSUBWEVHB": {0xE040 << 15, false}, "VSUBWEVWH": {0xE041 << 15, false},
|
|
"VSUBWEVVW": {0xE042 << 15, false}, "VSUBWEVQV": {0xE043 << 15, false},
|
|
"VADDWODHB": {0xE044 << 15, false}, "VADDWODWH": {0xE045 << 15, false},
|
|
"VADDWODVW": {0xE046 << 15, false}, "VADDWODQV": {0xE047 << 15, false},
|
|
"VSUBWODHB": {0xE048 << 15, false}, "VSUBWODWH": {0xE049 << 15, false},
|
|
"VSUBWODVW": {0xE04A << 15, false}, "VSUBWODQV": {0xE04B << 15, false},
|
|
"VSUBWEVHBU": {0xE060 << 15, false}, "VSUBWEVWHU": {0xE061 << 15, false},
|
|
"VSUBWEVVWU": {0xE062 << 15, false}, "VSUBWEVQVU": {0xE063 << 15, false},
|
|
"VADDWEVHBU": {0xE05C << 15, false}, "VADDWEVWHU": {0xE05D << 15, false},
|
|
"VADDWEVVWU": {0xE05E << 15, false}, "VADDWEVQVU": {0xE05F << 15, false},
|
|
"VADDWODHBU": {0xE064 << 15, false}, "VADDWODWHU": {0xE065 << 15, false},
|
|
"VADDWODVWU": {0xE066 << 15, false}, "VADDWODQVU": {0xE067 << 15, false},
|
|
"VSUBWODHBU": {0xE068 << 15, false}, "VSUBWODWHU": {0xE069 << 15, false},
|
|
"VSUBWODVWU": {0xE06A << 15, false}, "VSUBWODQVU": {0xE06B << 15, false},
|
|
"VSHUFH": {0xE2F5 << 15, false}, "VSHUFW": {0xE2F6 << 15, false},
|
|
"VSHUFV": {0xE2F7 << 15, false},
|
|
"XVSLLB": {0xE9D0 << 15, true}, "XVSLLH": {0xE9D1 << 15, true},
|
|
"XVSLLW": {0xE9D2 << 15, true}, "XVSLLV": {0xE9D3 << 15, true},
|
|
"XVSRLB": {0xE9D4 << 15, true}, "XVSRLH": {0xE9D5 << 15, true},
|
|
"XVSRLW": {0xE9D6 << 15, true}, "XVSRLV": {0xE9D7 << 15, true},
|
|
"XVSRAB": {0xE9D8 << 15, true}, "XVSRAH": {0xE9D9 << 15, true},
|
|
"XVSRAW": {0xE9DA << 15, true}, "XVSRAV": {0xE9DB << 15, true},
|
|
"XVROTRB": {0xE9DC << 15, true}, "XVROTRH": {0xE9DD << 15, true},
|
|
"XVROTRW": {0xE9DE << 15, true}, "XVROTRV": {0xE9DF << 15, true},
|
|
"XVILVLB": {0xEA34 << 15, true}, "XVILVLH": {0xEA35 << 15, true},
|
|
"XVILVLW": {0xEA36 << 15, true}, "XVILVLV": {0xEA37 << 15, true},
|
|
"XVILVHB": {0xEA38 << 15, true}, "XVILVHH": {0xEA39 << 15, true},
|
|
"XVILVHW": {0xEA3A << 15, true}, "XVILVHV": {0xEA3B << 15, true},
|
|
"XVBITCLRB": {0xEA18 << 15, true}, "XVBITCLRH": {0xEA19 << 15, true},
|
|
"XVBITCLRW": {0xEA1A << 15, true}, "XVBITCLRV": {0xEA1B << 15, true},
|
|
"XVBITSETB": {0xEA1C << 15, true}, "XVBITSETH": {0xEA1D << 15, true},
|
|
"XVBITSETW": {0xEA1E << 15, true}, "XVBITSETV": {0xEA1F << 15, true},
|
|
"XVBITREVB": {0xEA20 << 15, true}, "XVBITREVH": {0xEA21 << 15, true},
|
|
"XVBITREVW": {0xEA22 << 15, true}, "XVBITREVV": {0xEA23 << 15, true},
|
|
"XVORV": {0xEA4D << 15, true}, "XVNORV": {0xEA4F << 15, true},
|
|
"XVANDNV": {0xEA50 << 15, true}, "XVORNV": {0xEA51 << 15, true},
|
|
"XVSEQH": {0xE801 << 15, true}, "XVSEQW": {0xE802 << 15, true},
|
|
"XVSLTB": {0xE80C << 15, true}, "XVSLTH": {0xE80D << 15, true},
|
|
"XVSLTW": {0xE80E << 15, true}, "XVSLTV": {0xE80F << 15, true},
|
|
"XVSLTBU": {0xE810 << 15, true}, "XVSLTHU": {0xE811 << 15, true},
|
|
"XVSLTWU": {0xE812 << 15, true}, "XVSLTVU": {0xE813 << 15, true},
|
|
"XVADDWEVHB": {0xE83C << 15, true}, "XVADDWEVWH": {0xE83D << 15, true},
|
|
"XVADDWEVVW": {0xE83E << 15, true}, "XVADDWEVQV": {0xE83F << 15, true},
|
|
"XVSUBWEVHB": {0xE840 << 15, true}, "XVSUBWEVWH": {0xE841 << 15, true},
|
|
"XVSUBWEVVW": {0xE842 << 15, true}, "XVSUBWEVQV": {0xE843 << 15, true},
|
|
"XVADDWODHB": {0xE844 << 15, true}, "XVADDWODWH": {0xE845 << 15, true},
|
|
"XVADDWODVW": {0xE846 << 15, true}, "XVADDWODQV": {0xE847 << 15, true},
|
|
"XVSUBWODHB": {0xE848 << 15, true}, "XVSUBWODWH": {0xE849 << 15, true},
|
|
"XVSUBWODVW": {0xE84A << 15, true}, "XVSUBWODQV": {0xE84B << 15, true},
|
|
"XVADDWEVHBU": {0xE85C << 15, true}, "XVADDWEVWHU": {0xE85D << 15, true},
|
|
"XVADDWEVVWU": {0xE85E << 15, true}, "XVADDWEVQVU": {0xE85F << 15, true},
|
|
"XVSUBWEVHBU": {0xE860 << 15, true}, "XVSUBWEVWHU": {0xE861 << 15, true},
|
|
"XVSUBWEVVWU": {0xE862 << 15, true}, "XVSUBWEVQVU": {0xE863 << 15, true},
|
|
"XVADDWODHBU": {0xE864 << 15, true}, "XVADDWODWHU": {0xE865 << 15, true},
|
|
"XVADDWODVWU": {0xE866 << 15, true}, "XVADDWODQVU": {0xE867 << 15, true},
|
|
"XVSUBWODHBU": {0xE868 << 15, true}, "XVSUBWODWHU": {0xE869 << 15, true},
|
|
"XVSUBWODVWU": {0xE86A << 15, true}, "XVSUBWODQVU": {0xE86B << 15, true},
|
|
"XVSHUFH": {0xEAF5 << 15, true}, "XVSHUFW": {0xEAF6 << 15, true},
|
|
"XVSHUFV": {0xEAF7 << 15, true},
|
|
}
|
|
for _, tab := range []map[string]l64Vec3Enc{addsub, muldiv, vecmisc} {
|
|
for m, e := range tab {
|
|
if _, dup := vec3[m]; dup {
|
|
panic("loong64: duplicate vector mnemonic " + m)
|
|
}
|
|
vec3[m] = e
|
|
}
|
|
}
|
|
for m, e := range vec3 {
|
|
l64InstrTable[m] = l64Enc{format: l64Fvvv, op: e.op}
|
|
l64VecBank[m] = e.lasx
|
|
}
|
|
|
|
// Immediate forms: INSTR $imm, vj, vd (or INSTR $imm, vd). The immediate
|
|
// range, bias and field mask are the ones the toolchain encodes: vandi.b
|
|
// stores the raw 8-bit constant, vsrari.b stores imm+8 (lane-width
|
|
// bias), the si5 compares store 5-bit two's-complement values and vseqi.d
|
|
// a 7-bit field the toolchain range-checks down to si5.
|
|
// The mnemonics that also have a register form (the shifts, the bit
|
|
// clear/set/rev families, VSEQ and the logic immediates) keep their
|
|
// three-register entry in l64InstrTable; the dispatcher picks the
|
|
// immediate opcode from l64VecImmInfo by operand kind, so the immediate
|
|
// entries must not overwrite the table.
|
|
vecImm := map[string]l64VecImmEnc{
|
|
"VANDB": {0xE7A0 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVANDB": {0xEFA0 << 15, true, 0, 255, 0, 0xFF},
|
|
"VORB": {0xE7A8 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVORB": {0xEFA8 << 15, true, 0, 255, 0, 0xFF},
|
|
"VXORB": {0xE7B0 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVXORB": {0xEFB0 << 15, true, 0, 255, 0, 0xFF},
|
|
"VNORB": {0xE7B8 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVNORB": {0xEFB8 << 15, true, 0, 255, 0, 0xFF},
|
|
"VSEQB": {0xE500 << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSEQB": {0xE900 << 15, true, -16, 15, 0, 0x1F},
|
|
// vseqi.h/w accept the same si5 window as vseqi.b; vseqi.d carries a
|
|
// 7-bit field, but the toolchain range-checks it down to si5 as well
|
|
// (GOARCH=loong64 go tool asm rejects VSEQV $32 and VSEQV $-64).
|
|
"VSEQH": {0xE501 << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSEQH": {0xED01 << 15, true, -16, 15, 0, 0x1F},
|
|
"VSEQW": {0xE502 << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSEQW": {0xED02 << 15, true, -16, 15, 0, 0x1F},
|
|
"VSEQV": {0xE503 << 15, false, -16, 15, 0, 0x7F},
|
|
"XVSEQV": {0xE903 << 15, true, -16, 15, 0, 0x7F},
|
|
// vslti compares against a signed (or, in the U spellings, unsigned)
|
|
// si5/ui5 constant.
|
|
"VSLTB": {0xE50C << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSLTB": {0xED0C << 15, true, -16, 15, 0, 0x1F},
|
|
"VSLTH": {0xE50D << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSLTH": {0xED0D << 15, true, -16, 15, 0, 0x1F},
|
|
"VSLTW": {0xE50E << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSLTW": {0xED0E << 15, true, -16, 15, 0, 0x1F},
|
|
"VSLTV": {0xE50F << 15, false, -16, 15, 0, 0x1F},
|
|
"XVSLTV": {0xED0F << 15, true, -16, 15, 0, 0x1F},
|
|
"VSLTBU": {0xE510 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSLTBU": {0xED10 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSLTHU": {0xE511 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSLTHU": {0xED11 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSLTWU": {0xE512 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSLTWU": {0xED12 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSLTVU": {0xE513 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSLTVU": {0xED13 << 15, true, 0, 31, 0, 0x1F},
|
|
// vaddi/vsubi take ui5 constants for every width on this toolchain
|
|
// (VADDVU $32 is rejected by the oracle although the field is ui8).
|
|
"VADDBU": {0xE514 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVADDBU": {0xED14 << 15, true, 0, 31, 0, 0x1F},
|
|
"VADDHU": {0xE515 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVADDHU": {0xED15 << 15, true, 0, 31, 0, 0x1F},
|
|
"VADDWU": {0xE516 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVADDWU": {0xED16 << 15, true, 0, 31, 0, 0x1F},
|
|
"VADDVU": {0xE517 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVADDVU": {0xED17 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSUBBU": {0xE518 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSUBBU": {0xED18 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSUBHU": {0xE519 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSUBHU": {0xED19 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSUBWU": {0xE51A << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSUBWU": {0xED1A << 15, true, 0, 31, 0, 0x1F},
|
|
"VSUBVU": {0xE51B << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSUBVU": {0xED1B << 15, true, 0, 31, 0, 0x1F},
|
|
// The shift/rotate immediates ride in a width-sized field whose upper
|
|
// bits carry the lane-width code: vslli.b stores ui3 at [12:0] with
|
|
// bits [14:13] inside the opcode, vslli.h ui4 under a 4 bit mask, and
|
|
// the .w/.d spellings a raw ui5/ui6.
|
|
"VSLLB": {0x732C2000, false, 0, 7, 0, 0x7},
|
|
"XVSLLB": {0x772C2000, true, 0, 7, 0, 0x7},
|
|
"VSLLH": {0x732C4000, false, 0, 15, 0, 0xF},
|
|
"XVSLLH": {0x772C4000, true, 0, 15, 0, 0xF},
|
|
"VSLLW": {0xE659 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSLLW": {0xEE59 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSLLV": {0xE65A << 15, false, 0, 63, 0, 0x3F},
|
|
"XVSLLV": {0xEE5A << 15, true, 0, 63, 0, 0x3F},
|
|
"VSRLB": {0x73302000, false, 0, 7, 0, 0x7},
|
|
"XVSRLB": {0x77302000, true, 0, 7, 0, 0x7},
|
|
"VSRLH": {0x73304000, false, 0, 15, 0, 0xF},
|
|
"XVSRLH": {0x77304000, true, 0, 15, 0, 0xF},
|
|
"VSRLW": {0xE661 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSRLW": {0xEE61 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSRLV": {0xE662 << 15, false, 0, 63, 0, 0x3F},
|
|
"XVSRLV": {0xEE62 << 15, true, 0, 63, 0, 0x3F},
|
|
// vsrari/vrotri bias the field so the lane-width code rides above the
|
|
// shift amount (.b adds 8, .h 16, .w 32; .d is a raw ui6).
|
|
"VSRAB": {0xE668 << 15, false, 0, 7, 8, 0x1F},
|
|
"XVSRAB": {0xEE68 << 15, true, 0, 7, 8, 0x1F},
|
|
"VSRAH": {0x73344000, false, 0, 15, 0, 0xF},
|
|
"XVSRAH": {0x77344000, true, 0, 15, 0, 0xF},
|
|
"VSRAW": {0xE669 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVSRAW": {0xEE69 << 15, true, 0, 31, 0, 0x1F},
|
|
"VSRAV": {0xE66A << 15, false, 0, 63, 0, 0x3F},
|
|
"XVSRAV": {0xEE6A << 15, true, 0, 63, 0, 0x3F},
|
|
"VROTRB": {0x72A02000, false, 0, 7, 0, 0x7},
|
|
"XVROTRB": {0x76A02000, true, 0, 7, 0, 0x7},
|
|
"VROTRH": {0x72A04000, false, 0, 15, 0, 0xF},
|
|
"XVROTRH": {0x76A04000, true, 0, 15, 0, 0xF},
|
|
"VROTRW": {0xE541 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVROTRW": {0xED41 << 15, true, 0, 31, 0, 0x1F},
|
|
"VROTRV": {0xE542 << 15, false, 0, 63, 0, 0x3F},
|
|
"XVROTRV": {0xED42 << 15, true, 0, 63, 0, 0x3F},
|
|
// vbitclri/vbitseti/vbitrevi follow the same width-coded layout.
|
|
"VBITCLRB": {0x73102000, false, 0, 7, 0, 0x7},
|
|
"XVBITCLRB": {0x77102000, true, 0, 7, 0, 0x7},
|
|
"VBITCLRH": {0x73104000, false, 0, 15, 0, 0xF},
|
|
"XVBITCLRH": {0x77104000, true, 0, 15, 0, 0xF},
|
|
"VBITCLRW": {0xE621 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVBITCLRW": {0xEE21 << 15, true, 0, 31, 0, 0x1F},
|
|
"VBITCLRV": {0xE622 << 15, false, 0, 63, 0, 0x3F},
|
|
"XVBITCLRV": {0xEE22 << 15, true, 0, 63, 0, 0x3F},
|
|
"VBITSETB": {0x73142000, false, 0, 7, 0, 0x7},
|
|
"XVBITSETB": {0x77142000, true, 0, 7, 0, 0x7},
|
|
"VBITSETH": {0x73144000, false, 0, 15, 0, 0xF},
|
|
"XVBITSETH": {0x77144000, true, 0, 15, 0, 0xF},
|
|
"VBITSETW": {0xE629 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVBITSETW": {0xEE29 << 15, true, 0, 31, 0, 0x1F},
|
|
"VBITSETV": {0xE62A << 15, false, 0, 63, 0, 0x3F},
|
|
"XVBITSETV": {0xEE2A << 15, true, 0, 63, 0, 0x3F},
|
|
"VBITREVB": {0x73182000, false, 0, 7, 0, 0x7},
|
|
"XVBITREVB": {0x77182000, true, 0, 7, 0, 0x7},
|
|
"VBITREVH": {0x73184000, false, 0, 15, 0, 0xF},
|
|
"XVBITREVH": {0x77184000, true, 0, 15, 0, 0xF},
|
|
"VBITREVW": {0xE631 << 15, false, 0, 31, 0, 0x1F},
|
|
"XVBITREVW": {0xEE31 << 15, true, 0, 31, 0, 0x1F},
|
|
"VBITREVV": {0xE632 << 15, false, 0, 63, 0, 0x3F},
|
|
"XVBITREVV": {0xEE32 << 15, true, 0, 63, 0, 0x3F},
|
|
// The 4-bit-select shuffles and the byte-extract/insert permutations
|
|
// take ui8 (the .d shuffle ui4 range-checked to 0..15 by the
|
|
// toolchain) packing both position nibbles.
|
|
"VSHUF4IB": {0xE720 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVSHUF4IB": {0xEF20 << 15, true, 0, 255, 0, 0xFF},
|
|
"VSHUF4IH": {0xE728 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVSHUF4IH": {0xEF28 << 15, true, 0, 255, 0, 0xFF},
|
|
"VSHUF4IW": {0xE730 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVSHUF4IW": {0xEF30 << 15, true, 0, 255, 0, 0xFF},
|
|
"VSHUF4IV": {0xE738 << 15, false, 0, 15, 0, 0xFF},
|
|
"XVSHUF4IV": {0xEF38 << 15, true, 0, 15, 0, 0xFF},
|
|
"VPERMIW": {0xE7C8 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVPERMIW": {0xEFC8 << 15, true, 0, 255, 0, 0xFF},
|
|
"XVPERMIV": {0xEFD0 << 15, true, 0, 255, 0, 0xFF},
|
|
"XVPERMIQ": {0xEFD8 << 15, true, 0, 255, 0, 0xFF},
|
|
"VEXTRINSB": {0xE718 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVEXTRINSB": {0xEF18 << 15, true, 0, 255, 0, 0xFF},
|
|
"VEXTRINSH": {0xE710 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVEXTRINSH": {0xEF10 << 15, true, 0, 255, 0, 0xFF},
|
|
"VEXTRINSW": {0xE708 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVEXTRINSW": {0xEF08 << 15, true, 0, 255, 0, 0xFF},
|
|
"VEXTRINSV": {0xE700 << 15, false, 0, 255, 0, 0xFF},
|
|
"XVEXTRINSV": {0xEF00 << 15, true, 0, 255, 0, 0xFF},
|
|
}
|
|
for m, e := range vecImm {
|
|
l64VecImmInfo[m] = e
|
|
l64VecBank[m] = e.lasx
|
|
}
|
|
|
|
// Vector-to-condition flag: INSTR vj, FCCn (vsetnez.v, vsetanyeqz.*,
|
|
// vsetallnez.*): the sub-op rides in the rk field.
|
|
vecCf := map[string]uint32{
|
|
"VSETNEV": 0xE539<<15 | 7<<10, "XVSETNEV": 0xED39<<15 | 7<<10,
|
|
"VSETANYEQB": 0xE539<<15 | 8<<10, "XVSETANYEQB": 0xED39<<15 | 8<<10,
|
|
"VSETANYEQV": 0xE539<<15 | 11<<10, "XVSETANYEQV": 0xED39<<15 | 11<<10,
|
|
"VSETALLNEV": 0xE539<<15 | 15<<10, "XVSETALLNEV": 0xED39<<15 | 15<<10,
|
|
"VSETEQV": 0xE539<<15 | 6<<10, "XVSETEQV": 0xED39<<15 | 6<<10,
|
|
"VSETANYEQH": 0xE539<<15 | 9<<10, "XVSETANYEQH": 0xED39<<15 | 9<<10,
|
|
"VSETANYEQW": 0xE539<<15 | 10<<10, "XVSETANYEQW": 0xED39<<15 | 10<<10,
|
|
"VSETALLNEB": 0xE539<<15 | 12<<10, "XVSETALLNEB": 0xED39<<15 | 12<<10,
|
|
"VSETALLNEH": 0xE539<<15 | 13<<10, "XVSETALLNEH": 0xED39<<15 | 13<<10,
|
|
"VSETALLNEW": 0xE539<<15 | 14<<10, "XVSETALLNEW": 0xED39<<15 | 14<<10,
|
|
}
|
|
for m, op := range vecCf {
|
|
l64InstrTable[m] = l64Enc{format: l64Fvcf, op: op}
|
|
l64VecBank[m] = strings.HasPrefix(m, "XV")
|
|
}
|
|
|
|
// Lane popcount and the two-operand vector FP/unary spellings: INSTR vj,
|
|
// vd (the 2R layout with the opcode extending over the unused vk field;
|
|
// the low byte of each constant is the instruction's own sub-op).
|
|
vec2r := map[string]l64Vec3Enc{
|
|
"VPCNTV": {0x1CA70B << 10, false}, "XVPCNTV": {0x1DA70B << 10, true},
|
|
}
|
|
// The rest of the lane popcounts, the vector negations and the vector FP
|
|
// unary conversions (loong64enc1.s).
|
|
vec2rMore := map[string]l64Vec3Enc{
|
|
"VPCNTB": {0x1CA708 << 10, false}, "VPCNTH": {0x1CA709 << 10, false},
|
|
"VPCNTW": {0x1CA70A << 10, false},
|
|
"VNEGB": {0x1CA70C << 10, false}, "VNEGH": {0x1CA70D << 10, false},
|
|
"VNEGW": {0x1CA70E << 10, false}, "VNEGV": {0x1CA70F << 10, false},
|
|
"VFCLASSF": {0x1CA735 << 10, false}, "VFCLASSD": {0x1CA736 << 10, false},
|
|
"VFSQRTF": {0x1CA739 << 10, false}, "VFSQRTD": {0x1CA73A << 10, false},
|
|
"VFRECIPF": {0x1CA73D << 10, false}, "VFRECIPD": {0x1CA73E << 10, false},
|
|
"VFRSQRTF": {0x1CA741 << 10, false}, "VFRSQRTD": {0x1CA742 << 10, false},
|
|
"VFRINTF": {0x1CA74D << 10, false}, "VFRINTD": {0x1CA74E << 10, false},
|
|
"VFRINTRMF": {0x1CA751 << 10, false}, "VFRINTRMD": {0x1CA752 << 10, false},
|
|
"VFRINTRPF": {0x1CA755 << 10, false}, "VFRINTRPD": {0x1CA756 << 10, false},
|
|
"VFRINTRZF": {0x1CA759 << 10, false}, "VFRINTRZD": {0x1CA75A << 10, false},
|
|
"VFRINTRNEF": {0x1CA75D << 10, false}, "VFRINTRNED": {0x1CA75E << 10, false},
|
|
"XVPCNTB": {0x1DA708 << 10, true}, "XVPCNTH": {0x1DA709 << 10, true},
|
|
"XVPCNTW": {0x1DA70A << 10, true},
|
|
"XVNEGB": {0x1DA70C << 10, true}, "XVNEGH": {0x1DA70D << 10, true},
|
|
"XVNEGW": {0x1DA70E << 10, true}, "XVNEGV": {0x1DA70F << 10, true},
|
|
"XVFCLASSF": {0x1DA735 << 10, true}, "XVFCLASSD": {0x1DA736 << 10, true},
|
|
"XVFSQRTF": {0x1DA739 << 10, true}, "XVFSQRTD": {0x1DA73A << 10, true},
|
|
"XVFRECIPF": {0x1DA73D << 10, true}, "XVFRECIPD": {0x1DA73E << 10, true},
|
|
"XVFRSQRTF": {0x1DA741 << 10, true}, "XVFRSQRTD": {0x1DA742 << 10, true},
|
|
"XVFRINTF": {0x1DA74D << 10, true}, "XVFRINTD": {0x1DA74E << 10, true},
|
|
"XVFRINTRMF": {0x1DA751 << 10, true}, "XVFRINTRMD": {0x1DA752 << 10, true},
|
|
"XVFRINTRPF": {0x1DA755 << 10, true}, "XVFRINTRPD": {0x1DA756 << 10, true},
|
|
"XVFRINTRZF": {0x1DA759 << 10, true}, "XVFRINTRZD": {0x1DA75A << 10, true},
|
|
"XVFRINTRNEF": {0x1DA75D << 10, true}, "XVFRINTRNED": {0x1DA75E << 10, true},
|
|
}
|
|
maps.Copy(vec2r, vec2rMore)
|
|
for m, e := range vec2r {
|
|
l64InstrTable[m] = l64Enc{format: l64Frr, op: e.op}
|
|
l64VecBank[m] = e.lasx
|
|
l64Vec2R[m] = true
|
|
}
|
|
|
|
// The four-register byte shuffle: INSTR va, vk, vj, vd (the operand the
|
|
// table reads in each field position, va at bits [19:15]).
|
|
vec4r := map[string]l64Vec3Enc{
|
|
"VSHUFB": {0x0D50 << 16, false}, "XVSHUFB": {0x0D60 << 16, true},
|
|
}
|
|
for m, e := range vec4r {
|
|
l64InstrTable[m] = l64Enc{format: l64Fvvvv, op: e.op}
|
|
l64VecBank[m] = e.lasx
|
|
l64Vec4R[m] = true
|
|
}
|
|
}
|
|
|
|
// l64FpMovTable maps (mnemonic, from-class, to-class) to the 2R opcode of the
|
|
// register move between the integer and floating-point register banks, the
|
|
// MOVW/MOVV specials the Go assembler accepts.
|
|
var l64FpMovTable = map[string]uint32{
|
|
"MOVV.R.F": 0x452a << 10, // movgr2fr.d
|
|
"MOVV.R.FCC": 0x4536 << 10, // movgr2cf
|
|
"MOVV.R.FCSR": 0x4530 << 10, // movgr2fcsr
|
|
"MOVV.F.R": 0x452e << 10, // movfr2gr.d
|
|
"MOVV.F.FCC": 0x4534 << 10, // movfr2cf
|
|
"MOVV.FCC.R": 0x4537 << 10, // movcf2gr
|
|
"MOVV.FCC.F": 0x4535 << 10, // movcf2fr
|
|
"MOVV.FCSR.R": 0x4532 << 10, // movfcsr2gr
|
|
"MOVW.R.F": 0x4529 << 10, // movgr2fr.w
|
|
"MOVW.F.R": 0x452d << 10, // movfr2gr.s
|
|
}
|
|
|
|
// l64branchTable holds the 16-bit branch and jump encodings (2RI16).
|
|
var l64branchTable = map[string]uint32{
|
|
"BEQ": 0x16 << 26,
|
|
"BNE": 0x17 << 26,
|
|
"BLT": 0x18 << 26,
|
|
"BGE": 0x19 << 26,
|
|
"BLTU": 0x1a << 26,
|
|
"BGEU": 0x1b << 26,
|
|
"JIRL": 0x13 << 26,
|
|
}
|
|
|
|
// l64branch21Table holds the single-register branches with 21-bit offsets:
|
|
// the negative opcode constants the toolchain uses for the short forms.
|
|
var l64branch21Table = map[string]uint32{
|
|
"BEQZ": 0x10 << 26, // beq r0, rj → beqz
|
|
"BNEZ": 0x11 << 26, // bne r0, rj → bnez
|
|
"BLTZ": 0x18 << 26, // blt rj, r0 → bltz
|
|
"BGEZ": 0x19 << 26, // bge rj, r0 → bgez
|
|
"BGTZ": 0x18 << 26, // blt r0, rj → bgtz
|
|
"BLEZ": 0x19 << 26, // bge r0, rj → blez
|
|
"BFPT": 0x12<<26 | 0x1<<8,
|
|
"BFPF": 0x12<<26 | 0x0<<8,
|
|
}
|
|
|
|
// l64jumpTable maps the jump pseudo-instructions and their aliases to the
|
|
// B/BL opcode constants.
|
|
var l64jumpTable = map[string]uint32{
|
|
"JMP": 0x14 << 26, // b
|
|
"B": 0x14 << 26, // b
|
|
"JAL": 0x15 << 26, // bl
|
|
"CALL": 0x15 << 26, // bl
|
|
"BL": 0x15 << 26, // bl
|
|
}
|
|
|
|
// l64loadStoreTable maps the MOV width mnemonics to their load and store
|
|
// 2RI12 opcodes. The load opcode is the negated store opcode, exactly as
|
|
// the toolchain derives it.
|
|
var l64loadStoreTable = map[string]struct{ ld, st uint32 }{
|
|
"MOVB": {0x0a0 << 22, 0x0a4 << 22},
|
|
"MOVH": {0x0a1 << 22, 0x0a5 << 22},
|
|
"MOVW": {0x0a2 << 22, 0x0a6 << 22},
|
|
"MOVV": {0x0a3 << 22, 0x0a7 << 22},
|
|
"MOVBU": {0x0a8 << 22, 0x0a4 << 22},
|
|
"MOVHU": {0x0a9 << 22, 0x0a5 << 22},
|
|
"MOVWU": {0x0aa << 22, 0x0a6 << 22},
|
|
"MOVF": {0x0ac << 22, 0x0ad << 22},
|
|
"MOVD": {0x0ae << 22, 0x0af << 22},
|
|
}
|
|
|
|
// l64movRegTable maps a register-to-register MOV mnemonic to its expansion,
|
|
// matching the toolchain's case-1 encoding: MOVB → ext.w.b, MOVH → ext.w.h,
|
|
// MOVW → sll.w, MOVV → or, MOVBU → andi. MOVHU/MOVWU expand to bstrpick.d
|
|
// and are handled separately in the assembler.
|
|
type l64MovRegEnc struct {
|
|
rr bool // 2R format (ext.w.b/ext.w.h)
|
|
op uint32 // opcode constant (rr forms) or 3R/2RI12 opcode
|
|
imm int // 2RI12 immediate for MOVBU's andi
|
|
}
|
|
|
|
var l64movRegTable = map[string]l64MovRegEnc{
|
|
"MOVB": {true, 0x17 << 10, 0}, // ext.w.b rd, rj
|
|
"MOVH": {true, 0x16 << 10, 0}, // ext.w.h rd, rj
|
|
"MOVW": {false, 0x2e << 15, 0}, // sll.w rd, rj, r0
|
|
"MOVV": {false, 0x2a << 15, 0}, // or rd, rj, r0
|
|
"MOVBU": {false, 0x00d << 22, 0xff}, // andi rd, rj, $0xff
|
|
}
|
|
|
|
// l64movFpRegTable maps a floating-point register move mnemonic to its 2R
|
|
// opcode (fmov.s / fmov.d), used when both operands are F registers.
|
|
var l64movFpRegTable = map[string]uint32{
|
|
"MOVF": 0x4525 << 10,
|
|
"MOVD": 0x4526 << 10,
|
|
}
|
|
|
|
// l64RegClass discriminates integer (R), floating-point (F) and condition
|
|
// (FCC) registers for the MOV pseudo-instruction's register-move encoding.
|
|
type l64RegClass int
|
|
|
|
const (
|
|
l64ClsNone l64RegClass = iota
|
|
l64ClsGR
|
|
l64ClsFP
|
|
l64ClsFCC
|
|
l64ClsFCSR
|
|
)
|
|
|
|
// loong64RegClass reports the register class of a register operand name.
|
|
func loong64RegClass(name string) l64RegClass {
|
|
switch {
|
|
case name == "":
|
|
return l64ClsNone
|
|
case len(name) >= 3 && name[:3] == "FCC":
|
|
return l64ClsFCC
|
|
case len(name) >= 4 && name[:4] == "FCSR":
|
|
return l64ClsFCSR
|
|
case name[0] == 'F':
|
|
return l64ClsFP
|
|
default:
|
|
return l64ClsGR
|
|
}
|
|
}
|