feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+292
-55
@@ -308,6 +308,8 @@ const (
|
||||
a64CondLT = 0xb
|
||||
a64CondGT = 0xc
|
||||
a64CondLE = 0xd
|
||||
a64CondAL = 0xe
|
||||
a64CondNV = 0xf
|
||||
)
|
||||
|
||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
|
||||
"LT": a64CondLT,
|
||||
"GT": a64CondGT,
|
||||
"LE": a64CondLE,
|
||||
"AL": a64CondAL,
|
||||
"NV": a64CondNV,
|
||||
}
|
||||
|
||||
// ---- instruction format tags ----
|
||||
@@ -335,46 +339,47 @@ var arm64CondMap = map[string]uint32{
|
||||
type a64Format uint8
|
||||
|
||||
const (
|
||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||
a64FBranch // unconditional branch (B/BL)
|
||||
a64FBranchCond // conditional branch (B.cond)
|
||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||
a64FCRC32 // CRC32
|
||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
||||
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
||||
a64FCondCmp // conditional compare: CCMP, CCMN
|
||||
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
||||
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
||||
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
||||
a64FAcqRel // acquire/release: LDAR family, STLR family
|
||||
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
||||
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
||||
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
||||
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
||||
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
||||
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
||||
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
||||
a64FVTBL // SIMD table lookup: VTBL
|
||||
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
||||
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
||||
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
||||
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||
a64FBranch // unconditional branch (B/BL)
|
||||
a64FBranchCond // conditional branch (B.cond)
|
||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
|
||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||
a64FCRC32 // CRC32
|
||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
||||
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
||||
a64FCondCmp // conditional compare: CCMP, CCMN
|
||||
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
||||
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
||||
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
||||
a64FAcqRel // acquire/release: LDAR family, STLR family
|
||||
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
||||
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
||||
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
||||
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
||||
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
||||
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
||||
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
||||
a64FVTBL // SIMD table lookup: VTBL
|
||||
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
||||
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
||||
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
||||
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
||||
)
|
||||
|
||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||
@@ -487,6 +492,16 @@ func init() {
|
||||
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||
// The widening multiplies: a 64-bit result riding the same layout, the
|
||||
// three-operand forms reading the accumulate register as ZR.
|
||||
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
|
||||
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
|
||||
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
|
||||
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
|
||||
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
|
||||
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
|
||||
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
|
||||
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
|
||||
|
||||
// ---- move wide ----
|
||||
// MOVZ/MOVN/MOVK
|
||||
@@ -536,6 +551,15 @@ func init() {
|
||||
// ---- bitfield ----
|
||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
|
||||
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
|
||||
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
|
||||
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
|
||||
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
|
||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
@@ -716,6 +740,13 @@ func init() {
|
||||
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
||||
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
||||
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
||||
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
|
||||
// the source width.
|
||||
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
|
||||
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
|
||||
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
|
||||
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
|
||||
"REV16W": 0x5ac00400,
|
||||
}
|
||||
for m, op := range dp1 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
||||
@@ -734,7 +765,7 @@ func init() {
|
||||
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||
|
||||
// ---- system operations ----
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
||||
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||
}
|
||||
|
||||
@@ -785,6 +816,73 @@ func init() {
|
||||
for m, op := range lse {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
// The remaining width and ordering spellings of the same shapes, and the
|
||||
// CAS compare-and-swap family, word-verified against go tool asm.
|
||||
lseMore := map[string]uint32{
|
||||
"LDADDAB": 0x38a00000,
|
||||
"LDADDAH": 0x78a00000,
|
||||
"LDADDALB": 0x38e00000,
|
||||
"LDADDALH": 0x78e00000,
|
||||
"LDADDLB": 0x38600000,
|
||||
"LDADDLD": 0xf8600000,
|
||||
"LDADDLH": 0x78600000,
|
||||
"LDADDLW": 0xb8600000,
|
||||
"LDCLRAB": 0x38a01000,
|
||||
"LDCLRAH": 0x78a01000,
|
||||
"LDCLRALH": 0x78e01000,
|
||||
"LDCLRB": 0x38201000,
|
||||
"LDCLRD": 0xf8201000,
|
||||
"LDCLRH": 0x78201000,
|
||||
"LDCLRLB": 0x38601000,
|
||||
"LDCLRLD": 0xf8601000,
|
||||
"LDCLRLH": 0x78601000,
|
||||
"LDCLRLW": 0xb8601000,
|
||||
"LDCLRW": 0xb8201000,
|
||||
"LDEORAB": 0x38a02000,
|
||||
"LDEORAD": 0xf8a02000,
|
||||
"LDEORAH": 0x78a02000,
|
||||
"LDEORALB": 0x38e02000,
|
||||
"LDEORALH": 0x78e02000,
|
||||
"LDEORAW": 0xb8a02000,
|
||||
"LDEORB": 0x38202000,
|
||||
"LDEORD": 0xf8202000,
|
||||
"LDEORH": 0x78202000,
|
||||
"LDEORLB": 0x38602000,
|
||||
"LDEORLD": 0xf8602000,
|
||||
"LDEORLH": 0x78602000,
|
||||
"LDEORLW": 0xb8602000,
|
||||
"LDEORW": 0xb8202000,
|
||||
"LDORAB": 0x38a03000,
|
||||
"LDORAD": 0xf8a03000,
|
||||
"LDORAH": 0x78a03000,
|
||||
"LDORALH": 0x78e03000,
|
||||
"LDORAW": 0xb8a03000,
|
||||
"LDORB": 0x38203000,
|
||||
"LDORD": 0xf8203000,
|
||||
"LDORH": 0x78203000,
|
||||
"LDORLB": 0x38603000,
|
||||
"LDORLD": 0xf8603000,
|
||||
"LDORLH": 0x78603000,
|
||||
"LDORLW": 0xb8603000,
|
||||
"LDORW": 0xb8203000,
|
||||
"SWPAB": 0x38a08000,
|
||||
"SWPAD": 0xf8a08000,
|
||||
"SWPAH": 0x78a08000,
|
||||
"SWPALH": 0x78e08000,
|
||||
"SWPAW": 0xb8a08000,
|
||||
"SWPB": 0x38208000,
|
||||
"SWPH": 0x78208000,
|
||||
"SWPLB": 0x38608000,
|
||||
"SWPLD": 0xf8608000,
|
||||
"SWPLH": 0x78608000,
|
||||
"SWPLW": 0xb8608000,
|
||||
"CASAD": 0xc8e07c00,
|
||||
"CASALB": 0x08e0fc00,
|
||||
"CASLW": 0x88a0fc00,
|
||||
}
|
||||
for m, op := range lseMore {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
|
||||
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||
@@ -794,7 +892,12 @@ func init() {
|
||||
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
||||
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
||||
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
||||
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
|
||||
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
|
||||
"NGC": 0xda000000, "NGCW": 0x5a000000,
|
||||
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
|
||||
"NEGSW": 0x6b000000,
|
||||
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
||||
}
|
||||
for m, op := range dpsrExtra {
|
||||
@@ -835,6 +938,12 @@ func init() {
|
||||
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
||||
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
||||
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
||||
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
||||
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
||||
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
|
||||
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
||||
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
||||
// the across-vector reductions.
|
||||
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
||||
|
||||
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
||||
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
||||
// and pairwise compares among them. Word-verified against go tool asm.
|
||||
var a64SimdQOnly = map[string]bool{
|
||||
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
|
||||
"VFABS": true, "VFNEG": true, "VFSQRT": true,
|
||||
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
|
||||
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
|
||||
"VFMAXV": true, "VFMAXNMV": true,
|
||||
}
|
||||
|
||||
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
||||
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
||||
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
||||
@@ -918,19 +1044,96 @@ var a64ArrBits = [a64ArrCount]uint32{
|
||||
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
||||
// word and arrangement bit was read off go tool asm.
|
||||
var a64SimdVTable = map[string]a64SimdVSpec{
|
||||
"VADD": {0x0e208400, 0x7f, false},
|
||||
"VSUB": {0x2e208400, 0x7f, false},
|
||||
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
||||
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
||||
"VEOR": {0x2e201c00, 0x03, false},
|
||||
"VORR": {0x0ea01c00, 0x03, false},
|
||||
"VADDP": {0x0e20bc00, 0x7f, false},
|
||||
"VZIP1": {0x0e003800, 0x7f, false},
|
||||
"VZIP2": {0x0e007800, 0x7f, false},
|
||||
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||
"VADD": {0x0e208400, 0x7f, false},
|
||||
"VSUB": {0x2e208400, 0x7f, false},
|
||||
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
||||
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
||||
"VEOR": {0x2e201c00, 0x03, false},
|
||||
"VORR": {0x0ea01c00, 0x03, false},
|
||||
"VADDP": {0x0e20bc00, 0x7f, false},
|
||||
"VZIP1": {0x0e003800, 0x7f, false},
|
||||
"VZIP2": {0x0e007800, 0x7f, false},
|
||||
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||
"VCMGE": {0x0e203c00, 0x7f, false},
|
||||
"VCMGT": {0x0e203400, 0x7f, false},
|
||||
"VCMHI": {0x2e203400, 0x7f, false},
|
||||
"VCMHS": {0x2e203c00, 0x7f, false},
|
||||
// FP compares take H, S and D arrangements only (the toolchain rejects
|
||||
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
||||
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
// FP arithmetic shares the same arrangement restriction.
|
||||
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
||||
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
||||
// tool asm.
|
||||
"VBIC": {0x0e601c00, 0x7f, false},
|
||||
"VBIF": {0x2ee01c00, 0x7f, false},
|
||||
"VBIT": {0x6ea01c00, 0x7f, false},
|
||||
"VBSL": {0x6e601c00, 0x7f, false},
|
||||
"VCMTST": {0x0e208c00, 0x7f, false},
|
||||
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VMLA": {0x4ea09400, 0x7f, false},
|
||||
"VMLS": {0x6ea09400, 0x7f, false},
|
||||
"VORN": {0x4ee01c00, 0x7f, false},
|
||||
"VSHADD": {0x4ea00400, 0x7f, false},
|
||||
"VSRHADD": {0x4ea01400, 0x7f, false},
|
||||
"VUHADD": {0x6ea00400, 0x7f, false},
|
||||
"VURHADD": {0x6ea01400, 0x7f, false},
|
||||
"VSMAX": {0x4ea06400, 0x7f, false},
|
||||
"VSMIN": {0x4ea06c00, 0x7f, false},
|
||||
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
||||
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
||||
"VUMAX": {0x2e206400, 0x7f, false},
|
||||
"VUMIN": {0x2e206c00, 0x7f, false},
|
||||
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
||||
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
||||
"VSQADD": {0x4ea00c00, 0x7f, false},
|
||||
"VUQADD": {0x6ea00c00, 0x7f, false},
|
||||
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
||||
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
||||
"VSSHL": {0x4ee04400, 0x7f, false},
|
||||
"VUSHL": {0x6ee04400, 0x7f, false},
|
||||
"VUZP1": {0x0e001800, 0x7f, false},
|
||||
"VUZP2": {0x4ec05800, 0x7f, false},
|
||||
"VTRN1": {0x4ec02800, 0x7f, false},
|
||||
"VTRN2": {0x4ec06800, 0x7f, false},
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||
}
|
||||
|
||||
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
||||
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
|
||||
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
|
||||
// combination for them, so they stay out and the encoder rejects the shape.
|
||||
var a64SimdVZero = map[string]uint32{
|
||||
"VCMEQ": 0x0e209800,
|
||||
"VCMGT": 0x0e208800,
|
||||
"VCMGE": 0x2e208800,
|
||||
"VCMLT": 0x0e20a800,
|
||||
"VCMLE": 0x2e209800,
|
||||
// FP compares against (0.0): the register forms above carry the U and op
|
||||
// bits; the zero forms reshape them.
|
||||
"VFCMEQ": 0x0ea0d800,
|
||||
"VFCMGE": 0x2ea0c800,
|
||||
"VFCMGT": 0x0ea0c800,
|
||||
"VFCMLE": 0x2ea0d800,
|
||||
"VFCMLT": 0x0ea0e800,
|
||||
}
|
||||
|
||||
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
||||
@@ -939,8 +1142,41 @@ var a64SimdVTable = map[string]a64SimdVSpec{
|
||||
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||
"VREV64": {0x0e200800, 0x3f, false},
|
||||
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
// Two-register data-processing across one arrangement.
|
||||
"VABS": {0x0e20b800, 0x7f, false},
|
||||
"VNEG": {0x2e20b800, 0x7f, false},
|
||||
"VCLS": {0x0e204800, 0x7f, false},
|
||||
"VCLZ": {0x2e204800, 0x7f, false},
|
||||
"VCNT": {0x0e205800, 0x7f, false},
|
||||
"VNOT": {0x2e205800, 0x7f, false},
|
||||
"VSQABS": {0x0e207800, 0x7f, false},
|
||||
"VSQNEG": {0x2e207800, 0x7f, false},
|
||||
"VRBIT": {0x6e605800, 0x7f, false},
|
||||
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
||||
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
||||
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
||||
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
||||
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
||||
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
||||
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
||||
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
||||
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
||||
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
||||
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
||||
// Across-vector reductions: the operand arrangement rides as usual and
|
||||
// the destination stays a bare V register.
|
||||
"VADDV": {0x0e31b800, 0x3f, false},
|
||||
"VSMAXV": {0x0e30a800, 0x3f, false},
|
||||
"VSMINV": {0x0e31a800, 0x3f, false},
|
||||
"VUMAXV": {0x2e30a800, 0x3f, false},
|
||||
"VUMINV": {0x2e31a800, 0x3f, false},
|
||||
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
||||
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
||||
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
||||
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
||||
}
|
||||
|
||||
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
|
||||
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||
var a64MSRRegOps = map[string]uint32{
|
||||
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||
"ELR_EL1": 0xd5184020,
|
||||
}
|
||||
|
||||
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||
|
||||
Reference in New Issue
Block a user