1425 lines
56 KiB
Go
1425 lines
56 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
// arm64 (AArch64) instruction encoding.
|
|
//
|
|
// The encoder is data-driven: each mnemonic maps to an instruction format and
|
|
// an opcode constant, and the format selects the bit layout. The opcode
|
|
// constants and formats are transcribed from the Go toolchain's own arm64
|
|
// backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm`
|
|
// exactly, the ground-truth oracle for the verify suite.
|
|
//
|
|
// All AArch64 instructions are 32 bits, little-endian. The formats used here
|
|
// (per the ARM Architecture Reference Manual):
|
|
//
|
|
// DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
|
|
// DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd
|
|
// Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd
|
|
// Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd
|
|
// Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt
|
|
// LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt)
|
|
// LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt
|
|
// Branch-imm 0<<31 | 0x5<<26 | imm26 (B)
|
|
// Branch-imm 1<<31 | 0x5<<26 | imm26 (BL)
|
|
// Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond)
|
|
// Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET)
|
|
// ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd
|
|
|
|
import (
|
|
"maps"
|
|
"math/bits"
|
|
"strconv"
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
|
)
|
|
|
|
// arm64RegNum returns the 5-bit register number for an AArch64 register name:
|
|
// R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the
|
|
// runtime's assembly uses. Returns -1 for an unrecognised name.
|
|
func arm64RegNum(name string) int {
|
|
switch name {
|
|
case "R0":
|
|
return 0
|
|
case "R1":
|
|
return 1
|
|
case "R2":
|
|
return 2
|
|
case "R3":
|
|
return 3
|
|
case "R4":
|
|
return 4
|
|
case "R5":
|
|
return 5
|
|
case "R6":
|
|
return 6
|
|
case "R7":
|
|
return 7
|
|
case "R8":
|
|
return 8
|
|
case "R9":
|
|
return 9
|
|
case "R10":
|
|
return 10
|
|
case "R11":
|
|
return 11
|
|
case "R12":
|
|
return 12
|
|
case "R13":
|
|
return 13
|
|
case "R14":
|
|
return 14
|
|
case "R15":
|
|
return 15
|
|
case "R16":
|
|
return 16
|
|
case "R17":
|
|
return 17
|
|
case "R18":
|
|
return 18
|
|
case "R18_PLATFORM":
|
|
// The toolchain's Windows spelling: R18 is renamed R18_PLATFORM in
|
|
// cmd/asm/internal/arch so assembly cannot use it by accident, and
|
|
// sys_windows_arm64.s references it only through this name.
|
|
return 18
|
|
case "R19":
|
|
return 19
|
|
case "R20":
|
|
return 20
|
|
case "R21":
|
|
return 21
|
|
case "R22":
|
|
return 22
|
|
case "R23":
|
|
return 23
|
|
case "R24":
|
|
return 24
|
|
case "R25":
|
|
return 25
|
|
case "R26", "REGCTXT", "CTXT":
|
|
return 26
|
|
case "R27", "REGTMP", "TMP":
|
|
return 27
|
|
case "R28", "REGG", "g":
|
|
return 28
|
|
case "R29", "FP":
|
|
return 29
|
|
case "R30", "LR", "LINK":
|
|
return 30
|
|
case "R31", "ZR":
|
|
return 31
|
|
case "SP", "RSP":
|
|
// RSP is the toolchain's spelling for register 31 (it rejects
|
|
// R31 in an operand); SP stays for sources that spell it the
|
|
// amd64 way. SP and ZR share encoding 31; context determines
|
|
// the meaning.
|
|
return 31
|
|
}
|
|
// F0-F31.
|
|
if len(name) >= 1 && name[0] == 'F' {
|
|
n := 0
|
|
for i := 1; i < len(name); i++ {
|
|
if name[i] < '0' || name[i] > '9' {
|
|
return -1
|
|
}
|
|
n = n*10 + int(name[i]-'0')
|
|
}
|
|
if n <= 31 {
|
|
return n
|
|
}
|
|
}
|
|
return -1
|
|
}
|
|
|
|
// ---- format helpers ----
|
|
|
|
// a64wordLE encodes a uint32 as 4 little-endian bytes.
|
|
func a64wordLE(w uint32) []byte {
|
|
return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)}
|
|
}
|
|
|
|
// a64WordsLE concatenates one or more instruction words as little-endian bytes.
|
|
func a64WordsLE(ws ...uint32) []byte {
|
|
var out []byte
|
|
for _, w := range ws {
|
|
out = append(out, a64wordLE(w)...)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// ---- data-processing (immediate) ----
|
|
|
|
// a64AddSub encodes an ADD/SUB (immediate) instruction:
|
|
// sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd.
|
|
func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 {
|
|
return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd
|
|
}
|
|
|
|
// ---- move wide ----
|
|
|
|
// a64MoveWide encodes a MOVZ/MOVK/MOVN instruction:
|
|
// sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd.
|
|
func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 {
|
|
return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd
|
|
}
|
|
|
|
// ---- logical immediate ----
|
|
|
|
// a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the
|
|
// given lane width (32 or 64): it returns the N, immr and imms fields of the
|
|
// imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's
|
|
// encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest
|
|
// repeating element, find the run of ones and its rotation. ok is false when
|
|
// v is not expressible (all zeros, all ones, or not a single cyclic run).
|
|
func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) {
|
|
u := uint64(v)
|
|
if width == 32 {
|
|
u &= 0xFFFFFFFF
|
|
}
|
|
size := uint64(width)
|
|
mask := ^uint64(0)
|
|
if size < 64 {
|
|
mask = uint64(1)<<size - 1
|
|
}
|
|
u &= mask
|
|
// All zeros and all ones are MOV territory, not bitmask immediates.
|
|
if u == 0 || u == mask {
|
|
return 0, 0, 0, false
|
|
}
|
|
// Shrink to the smallest repeating element.
|
|
for size > 2 {
|
|
half := size / 2
|
|
hm := uint64(1)<<half - 1
|
|
if u&hm == u>>half&hm {
|
|
size = half
|
|
u &= hm
|
|
} else {
|
|
break
|
|
}
|
|
}
|
|
ones := bits.OnesCount64(u)
|
|
// Find the right-rotation that lays the ones out contiguously at the
|
|
// bottom of the element; the hardware applies the inverse rotation.
|
|
em := uint64(1)<<size - 1
|
|
expected := uint64(1)<<ones - 1
|
|
rot := -1
|
|
for r := 0; r < int(size); r++ {
|
|
rotated := u>>r | u<<(int(size)-r)
|
|
if size < 64 {
|
|
rotated &= em
|
|
}
|
|
if rotated == expected {
|
|
rot = r
|
|
break
|
|
}
|
|
}
|
|
if rot < 0 {
|
|
return 0, 0, 0, false
|
|
}
|
|
if size == 64 {
|
|
n = 1
|
|
}
|
|
immr = uint32((int(size) - rot) % int(size))
|
|
imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1)
|
|
return n, immr, imms, true
|
|
}
|
|
|
|
// ---- load/store (unsigned immediate, scaled) ----
|
|
|
|
// a64LSU encodes a load/store register (unsigned immediate, scaled):
|
|
// size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt.
|
|
// (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.)
|
|
func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 {
|
|
return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt
|
|
}
|
|
|
|
// ---- load/store (unscaled immediate) ----
|
|
|
|
// a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed):
|
|
// size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt.
|
|
// Note: the 0<<24 distinguishes unscaled from the pre/post-index forms.
|
|
func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 {
|
|
return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 |
|
|
(uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31)
|
|
}
|
|
|
|
// ---- load/store pair ----
|
|
|
|
// a64LSP encodes a load/store pair instruction (signed offset):
|
|
// opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt.
|
|
// opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD.
|
|
// L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs).
|
|
func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 {
|
|
return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt
|
|
}
|
|
|
|
// ---- branches ----
|
|
|
|
// a64Branch encodes an unconditional branch (B/BL):
|
|
// op<<31 | 0x5<<26 | imm26.
|
|
func a64Branch(op uint32, imm26 int32) uint32 {
|
|
return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF)
|
|
}
|
|
|
|
// a64BranchCond encodes a conditional branch (B.cond):
|
|
// 0x2A<<25 | imm19<<5 | cond.
|
|
func a64BranchCond(imm19 int32, cond uint32) uint32 {
|
|
return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF
|
|
}
|
|
|
|
// a64UncondBranch encodes an unconditional branch register (BR/BLR/RET):
|
|
// 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd.
|
|
// opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30).
|
|
func a64UncondBranch(opc, rn, rd uint32) uint32 {
|
|
return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd
|
|
}
|
|
|
|
// ---- ADR/ADRP ----
|
|
|
|
// a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1):
|
|
// p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd.
|
|
func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 {
|
|
return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd
|
|
}
|
|
|
|
// ---- system ----
|
|
|
|
// a64NOP encodes a NOP: 0xd503201f.
|
|
const a64NOP uint32 = 0xd503201f
|
|
|
|
// a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5.
|
|
func a64BRK(imm16 uint32) uint32 {
|
|
return 0xd4200000 | imm16<<5
|
|
}
|
|
|
|
// ---- condition codes ----
|
|
|
|
const (
|
|
a64CondEQ = 0x0
|
|
a64CondNE = 0x1
|
|
a64CondCS = 0x2
|
|
a64CondHS = 0x2
|
|
a64CondCC = 0x3
|
|
a64CondLO = 0x3
|
|
a64CondMI = 0x4
|
|
a64CondPL = 0x5
|
|
a64CondVS = 0x6
|
|
a64CondVC = 0x7
|
|
a64CondHI = 0x8
|
|
a64CondLS = 0x9
|
|
a64CondGE = 0xa
|
|
a64CondLT = 0xb
|
|
a64CondGT = 0xc
|
|
a64CondLE = 0xd
|
|
a64CondAL = 0xe
|
|
a64CondNV = 0xf
|
|
)
|
|
|
|
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
|
var arm64CondMap = map[string]uint32{
|
|
"EQ": a64CondEQ,
|
|
"NE": a64CondNE,
|
|
"CS": a64CondCS,
|
|
"HS": a64CondHS,
|
|
"CC": a64CondCC,
|
|
"LO": a64CondLO,
|
|
"MI": a64CondMI,
|
|
"PL": a64CondPL,
|
|
"VS": a64CondVS,
|
|
"VC": a64CondVC,
|
|
"HI": a64CondHI,
|
|
"LS": a64CondLS,
|
|
"GE": a64CondGE,
|
|
"LT": a64CondLT,
|
|
"GT": a64CondGT,
|
|
"LE": a64CondLE,
|
|
"AL": a64CondAL,
|
|
"NV": a64CondNV,
|
|
}
|
|
|
|
// ---- instruction format tags ----
|
|
|
|
type a64Format uint8
|
|
|
|
const (
|
|
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
|
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
|
a64FBranch // unconditional branch (B/BL)
|
|
a64FBranchCond // conditional branch (B.cond)
|
|
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
|
a64FADR // ADR/ADRP
|
|
a64FEXTR // EXTR
|
|
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
|
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
|
|
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
|
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
|
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
|
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
|
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
|
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
|
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
|
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
|
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
|
a64FCRC32 // CRC32
|
|
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
|
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
|
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
|
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
|
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
|
a64FCondCmp // conditional compare: CCMP, CCMN
|
|
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
|
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
|
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
|
a64FAcqRel // acquire/release: LDAR family, STLR family
|
|
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
|
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
|
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
|
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
|
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
|
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
|
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
|
a64FVTBL // SIMD table lookup: VTBL
|
|
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
|
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
|
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
|
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
|
)
|
|
|
|
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
|
// opcode constant, positioned at its exact bit range.
|
|
type a64Enc struct {
|
|
format a64Format
|
|
op uint32 // the pre-positioned opcode bits
|
|
}
|
|
|
|
// a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to
|
|
// their encoding. The base integer, memory, floating-point and SIMD
|
|
// instruction sets are covered.
|
|
var a64InstrTable = map[string]a64Enc{}
|
|
|
|
func init() {
|
|
// ---- data-processing (shifted register) ----
|
|
// Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd
|
|
dpsr := map[string]uint32{
|
|
// Add/Sub
|
|
"ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default)
|
|
"ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0
|
|
"ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24,
|
|
"ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24,
|
|
"SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24,
|
|
"SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24,
|
|
"SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24,
|
|
"SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24,
|
|
// Logical (shifted register)
|
|
"AND": 1<<31 | 0<<29 | 0x0a<<24,
|
|
"ANDW": 0<<31 | 0<<29 | 0x0a<<24,
|
|
"BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21,
|
|
"BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21,
|
|
"ORR": 1<<31 | 1<<29 | 0x0a<<24,
|
|
"ORRW": 0<<31 | 1<<29 | 0x0a<<24,
|
|
"ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21,
|
|
"ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21,
|
|
"EOR": 1<<31 | 2<<29 | 0x0a<<24,
|
|
"EORW": 0<<31 | 2<<29 | 0x0a<<24,
|
|
"EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21,
|
|
"EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21,
|
|
"ANDS": 1<<31 | 3<<29 | 0x0a<<24,
|
|
"ANDSW": 0<<31 | 3<<29 | 0x0a<<24,
|
|
"BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
|
"BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21,
|
|
// Divide (data-processing 2 source): the opcode occupies bits 15:10
|
|
// of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM
|
|
// "Data-processing (2 source)"; the toolchain spells them OPDP2(2)
|
|
// and OPDP2(3)). sf=1 selects the X forms.
|
|
"SDIV": 1<<31 | 0xd6<<21 | 3<<10,
|
|
"SDIVW": 0<<31 | 0xd6<<21 | 3<<10,
|
|
"UDIV": 1<<31 | 0xd6<<21 | 2<<10,
|
|
"UDIVW": 0<<31 | 0xd6<<21 | 2<<10,
|
|
// Conditional select
|
|
"CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
|
"CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10,
|
|
"CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10,
|
|
"CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10,
|
|
"CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10,
|
|
"CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10,
|
|
"CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10,
|
|
"CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10,
|
|
}
|
|
for m, op := range dpsr {
|
|
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
|
|
}
|
|
|
|
// Aliases that map to the same encoding as their target.
|
|
a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
|
|
a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]}
|
|
a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]}
|
|
a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]}
|
|
a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]}
|
|
a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]}
|
|
a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]}
|
|
a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]}
|
|
a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]}
|
|
a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]}
|
|
a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]}
|
|
a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]}
|
|
a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]}
|
|
|
|
// ---- shifts ----
|
|
// The mnemonic serves both forms: with an immediate the aliases of the
|
|
// data-processing (immediate) group apply (ARM ARM "Shifts"), with a
|
|
// register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op
|
|
// field carries the immediate-alias base; encodeARM64Shift derives both
|
|
// it and the two-source opcode. Identities, W = 64 (X) or 32 (W):
|
|
//
|
|
// LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
|
|
// LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1)
|
|
// ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1)
|
|
// ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh
|
|
shifts := map[string]a64Enc{
|
|
"LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
|
|
"LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
|
|
"LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X
|
|
"LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W
|
|
"ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X
|
|
"ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W
|
|
"ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X
|
|
"RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W
|
|
}
|
|
maps.Copy(a64InstrTable, shifts)
|
|
|
|
// ---- multiply accumulate ----
|
|
// MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The
|
|
// toolchain's optab has no shorter row, so all four operands are
|
|
// mandatory, and Ra is the SECOND operand.
|
|
a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24}
|
|
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
|
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
|
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
|
// The widening multiplies: a 64-bit result riding the same layout, the
|
|
// three-operand forms reading the accumulate register as ZR.
|
|
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
|
|
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
|
|
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
|
|
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
|
|
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
|
|
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
|
|
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
|
|
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
|
|
|
|
// ---- move wide ----
|
|
// MOVZ/MOVN/MOVK
|
|
a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23}
|
|
a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23}
|
|
a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23}
|
|
a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23}
|
|
a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23}
|
|
a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23}
|
|
|
|
// ---- ADR/ADRP ----
|
|
a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0}
|
|
a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1}
|
|
|
|
// Load/store mnemonics never enter this table: the MOV pseudo-instruction
|
|
// dispatch handles them through a64LoadTable, which also carries the store
|
|
// opcode (integer and FP stores both use opc=00, differing only in V).
|
|
|
|
// ---- branches ----
|
|
a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26}
|
|
a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26}
|
|
|
|
// Conditional branches.
|
|
condBranches := map[string]uint32{
|
|
"BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2,
|
|
"BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5,
|
|
"BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9,
|
|
"BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd,
|
|
}
|
|
for name, cond := range condBranches {
|
|
a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond}
|
|
}
|
|
|
|
// Unconditional branch register (BR/BLR/RET).
|
|
a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21}
|
|
a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21}
|
|
a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21}
|
|
|
|
// ---- system ----
|
|
// NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch,
|
|
// so they carry no table entry; a64NOP and a64BRK are the encoders.
|
|
|
|
// ---- EXTR ----
|
|
a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22}
|
|
a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22}
|
|
|
|
// ---- bitfield ----
|
|
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
|
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
|
|
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
|
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
|
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
|
|
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
|
|
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
|
|
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
|
|
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
|
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
|
|
a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}
|
|
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
|
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
|
|
|
// ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ----
|
|
fp3 := map[string]uint32{
|
|
"FADDS": 0x1e202800, "FADDD": 0x1e602800,
|
|
"FSUBS": 0x1e203800, "FSUBD": 0x1e603800,
|
|
"FMULS": 0x1e200800, "FMULD": 0x1e600800,
|
|
"FDIVS": 0x1e201800, "FDIVD": 0x1e601800,
|
|
"FMAXS": 0x1e204800, "FMAXD": 0x1e604800,
|
|
"FMINS": 0x1e205800, "FMIND": 0x1e605800,
|
|
"FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800,
|
|
"FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800,
|
|
"FNMULS": 0x1e208800, "FNMULD": 0x1e608800,
|
|
}
|
|
for m, op := range fp3 {
|
|
a64InstrTable[m] = a64Enc{format: a64FFP3, op: op}
|
|
}
|
|
|
|
// ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ----
|
|
fp1 := map[string]uint32{
|
|
"FMOVS": 0x1e204000, "FMOVD": 0x1e604000,
|
|
"FABSS": 0x1e20c000, "FABSD": 0x1e60c000,
|
|
"FNEGS": 0x1e214000, "FNEGD": 0x1e614000,
|
|
"FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000,
|
|
"FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000,
|
|
"FRINTNS": 0x1e244000, "FRINTND": 0x1e644000,
|
|
"FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000,
|
|
"FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000,
|
|
"FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000,
|
|
"FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000,
|
|
"FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000,
|
|
"FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000,
|
|
}
|
|
for m, op := range fp1 {
|
|
a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op}
|
|
}
|
|
|
|
// ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ----
|
|
fp4 := map[string]uint32{
|
|
"FMADDS": 0x1f000000, "FMADDD": 0x1f400000,
|
|
"FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000,
|
|
"FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000,
|
|
"FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000,
|
|
}
|
|
for m, op := range fp4 {
|
|
a64InstrTable[m] = a64Enc{format: a64FFP4, op: op}
|
|
}
|
|
|
|
// ---- FP compare (Rm, Rn or #0, Rn) ----
|
|
fpcmp := map[string]uint32{
|
|
"FCMPS": 0x1e202000, "FCMPD": 0x1e602000,
|
|
"FCMPES": 0x1e202010, "FCMPED": 0x1e602010,
|
|
}
|
|
for m, op := range fpcmp {
|
|
a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op}
|
|
}
|
|
|
|
// ---- FP conditional compare (Rm, Rn, #nzcv, cond) ----
|
|
fpccmp := map[string]uint32{
|
|
"FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400,
|
|
"FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410,
|
|
}
|
|
for m, op := range fpccmp {
|
|
a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op}
|
|
}
|
|
|
|
// ---- FP conditional select (Rm, Rn, Rd, cond) ----
|
|
a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00}
|
|
a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00}
|
|
|
|
// ---- FP ↔ integer conversion ----
|
|
fpcvt := map[string]uint32{
|
|
"FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000,
|
|
"FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000,
|
|
"FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000,
|
|
"FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000,
|
|
"SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000,
|
|
"SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000,
|
|
"UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000,
|
|
"UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000,
|
|
}
|
|
for m, op := range fpcvt {
|
|
a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op}
|
|
}
|
|
|
|
// FMOV between GP and FP registers needs no table entry: the MOV
|
|
// pseudo-instruction dispatches it by operand class (encodeARM64RegMove).
|
|
|
|
// ---- conditional select: CSEL, CSINC, CSINV, CSNEG ----
|
|
csel := map[string]uint32{
|
|
"CSEL": 0x9a800000, "CSELW": 0x1a800000,
|
|
"CSINC": 0x9a800400, "CSINCW": 0x1a800400,
|
|
"CSINV": 0xda800000, "CSINVW": 0x5a800000,
|
|
"CSNEG": 0xda800400, "CSNEGW": 0x5a800400,
|
|
}
|
|
for m, op := range csel {
|
|
a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op}
|
|
}
|
|
// Aliases
|
|
a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
|
|
a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
|
|
a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000}
|
|
a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
|
|
a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400}
|
|
a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400}
|
|
a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000}
|
|
a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000}
|
|
a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400}
|
|
a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400}
|
|
|
|
// ---- CRC32 ----
|
|
crc32 := map[string]uint32{
|
|
"CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400,
|
|
"CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00,
|
|
"CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400,
|
|
"CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00,
|
|
}
|
|
for m, op := range crc32 {
|
|
a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op}
|
|
}
|
|
|
|
// ---- exclusive load/store ----
|
|
// Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the
|
|
// 0x7c00/0x1f0000 halves of the constants below); the register-pair
|
|
// forms carry a real Rt2 in bits 14:10, so their opcodes pre-set
|
|
// neither field.
|
|
a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00}
|
|
a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00}
|
|
a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00}
|
|
a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00}
|
|
a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00}
|
|
a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00}
|
|
a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00}
|
|
a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00}
|
|
// Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have
|
|
// o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31.
|
|
a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000}
|
|
a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000}
|
|
a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000}
|
|
a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000}
|
|
a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00}
|
|
a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00}
|
|
a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00}
|
|
a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00}
|
|
a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00}
|
|
a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00}
|
|
a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00}
|
|
a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00}
|
|
// Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0,
|
|
// STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields.
|
|
a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000}
|
|
a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000}
|
|
a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000}
|
|
a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000}
|
|
|
|
// ---- LSE atomics ----
|
|
a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10}
|
|
a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10}
|
|
a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10}
|
|
a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10}
|
|
a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10}
|
|
a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10}
|
|
a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10}
|
|
a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10}
|
|
|
|
// ---- SIMD: the arrangement-aware tables in this file carry VADD,
|
|
// VSUB, VMUL and every other three-register vector op. ----
|
|
|
|
// ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ----
|
|
dp1 := map[string]uint32{
|
|
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
|
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
|
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
|
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
|
|
// the source width.
|
|
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
|
|
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
|
|
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
|
|
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
|
|
"REV16W": 0x5ac00400,
|
|
}
|
|
for m, op := range dp1 {
|
|
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
|
}
|
|
|
|
// ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ----
|
|
a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000}
|
|
a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000}
|
|
a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000}
|
|
a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000}
|
|
|
|
// ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ----
|
|
a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000}
|
|
a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000}
|
|
a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000}
|
|
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
|
|
|
// ---- system operations ----
|
|
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
|
a64InstrTable[m] = a64Enc{format: a64FSys}
|
|
}
|
|
|
|
// ---- compare/test and branch ----
|
|
a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000}
|
|
a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000}
|
|
a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000}
|
|
a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000}
|
|
a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000}
|
|
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
|
|
|
|
// ---- load/store pair (signed offset) ----
|
|
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
|
|
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
|
|
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
|
|
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
|
|
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
|
|
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
|
|
|
|
// ---- acquire/release loads and stores ----
|
|
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
|
|
a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00}
|
|
a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00}
|
|
a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00}
|
|
a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00}
|
|
a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00}
|
|
a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00}
|
|
a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00}
|
|
|
|
// ---- LSE atomics with acquire and release semantics ----
|
|
// CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/
|
|
// LDOR/SWP families leave Rs free for the returned value.
|
|
lse := map[string]uint32{
|
|
"CASALD": 0xc8e0fc00,
|
|
"CASALW": 0x88e0fc00,
|
|
"LDADDALD": 0xf8e00000,
|
|
"LDADDALW": 0xb8e00000,
|
|
"LDCLRALB": 0x38e01000,
|
|
"LDCLRALW": 0xb8e01000,
|
|
"LDCLRALD": 0xf8e01000,
|
|
"LDORALB": 0x38e03000,
|
|
"LDORALW": 0xb8e03000,
|
|
"LDORALD": 0xf8e03000,
|
|
"SWPALB": 0x38e08000,
|
|
"SWPALW": 0xb8e08000,
|
|
"SWPALD": 0xf8e08000,
|
|
}
|
|
for m, op := range lse {
|
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
|
}
|
|
// The remaining width and ordering spellings of the same shapes, and the
|
|
// CAS compare-and-swap family, word-verified against go tool asm.
|
|
lseMore := map[string]uint32{
|
|
"LDADDAB": 0x38a00000,
|
|
"LDADDAH": 0x78a00000,
|
|
"LDADDALB": 0x38e00000,
|
|
"LDADDALH": 0x78e00000,
|
|
"LDADDLB": 0x38600000,
|
|
"LDADDLD": 0xf8600000,
|
|
"LDADDLH": 0x78600000,
|
|
"LDADDLW": 0xb8600000,
|
|
"LDCLRAB": 0x38a01000,
|
|
"LDCLRAH": 0x78a01000,
|
|
"LDCLRALH": 0x78e01000,
|
|
"LDCLRB": 0x38201000,
|
|
"LDCLRD": 0xf8201000,
|
|
"LDCLRH": 0x78201000,
|
|
"LDCLRLB": 0x38601000,
|
|
"LDCLRLD": 0xf8601000,
|
|
"LDCLRLH": 0x78601000,
|
|
"LDCLRLW": 0xb8601000,
|
|
"LDCLRW": 0xb8201000,
|
|
"LDEORAB": 0x38a02000,
|
|
"LDEORAD": 0xf8a02000,
|
|
"LDEORAH": 0x78a02000,
|
|
"LDEORALB": 0x38e02000,
|
|
"LDEORALH": 0x78e02000,
|
|
"LDEORAW": 0xb8a02000,
|
|
"LDEORB": 0x38202000,
|
|
"LDEORD": 0xf8202000,
|
|
"LDEORH": 0x78202000,
|
|
"LDEORLB": 0x38602000,
|
|
"LDEORLD": 0xf8602000,
|
|
"LDEORLH": 0x78602000,
|
|
"LDEORLW": 0xb8602000,
|
|
"LDEORW": 0xb8202000,
|
|
"LDORAB": 0x38a03000,
|
|
"LDORAD": 0xf8a03000,
|
|
"LDORAH": 0x78a03000,
|
|
"LDORALH": 0x78e03000,
|
|
"LDORAW": 0xb8a03000,
|
|
"LDORB": 0x38203000,
|
|
"LDORD": 0xf8203000,
|
|
"LDORH": 0x78203000,
|
|
"LDORLB": 0x38603000,
|
|
"LDORLD": 0xf8603000,
|
|
"LDORLH": 0x78603000,
|
|
"LDORLW": 0xb8603000,
|
|
"LDORW": 0xb8203000,
|
|
"SWPAB": 0x38a08000,
|
|
"SWPAD": 0xf8a08000,
|
|
"SWPAH": 0x78a08000,
|
|
"SWPALH": 0x78e08000,
|
|
"SWPAW": 0xb8a08000,
|
|
"SWPB": 0x38208000,
|
|
"SWPH": 0x78208000,
|
|
"SWPLB": 0x38608000,
|
|
"SWPLD": 0xf8608000,
|
|
"SWPLH": 0x78608000,
|
|
"SWPLW": 0xb8608000,
|
|
"CASAD": 0xc8e07c00,
|
|
"CASALB": 0x08e0fc00,
|
|
"CASLW": 0x88a0fc00,
|
|
}
|
|
for m, op := range lseMore {
|
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
|
}
|
|
|
|
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
|
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
|
// register preset to ZR (bits 14:10 = 11111).
|
|
dpsrExtra := map[string]uint32{
|
|
"ADC": 0x9a000000, "ADCW": 0x1a000000,
|
|
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
|
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
|
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
|
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
|
|
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
|
|
"NGC": 0xda000000, "NGCW": 0x5a000000,
|
|
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
|
|
"NEGSW": 0x6b000000,
|
|
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
|
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
|
}
|
|
for m, op := range dpsrExtra {
|
|
a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op}
|
|
}
|
|
|
|
// ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ----
|
|
crypto2 := map[string]uint32{
|
|
"AESD": 0x4e285800, "AESE": 0x4e284800,
|
|
"AESIMC": 0x4e287800, "AESMC": 0x4e286800,
|
|
"SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800,
|
|
"SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000,
|
|
}
|
|
for m, op := range crypto2 {
|
|
a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op}
|
|
}
|
|
crypto3 := map[string]uint32{
|
|
"SHA1C": 0x5e000000, "SHA1P": 0x5e001000,
|
|
"SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000,
|
|
"SHA256H": 0x5e004000, "SHA256H2": 0x5e005000,
|
|
"SHA256SU1": 0x5e006000, "SHA512H": 0xce608000,
|
|
"SHA512H2": 0xce608400, "SHA512SU1": 0xce608800,
|
|
}
|
|
for m, op := range crypto3 {
|
|
a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op}
|
|
}
|
|
|
|
// ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ----
|
|
a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000}
|
|
a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000}
|
|
a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000}
|
|
a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000}
|
|
a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL}
|
|
a64InstrTable["VDUP"] = a64Enc{format: a64FDUP}
|
|
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
|
|
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
|
|
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
|
|
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
|
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
|
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
|
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
|
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
|
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
|
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
|
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
|
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
|
|
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
|
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
|
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
|
a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
|
a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST}
|
|
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
|
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
|
|
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
|
|
}
|
|
|
|
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
|
|
// the set of arrangements it accepts as a bitmask over the a64Arr index and,
|
|
// for instructions that exist at a single arrangement and carry that
|
|
// arrangement's bits inside the base already, the fixed flag.
|
|
type a64SimdVSpec struct {
|
|
base uint32
|
|
arrs uint16
|
|
fixed bool
|
|
}
|
|
|
|
// a64Arr names the vector arrangements the encoders deal with, indexed by
|
|
// a64Arr. The source spellings put the element letter first: B8, H4, S2,
|
|
// D1 and the 128-bit halves B16, H8, S4, D2.
|
|
const (
|
|
a64Arr8B = iota
|
|
a64Arr16B
|
|
a64Arr4H
|
|
a64Arr8H
|
|
a64Arr2S
|
|
a64Arr4S
|
|
a64Arr2D
|
|
a64ArrD1
|
|
a64ArrQ1
|
|
a64ArrCount
|
|
)
|
|
|
|
// a64ArrNames maps an arrangement to its source spelling (element letter
|
|
// first, as the toolchain writes it).
|
|
var a64ArrNames = [a64ArrCount]string{
|
|
a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8",
|
|
a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1",
|
|
}
|
|
|
|
// a64ArrIndex resolves a source spelling to its a64Arr index, -1 when
|
|
// unknown.
|
|
func a64ArrIndex(s string) int {
|
|
for i, n := range a64ArrNames {
|
|
if n == s {
|
|
return i
|
|
}
|
|
}
|
|
return -1
|
|
}
|
|
|
|
// a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q)
|
|
// as it appears in element operands such as V13.S[0].
|
|
func a64ElemLetter(s string) bool {
|
|
switch s {
|
|
case "B", "H", "S", "D", "Q":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
|
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
|
// the across-vector reductions.
|
|
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
|
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
|
|
|
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
|
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
|
// and pairwise compares among them. Word-verified against go tool asm.
|
|
var a64SimdQOnly = map[string]bool{
|
|
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
|
|
"VFABS": true, "VFNEG": true, "VFSQRT": true,
|
|
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
|
|
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
|
|
"VFMAXV": true, "VFMAXNMV": true,
|
|
}
|
|
|
|
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
|
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
|
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
|
var a64ArrBits = [a64ArrCount]uint32{
|
|
a64Arr8B: 0,
|
|
a64Arr16B: 1 << 30,
|
|
a64Arr4H: 1 << 22,
|
|
a64Arr8H: 1<<30 | 1<<22,
|
|
a64Arr2S: 1 << 23,
|
|
a64Arr4S: 1<<30 | 1<<23,
|
|
a64Arr2D: 1<<30 | 1<<23 | 1<<22,
|
|
a64ArrD1: 1<<23 | 1<<22,
|
|
a64ArrQ1: 0,
|
|
}
|
|
|
|
// a64SimdVTable holds the arrangement-aware three-register SIMD
|
|
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
|
// word and arrangement bit was read off go tool asm.
|
|
var a64SimdVTable = map[string]a64SimdVSpec{
|
|
"VADD": {0x0e208400, 0x7f, false},
|
|
"VSUB": {0x2e208400, 0x7f, false},
|
|
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
|
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
|
"VEOR": {0x2e201c00, 0x03, false},
|
|
"VORR": {0x0ea01c00, 0x03, false},
|
|
"VADDP": {0x0e20bc00, 0x7f, false},
|
|
"VZIP1": {0x0e003800, 0x7f, false},
|
|
"VZIP2": {0x0e007800, 0x7f, false},
|
|
"VCMEQ": {0x2e208c00, 0x7f, false},
|
|
"VCMGE": {0x0e203c00, 0x7f, false},
|
|
"VCMGT": {0x0e203400, 0x7f, false},
|
|
"VCMHI": {0x2e203400, 0x7f, false},
|
|
"VCMHS": {0x2e203c00, 0x7f, false},
|
|
// FP compares take H, S and D arrangements only (the toolchain rejects
|
|
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
|
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
// FP arithmetic shares the same arrangement restriction.
|
|
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
|
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
|
// tool asm.
|
|
"VBIC": {0x0e601c00, 0x7f, false},
|
|
"VBIF": {0x2ee01c00, 0x7f, false},
|
|
"VBIT": {0x6ea01c00, 0x7f, false},
|
|
"VBSL": {0x6e601c00, 0x7f, false},
|
|
"VCMTST": {0x0e208c00, 0x7f, false},
|
|
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
|
"VMLA": {0x4ea09400, 0x7f, false},
|
|
"VMLS": {0x6ea09400, 0x7f, false},
|
|
"VORN": {0x4ee01c00, 0x7f, false},
|
|
"VSHADD": {0x4ea00400, 0x7f, false},
|
|
"VSRHADD": {0x4ea01400, 0x7f, false},
|
|
"VUHADD": {0x6ea00400, 0x7f, false},
|
|
"VURHADD": {0x6ea01400, 0x7f, false},
|
|
"VSMAX": {0x4ea06400, 0x7f, false},
|
|
"VSMIN": {0x4ea06c00, 0x7f, false},
|
|
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
|
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
|
"VUMAX": {0x2e206400, 0x7f, false},
|
|
"VUMIN": {0x2e206c00, 0x7f, false},
|
|
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
|
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
|
"VSQADD": {0x4ea00c00, 0x7f, false},
|
|
"VUQADD": {0x6ea00c00, 0x7f, false},
|
|
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
|
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
|
"VSSHL": {0x4ee04400, 0x7f, false},
|
|
"VUSHL": {0x6ee04400, 0x7f, false},
|
|
"VUZP1": {0x0e001800, 0x7f, false},
|
|
"VUZP2": {0x4ec05800, 0x7f, false},
|
|
"VTRN1": {0x4ec02800, 0x7f, false},
|
|
"VTRN2": {0x4ec06800, 0x7f, false},
|
|
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
|
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
|
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
|
}
|
|
|
|
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
|
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
|
|
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
|
|
// combination for them, so they stay out and the encoder rejects the shape.
|
|
var a64SimdVZero = map[string]uint32{
|
|
"VCMEQ": 0x0e209800,
|
|
"VCMGT": 0x0e208800,
|
|
"VCMGE": 0x2e208800,
|
|
"VCMLT": 0x0e20a800,
|
|
"VCMLE": 0x2e209800,
|
|
// FP compares against (0.0): the register forms above carry the U and op
|
|
// bits; the zero forms reshape them.
|
|
"VFCMEQ": 0x0ea0d800,
|
|
"VFCMGE": 0x2ea0c800,
|
|
"VFCMGT": 0x0ea0c800,
|
|
"VFCMLE": 0x2ea0d800,
|
|
"VFCMLT": 0x0ea0e800,
|
|
}
|
|
|
|
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
|
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
|
|
// the register pair spelling ORR Vd, Vn, Vm.
|
|
var a64SimdV2Table = map[string]a64SimdVSpec{
|
|
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
|
"VREV64": {0x0e200800, 0x3f, false},
|
|
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
|
"VUADDLV": {0x2e303800, 0x3f, false},
|
|
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
|
// Two-register data-processing across one arrangement.
|
|
"VABS": {0x0e20b800, 0x7f, false},
|
|
"VNEG": {0x2e20b800, 0x7f, false},
|
|
"VCLS": {0x0e204800, 0x7f, false},
|
|
"VCLZ": {0x2e204800, 0x7f, false},
|
|
"VCNT": {0x0e205800, 0x7f, false},
|
|
"VNOT": {0x2e205800, 0x7f, false},
|
|
"VSQABS": {0x0e207800, 0x7f, false},
|
|
"VSQNEG": {0x2e207800, 0x7f, false},
|
|
"VRBIT": {0x6e605800, 0x7f, false},
|
|
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
|
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
|
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
|
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
|
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
|
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
|
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
|
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
|
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
|
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
|
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
|
// Across-vector reductions: the operand arrangement rides as usual and
|
|
// the destination stays a bare V register.
|
|
"VADDV": {0x0e31b800, 0x3f, false},
|
|
"VSMAXV": {0x0e30a800, 0x3f, false},
|
|
"VSMINV": {0x0e31a800, 0x3f, false},
|
|
"VUMAXV": {0x2e30a800, 0x3f, false},
|
|
"VUMINV": {0x2e31a800, 0x3f, false},
|
|
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
|
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
|
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
|
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
|
}
|
|
|
|
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
|
// carry when they spell one at all; a bare V/F spelling is accepted as is.
|
|
var a64CryptoArr = map[string]int{
|
|
"AESD": a64Arr16B, "AESE": a64Arr16B, "AESIMC": a64Arr16B, "AESMC": a64Arr16B,
|
|
"SHA1H": a64Arr4S, "SHA1SU1": a64Arr4S, "SHA256SU0": a64Arr4S, "SHA512SU0": a64Arr2D,
|
|
"SHA1C": a64Arr4S, "SHA1P": a64Arr4S, "SHA1M": a64Arr4S, "SHA1SU0": a64Arr4S,
|
|
"SHA256H": a64Arr4S, "SHA256H2": a64Arr4S, "SHA256SU1": a64Arr4S,
|
|
"SHA512H": a64Arr2D, "SHA512H2": a64Arr2D, "SHA512SU1": a64Arr2D,
|
|
}
|
|
|
|
// a64DCOps maps the data-cache maintenance operation names to their fixed
|
|
// word (the register rides bits 4:0).
|
|
var a64DCOps = map[string]uint32{
|
|
"IVAC": 0xd5087620, "ZVA": 0xd50b7420,
|
|
"CVAC": 0xd50b7a20, "CVAU": 0xd50b7b20, "CIVAC": 0xd50b7e20,
|
|
}
|
|
|
|
// a64MRSOps maps the system register names GOROOT reads to their fixed word
|
|
// (the destination register rides bits 4:0).
|
|
var a64MRSOps = map[string]uint32{
|
|
"ELR_EL1": 0xd5384020, "MIDR_EL1": 0xd5380000,
|
|
"ID_AA64PFR0_EL1": 0xd5380400, "ID_AA64ISAR0_EL1": 0xd5380600,
|
|
"ID_AA64ISAR1_EL1": 0xd5380620, "CNTFRQ_EL0": 0xd53be000,
|
|
"CNTPCT_EL0": 0xd53be020, "CNTVCT_EL0": 0xd53be040,
|
|
"DCZID_EL0": 0xd53b00e0, "DIT": 0xd53b42a0, "ID_AA64ZFR0_EL1": 0xd5380480,
|
|
"NZCV": 0xd53b4200, "FPCR": 0xd53b4400, "FPSR": 0xd53b4420,
|
|
}
|
|
|
|
// a64MSRRegOps maps the system register names GOROOT writes through the
|
|
// MSR (register) form, spelled in Go assembly as MOVD Rn, <sysreg> or
|
|
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
|
var a64MSRRegOps = map[string]uint32{
|
|
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
|
"ELR_EL1": 0xd5184020,
|
|
}
|
|
|
|
// a64MSROps maps the system register names GOROOT writes to their fixed
|
|
// word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111.
|
|
var a64MSROps = map[string]uint32{
|
|
"SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040,
|
|
}
|
|
|
|
// a64PRFOps maps the prefetch operation names to their prfop immediate
|
|
// (word = 0xf9800000 | Rn<<5 | prfop).
|
|
var a64PRFOps = map[string]int{
|
|
"PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03,
|
|
"PLDL3KEEP": 0x04, "PLDL3STRM": 0x05,
|
|
"PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b,
|
|
"PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d,
|
|
"PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13,
|
|
"PSTL3KEEP": 0x14, "PSTL3STRM": 0x15,
|
|
}
|
|
|
|
// a64VLD1Base holds the fixed words of the multi-register structure
|
|
// accesses, indexed by register count 1..4, before the Q and size bits.
|
|
// Post-index spellings add 0x9f0000 (post bit and Rm = 11111).
|
|
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
|
|
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
|
|
|
|
// a64Vec is a parsed vector operand: the register number, the arrangement
|
|
// ("" when the operand spells none) and, for element forms, the lane index.
|
|
type a64Vec struct {
|
|
reg int
|
|
arr string
|
|
idx int
|
|
hasIdx bool
|
|
}
|
|
|
|
// a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias,
|
|
// the same architectural registers the scalar floating-point spellings use),
|
|
// optionally with an arrangement suffix such as V0.B16 and, for element
|
|
// forms, a lane index such as V13.S[0]. It reports ok=false for anything
|
|
// else, including X/W and R spellings, which the toolchain's vector
|
|
// operands reject as well.
|
|
func a64VecReg(name string) (v a64Vec, ok bool) {
|
|
s := strings.TrimSpace(name)
|
|
if i := strings.IndexByte(s, '.'); i >= 0 {
|
|
v.arr = strings.TrimSpace(s[i+1:])
|
|
s = s[:i]
|
|
}
|
|
if v.arr != "" {
|
|
// Element form: B[3], S[2] and friends.
|
|
if j := strings.IndexByte(v.arr, '['); j >= 0 {
|
|
k := strings.LastIndexByte(v.arr, ']')
|
|
if k < j {
|
|
return v, false
|
|
}
|
|
n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k]))
|
|
if err != nil || n < 0 {
|
|
return v, false
|
|
}
|
|
v.idx, v.hasIdx = n, true
|
|
v.arr = strings.TrimSpace(v.arr[:j])
|
|
}
|
|
if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) {
|
|
return v, false
|
|
}
|
|
}
|
|
if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') {
|
|
return v, false
|
|
}
|
|
n := 0
|
|
for i := 1; i < len(s); i++ {
|
|
if s[i] < '0' || s[i] > '9' {
|
|
return v, false
|
|
}
|
|
n = n*10 + int(s[i]-'0')
|
|
}
|
|
if n > 31 {
|
|
return v, false
|
|
}
|
|
v.reg = n
|
|
return v, true
|
|
}
|
|
|
|
// a64ElemField encodes a lane index for the copy/insert group: imm5 = the
|
|
// index shifted by the element scale, with the scale's own bit set. B gets
|
|
// shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4.
|
|
func a64ElemField(arr string, idx int) (uint32, bool) {
|
|
var shift, low uint32
|
|
switch arr {
|
|
case "B8", "B16", "B":
|
|
shift, low = 1, 1
|
|
case "H4", "H8", "H":
|
|
shift, low = 2, 2
|
|
case "S2", "S4", "S":
|
|
shift, low = 3, 4
|
|
case "D1", "D2", "D":
|
|
shift, low = 4, 8
|
|
default:
|
|
return 0, false
|
|
}
|
|
if idx < 0 || idx >= 1<<(5-shift) {
|
|
return 0, false
|
|
}
|
|
return uint32(idx)<<shift | low, true
|
|
}
|
|
|
|
// a64VecListOf recovers the register list of a VLD1/VST1/VTBL operand run.
|
|
// The parser keeps parenthesised groups whole but splits bracketed lists on
|
|
// the commas, so a list arrives as one operand run whose first Raw starts
|
|
// with "[" and whose last Raw ends with "]". It returns the parsed
|
|
// registers with the brackets and spaces removed.
|
|
func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool) {
|
|
if start >= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") {
|
|
return nil, 0, false
|
|
}
|
|
end = start
|
|
for end < len(ops) {
|
|
if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") {
|
|
break
|
|
}
|
|
end++
|
|
}
|
|
if end >= len(ops) {
|
|
return nil, 0, false
|
|
}
|
|
for i := start; i <= end; i++ {
|
|
s := strings.TrimSpace(ops[i].Raw)
|
|
s = strings.TrimPrefix(s, "[")
|
|
s = strings.TrimSuffix(s, "]")
|
|
if s == "" && len(ops) > start+1 {
|
|
return nil, 0, false
|
|
}
|
|
for part := range strings.SplitSeq(s, ",") {
|
|
v, ok := a64VecReg(part)
|
|
if !ok {
|
|
return nil, 0, false
|
|
}
|
|
vs = append(vs, v)
|
|
}
|
|
}
|
|
return vs, end, true
|
|
}
|
|
|
|
// ---- load/store helper tables ----
|
|
|
|
// a64LSType describes the load/store parameters for a MOV width mnemonic.
|
|
type a64LSType struct {
|
|
size int // 0=byte, 1=half, 2=word, 3=dword
|
|
V int // 0=integer, 1=FP
|
|
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
|
|
}
|
|
|
|
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
|
|
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
|
|
var a64LoadTable = map[string]a64LSType{
|
|
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
|
|
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
|
|
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
|
|
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
|
|
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
|
|
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
|
|
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
|
|
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
|
|
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
|
|
}
|
|
|
|
// a64StoreOpc returns the store opc for a given load type: integer and FP
|
|
// stores both encode opc=00 (the load's signedness bit sits in opc[1], which
|
|
// the store form clears; FP registers are selected by V, not opc).
|
|
func a64StoreOpc(t a64LSType) int {
|
|
return 0
|
|
}
|
|
|
|
// arm64RegClass discriminates integer (R), floating-point (F) registers for
|
|
// the MOV pseudo-instruction.
|
|
type arm64RegClass int
|
|
|
|
const (
|
|
arm64ClsNone arm64RegClass = iota
|
|
arm64ClsGR
|
|
arm64ClsFP
|
|
)
|
|
|
|
// arm64RegClassOf reports the register class of a register operand name.
|
|
func arm64RegClassOf(name string) arm64RegClass {
|
|
switch {
|
|
case name == "":
|
|
return arm64ClsNone
|
|
case len(name) >= 1 && name[0] == 'F':
|
|
return arm64ClsFP
|
|
default:
|
|
return arm64ClsGR
|
|
}
|
|
}
|
|
|
|
// arm64Movcon returns the shift (in units of 16 bits) at which a non-zero
|
|
// 16-bit chunk of v sits, or -1 if v cannot be represented as a single
|
|
// MOVZ/MOVN immediate. This is the Go toolchain's movcon function.
|
|
func arm64Movcon(v int64) int {
|
|
for s := 0; s < 64; s += 16 {
|
|
if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 {
|
|
return s
|
|
}
|
|
}
|
|
return -1
|
|
}
|