// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm // arm64 (AArch64) instruction encoding. // // The encoder is data-driven: each mnemonic maps to an instruction format and // an opcode constant, and the format selects the bit layout. The opcode // constants and formats are transcribed from the Go toolchain's own arm64 // backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm` // exactly, the ground-truth oracle for the verify suite. // // All AArch64 instructions are 32 bits, little-endian. The formats used here // (per the ARM Architecture Reference Manual): // // DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd // DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd // Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd // Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd // Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt // LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt) // LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt // Branch-imm 0<<31 | 0x5<<26 | imm26 (B) // Branch-imm 1<<31 | 0x5<<26 | imm26 (BL) // Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond) // Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET) // ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd import ( "maps" "math/bits" "strconv" "strings" "sourcedock.dev/petrbalvin/gasm-devkit/ast" ) // arm64RegNum returns the 5-bit register number for an AArch64 register name: // R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the // runtime's assembly uses. Returns -1 for an unrecognised name. func arm64RegNum(name string) int { switch name { case "R0": return 0 case "R1": return 1 case "R2": return 2 case "R3": return 3 case "R4": return 4 case "R5": return 5 case "R6": return 6 case "R7": return 7 case "R8": return 8 case "R9": return 9 case "R10": return 10 case "R11": return 11 case "R12": return 12 case "R13": return 13 case "R14": return 14 case "R15": return 15 case "R16": return 16 case "R17": return 17 case "R18": return 18 case "R19": return 19 case "R20": return 20 case "R21": return 21 case "R22": return 22 case "R23": return 23 case "R24": return 24 case "R25": return 25 case "R26", "REGCTXT", "CTXT": return 26 case "R27", "REGTMP", "TMP": return 27 case "R28", "REGG", "g": return 28 case "R29", "FP": return 29 case "R30", "LR", "LINK": return 30 case "R31", "ZR": return 31 case "SP", "RSP": // RSP is the toolchain's spelling for register 31 (it rejects // R31 in an operand); SP stays for sources that spell it the // amd64 way. SP and ZR share encoding 31; context determines // the meaning. return 31 } // F0-F31. if len(name) >= 1 && name[0] == 'F' { n := 0 for i := 1; i < len(name); i++ { if name[i] < '0' || name[i] > '9' { return -1 } n = n*10 + int(name[i]-'0') } if n <= 31 { return n } } return -1 } // ---- format helpers ---- // a64wordLE encodes a uint32 as 4 little-endian bytes. func a64wordLE(w uint32) []byte { return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} } // a64WordsLE concatenates one or more instruction words as little-endian bytes. func a64WordsLE(ws ...uint32) []byte { var out []byte for _, w := range ws { out = append(out, a64wordLE(w)...) } return out } // ---- data-processing (immediate) ---- // a64AddSub encodes an ADD/SUB (immediate) instruction: // sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd. func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 { return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd } // ---- move wide ---- // a64MoveWide encodes a MOVZ/MOVK/MOVN instruction: // sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd. func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 { return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd } // ---- logical immediate ---- // a64LogicalImm encodes v as the AArch64 logical (bitmask) immediate for the // given lane width (32 or 64): it returns the N, immr and imms fields of the // imm13 encoding. The algorithm mirrors cmd/internal/obj/arm64's // encodeLogicalImmArrEncoding: replicate the value, shrink it to the smallest // repeating element, find the run of ones and its rotation. ok is false when // v is not expressible (all zeros, all ones, or not a single cyclic run). func a64LogicalImm(v int64, width int) (n, immr, imms uint32, ok bool) { u := uint64(v) if width == 32 { u &= 0xFFFFFFFF } size := uint64(width) mask := ^uint64(0) if size < 64 { mask = uint64(1)< 2 { half := size / 2 hm := uint64(1)<>half&hm { size = half u &= hm } else { break } } ones := bits.OnesCount64(u) // Find the right-rotation that lays the ones out contiguously at the // bottom of the element; the hardware applies the inverse rotation. em := uint64(1)<>r | u<<(int(size)-r) if size < 64 { rotated &= em } if rotated == expected { rot = r break } } if rot < 0 { return 0, 0, 0, false } if size == 64 { n = 1 } immr = uint32((int(size) - rot) % int(size)) imms = ^uint32(uint32(size*2-1))&0x3F | uint32(ones-1) return n, immr, imms, true } // ---- load/store (unsigned immediate, scaled) ---- // a64LSU encodes a load/store register (unsigned immediate, scaled): // size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt. // (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.) func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 { return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt } // ---- load/store (unscaled immediate) ---- // a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed): // size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt. // Note: the 0<<24 distinguishes unscaled from the pre/post-index forms. func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } // ---- load/store pair ---- // a64LSP encodes a load/store pair instruction (signed offset): // opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt. // opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD. // L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs). func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 { return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt } // ---- branches ---- // a64Branch encodes an unconditional branch (B/BL): // op<<31 | 0x5<<26 | imm26. func a64Branch(op uint32, imm26 int32) uint32 { return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF) } // a64BranchCond encodes a conditional branch (B.cond): // 0x2A<<25 | imm19<<5 | cond. func a64BranchCond(imm19 int32, cond uint32) uint32 { return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF } // a64UncondBranch encodes an unconditional branch register (BR/BLR/RET): // 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd. // opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30). func a64UncondBranch(opc, rn, rd uint32) uint32 { return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd } // ---- ADR/ADRP ---- // a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1): // p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd. func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 { return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd } // ---- system ---- // a64NOP encodes a NOP: 0xd503201f. const a64NOP uint32 = 0xd503201f // a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5. func a64BRK(imm16 uint32) uint32 { return 0xd4200000 | imm16<<5 } // ---- condition codes ---- const ( a64CondEQ = 0x0 a64CondNE = 0x1 a64CondCS = 0x2 a64CondHS = 0x2 a64CondCC = 0x3 a64CondLO = 0x3 a64CondMI = 0x4 a64CondPL = 0x5 a64CondVS = 0x6 a64CondVC = 0x7 a64CondHI = 0x8 a64CondLS = 0x9 a64CondGE = 0xa a64CondLT = 0xb a64CondGT = 0xc a64CondLE = 0xd a64CondAL = 0xe a64CondNV = 0xf ) // arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes. var arm64CondMap = map[string]uint32{ "EQ": a64CondEQ, "NE": a64CondNE, "CS": a64CondCS, "HS": a64CondHS, "CC": a64CondCC, "LO": a64CondLO, "MI": a64CondMI, "PL": a64CondPL, "VS": a64CondVS, "VC": a64CondVC, "HI": a64CondHI, "LS": a64CondLS, "GE": a64CondGE, "LT": a64CondLT, "GT": a64CondGT, "LE": a64CondLE, "AL": a64CondAL, "NV": a64CondNV, } // ---- instruction format tags ---- type a64Format uint8 const ( a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc. a64FMovWide // move wide: MOVZ, MOVN, MOVK a64FBranch // unconditional branch (B/BL) a64FBranchCond // conditional branch (B.cond) a64FUncondBranch // unconditional branch register (BR/BLR/RET) a64FADR // ADR/ADRP a64FEXTR // EXTR a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd) a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10 a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT* a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc. a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc. a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL a64FCRC32 // CRC32 a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP a64FLSE // LSE atomics: LDADD, CAS, SWP a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms a64FCondCmp // conditional compare: CCMP, CCMN a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD a64FAcqRel // acquire/release: LDAR family, STLR family a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ... a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ... a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ... a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT a64FVTBL // SIMD table lookup: VTBL a64FDUP // SIMD element moves: VDUP, VMOV with element indices a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool) ) // a64Enc is one instruction's encoding: its bit layout (format) and the // opcode constant, positioned at its exact bit range. type a64Enc struct { format a64Format op uint32 // the pre-positioned opcode bits } // a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to // their encoding. The base integer, memory, floating-point and SIMD // instruction sets are covered. var a64InstrTable = map[string]a64Enc{} func init() { // ---- data-processing (shifted register) ---- // Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd dpsr := map[string]uint32{ // Add/Sub "ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default) "ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0 "ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24, "ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24, "SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24, "SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24, "SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24, "SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24, // Logical (shifted register) "AND": 1<<31 | 0<<29 | 0x0a<<24, "ANDW": 0<<31 | 0<<29 | 0x0a<<24, "BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21, "BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21, "ORR": 1<<31 | 1<<29 | 0x0a<<24, "ORRW": 0<<31 | 1<<29 | 0x0a<<24, "ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21, "ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21, "EOR": 1<<31 | 2<<29 | 0x0a<<24, "EORW": 0<<31 | 2<<29 | 0x0a<<24, "EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21, "EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21, "ANDS": 1<<31 | 3<<29 | 0x0a<<24, "ANDSW": 0<<31 | 3<<29 | 0x0a<<24, "BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21, "BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21, // Divide (data-processing 2 source): the opcode occupies bits 15:10 // of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM // "Data-processing (2 source)"; the toolchain spells them OPDP2(2) // and OPDP2(3)). sf=1 selects the X forms. "SDIV": 1<<31 | 0xd6<<21 | 3<<10, "SDIVW": 0<<31 | 0xd6<<21 | 3<<10, "UDIV": 1<<31 | 0xd6<<21 | 2<<10, "UDIVW": 0<<31 | 0xd6<<21 | 2<<10, // Conditional select "CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10, "CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10, "CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10, "CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10, "CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10, "CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10, "CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10, "CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10, } for m, op := range dpsr { a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op} } // Aliases that map to the same encoding as their target. a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]} a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]} a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]} a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]} a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]} a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]} a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]} a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]} a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]} a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]} a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]} a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]} a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]} // ---- shifts ---- // The mnemonic serves both forms: with an immediate the aliases of the // data-processing (immediate) group apply (ARM ARM "Shifts"), with a // register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op // field carries the immediate-alias base; encodeARM64Shift derives both // it and the two-source opcode. Identities, W = 64 (X) or 32 (W): // // LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh // LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1) // ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1) // ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh shifts := map[string]a64Enc{ "LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X "LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W "LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X "LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W "ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X "ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W "ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X "RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W } maps.Copy(a64InstrTable, shifts) // ---- multiply accumulate ---- // MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The // toolchain's optab has no shorter row, so all four operands are // mandatory, and Ra is the SECOND operand. a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24} a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24} a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15} a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15} // The widening multiplies: a 64-bit result riding the same layout, the // three-operand forms reading the accumulate register as ZR. a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21} a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23} a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15} a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15} a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10} a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10} a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10} a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10} // ---- move wide ---- // MOVZ/MOVN/MOVK a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23} a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23} a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23} a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23} a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23} a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23} // ---- ADR/ADRP ---- a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0} a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1} // Load/store mnemonics never enter this table: the MOV pseudo-instruction // dispatch handles them through a64LoadTable, which also carries the store // opcode (integer and FP stores both use opc=00, differing only in V). // ---- branches ---- a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26} a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26} // Conditional branches. condBranches := map[string]uint32{ "BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2, "BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5, "BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9, "BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd, } for name, cond := range condBranches { a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond} } // Unconditional branch register (BR/BLR/RET). a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21} a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21} a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21} // ---- system ---- // NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch, // so they carry no table entry; a64NOP and a64BRK are the encoders. // ---- EXTR ---- a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22} a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22} // ---- bitfield ---- a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} // The four-operand bitfield aliases: ($lsb, Rn, $width, Rd). a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23} a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23} a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000} a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000} a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000} a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000} a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22} a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22} a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} // ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ---- fp3 := map[string]uint32{ "FADDS": 0x1e202800, "FADDD": 0x1e602800, "FSUBS": 0x1e203800, "FSUBD": 0x1e603800, "FMULS": 0x1e200800, "FMULD": 0x1e600800, "FDIVS": 0x1e201800, "FDIVD": 0x1e601800, "FMAXS": 0x1e204800, "FMAXD": 0x1e604800, "FMINS": 0x1e205800, "FMIND": 0x1e605800, "FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800, "FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800, "FNMULS": 0x1e208800, "FNMULD": 0x1e608800, } for m, op := range fp3 { a64InstrTable[m] = a64Enc{format: a64FFP3, op: op} } // ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ---- fp1 := map[string]uint32{ "FMOVS": 0x1e204000, "FMOVD": 0x1e604000, "FABSS": 0x1e20c000, "FABSD": 0x1e60c000, "FNEGS": 0x1e214000, "FNEGD": 0x1e614000, "FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000, "FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000, "FRINTNS": 0x1e244000, "FRINTND": 0x1e644000, "FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000, "FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000, "FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000, "FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000, "FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000, "FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000, } for m, op := range fp1 { a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op} } // ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ---- fp4 := map[string]uint32{ "FMADDS": 0x1f000000, "FMADDD": 0x1f400000, "FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000, "FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000, "FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000, } for m, op := range fp4 { a64InstrTable[m] = a64Enc{format: a64FFP4, op: op} } // ---- FP compare (Rm, Rn or #0, Rn) ---- fpcmp := map[string]uint32{ "FCMPS": 0x1e202000, "FCMPD": 0x1e602000, "FCMPES": 0x1e202010, "FCMPED": 0x1e602010, } for m, op := range fpcmp { a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op} } // ---- FP conditional compare (Rm, Rn, #nzcv, cond) ---- fpccmp := map[string]uint32{ "FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400, "FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410, } for m, op := range fpccmp { a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op} } // ---- FP conditional select (Rm, Rn, Rd, cond) ---- a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00} a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00} // ---- FP ↔ integer conversion ---- fpcvt := map[string]uint32{ "FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000, "FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000, "FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000, "FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000, "SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000, "SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000, "UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000, "UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000, } for m, op := range fpcvt { a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op} } // FMOV between GP and FP registers needs no table entry: the MOV // pseudo-instruction dispatches it by operand class (encodeARM64RegMove). // ---- conditional select: CSEL, CSINC, CSINV, CSNEG ---- csel := map[string]uint32{ "CSEL": 0x9a800000, "CSELW": 0x1a800000, "CSINC": 0x9a800400, "CSINCW": 0x1a800400, "CSINV": 0xda800000, "CSINVW": 0x5a800000, "CSNEG": 0xda800400, "CSNEGW": 0x5a800400, } for m, op := range csel { a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op} } // Aliases a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400} a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400} a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000} a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000} a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400} a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400} a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000} a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000} a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400} a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400} // ---- CRC32 ---- crc32 := map[string]uint32{ "CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400, "CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00, "CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400, "CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00, } for m, op := range crc32 { a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op} } // ---- exclusive load/store ---- // Single-register forms pre-set the unused Rs and Rt2 fields to 31 (the // 0x7c00/0x1f0000 halves of the constants below); the register-pair // forms carry a real Rt2 in bits 14:10, so their opcodes pre-set // neither field. a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00} a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00} a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00} a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00} a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00} a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00} a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00} a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00} // Pair loads, LDSTX(sz, 0, l=1, o1=1, o0) in asm7.go: LDXP/ LDXPW have // o0=0, LDAXP/LDAXPW o0=1 (bit 15). Rs (bits 20:16) stays 31. a64InstrTable["LDXP"] = a64Enc{format: a64FExcl, op: 0xc8600000} a64InstrTable["LDXPW"] = a64Enc{format: a64FExcl, op: 0x88600000} a64InstrTable["LDAXP"] = a64Enc{format: a64FExcl, op: 0xc8608000} a64InstrTable["LDAXPW"] = a64Enc{format: a64FExcl, op: 0x88608000} a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00} a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00} a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00} a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00} a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00} a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00} a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00} a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00} // Pair stores, LDSTX(sz, 0, l=0, o1=1, o0): STXP/STXPW have o0=0, // STLXP/STLXPW o0=1 (bit 15). Both Rs and Rt2 are real fields. a64InstrTable["STXP"] = a64Enc{format: a64FExcl, op: 0xc8200000} a64InstrTable["STXPW"] = a64Enc{format: a64FExcl, op: 0x88200000} a64InstrTable["STLXP"] = a64Enc{format: a64FExcl, op: 0xc8208000} a64InstrTable["STLXPW"] = a64Enc{format: a64FExcl, op: 0x88208000} // ---- LSE atomics ---- a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10} a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10} a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10} a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10} // ---- SIMD: the arrangement-aware tables in this file carry VADD, // VSUB, VMUL and every other three-register vector op. ---- // ---- data-processing (1 source): sf 10 11010110 opcode 00000 Rn Rd ---- dp1 := map[string]uint32{ "RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800, "REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400, "RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400, // Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing // the source width. "SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00, "SXTHW": 0x13003c00, "SXTW": 0x93407c00, "UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00, "UXTHW": 0x53003c00, "UXTW": 0x53407c00, "REV16W": 0x5ac00400, } for m, op := range dp1 { a64InstrTable[m] = a64Enc{format: a64FDP1, op: op} } // ---- bitfield extract: the UBFM/SBFM bases, immediate operands wrap ---- a64InstrTable["UBFX"] = a64Enc{format: a64FBitfield2, op: 0xd3400000} a64InstrTable["SBFX"] = a64Enc{format: a64FBitfield2, op: 0x93400000} a64InstrTable["UBFXW"] = a64Enc{format: a64FBitfield2, op: 0x53000000} a64InstrTable["SBFXW"] = a64Enc{format: a64FBitfield2, op: 0x13000000} // ---- conditional compare: sf 1 1 101001 0 imm5/Rm cond op2 Rn nzcv ---- a64InstrTable["CCMP"] = a64Enc{format: a64FCondCmp, op: 0xfa400000} a64InstrTable["CCMN"] = a64Enc{format: a64FCondCmp, op: 0xba400000} a64InstrTable["CCMPW"] = a64Enc{format: a64FCondCmp, op: 0x7a400000} a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000} // ---- system operations ---- for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} { a64InstrTable[m] = a64Enc{format: a64FSys} } // ---- compare/test and branch ---- a64InstrTable["CBZ"] = a64Enc{format: a64FBranch19, op: 0xb4000000} a64InstrTable["CBZW"] = a64Enc{format: a64FBranch19, op: 0x34000000} a64InstrTable["CBNZ"] = a64Enc{format: a64FBranch19, op: 0xb5000000} a64InstrTable["CBNZW"] = a64Enc{format: a64FBranch19, op: 0x35000000} a64InstrTable["TBZ"] = a64Enc{format: a64FTestBranch, op: 0x36000000} a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000} // ---- load/store pair (signed offset) ---- a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000} a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000} a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000} a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000} a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000} a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000} // ---- acquire/release loads and stores ---- a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00} a64InstrTable["LDARB"] = a64Enc{format: a64FAcqRel, op: 0x08dffc00} a64InstrTable["LDARH"] = a64Enc{format: a64FAcqRel, op: 0x48dffc00} a64InstrTable["LDARW"] = a64Enc{format: a64FAcqRel, op: 0x88dffc00} a64InstrTable["STLR"] = a64Enc{format: a64FAcqRel, op: 0xc89ffc00} a64InstrTable["STLRB"] = a64Enc{format: a64FAcqRel, op: 0x089ffc00} a64InstrTable["STLRH"] = a64Enc{format: a64FAcqRel, op: 0x489ffc00} a64InstrTable["STLRW"] = a64Enc{format: a64FAcqRel, op: 0x889ffc00} // ---- LSE atomics with acquire and release semantics ---- // CAS carries a preset fixed op field and a real Rs; the LDADD/LDCLR/ // LDOR/SWP families leave Rs free for the returned value. lse := map[string]uint32{ "CASALD": 0xc8e0fc00, "CASALW": 0x88e0fc00, "LDADDALD": 0xf8e00000, "LDADDALW": 0xb8e00000, "LDCLRALB": 0x38e01000, "LDCLRALW": 0xb8e01000, "LDCLRALD": 0xf8e01000, "LDORALB": 0x38e03000, "LDORALW": 0xb8e03000, "LDORALD": 0xf8e03000, "SWPALB": 0x38e08000, "SWPALW": 0xb8e08000, "SWPALD": 0xf8e08000, } for m, op := range lse { a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} } // The remaining width and ordering spellings of the same shapes, and the // CAS compare-and-swap family, word-verified against go tool asm. lseMore := map[string]uint32{ "LDADDAB": 0x38a00000, "LDADDAH": 0x78a00000, "LDADDALB": 0x38e00000, "LDADDALH": 0x78e00000, "LDADDLB": 0x38600000, "LDADDLD": 0xf8600000, "LDADDLH": 0x78600000, "LDADDLW": 0xb8600000, "LDCLRAB": 0x38a01000, "LDCLRAH": 0x78a01000, "LDCLRALH": 0x78e01000, "LDCLRB": 0x38201000, "LDCLRD": 0xf8201000, "LDCLRH": 0x78201000, "LDCLRLB": 0x38601000, "LDCLRLD": 0xf8601000, "LDCLRLH": 0x78601000, "LDCLRLW": 0xb8601000, "LDCLRW": 0xb8201000, "LDEORAB": 0x38a02000, "LDEORAD": 0xf8a02000, "LDEORAH": 0x78a02000, "LDEORALB": 0x38e02000, "LDEORALH": 0x78e02000, "LDEORAW": 0xb8a02000, "LDEORB": 0x38202000, "LDEORD": 0xf8202000, "LDEORH": 0x78202000, "LDEORLB": 0x38602000, "LDEORLD": 0xf8602000, "LDEORLH": 0x78602000, "LDEORLW": 0xb8602000, "LDEORW": 0xb8202000, "LDORAB": 0x38a03000, "LDORAD": 0xf8a03000, "LDORAH": 0x78a03000, "LDORALH": 0x78e03000, "LDORAW": 0xb8a03000, "LDORB": 0x38203000, "LDORD": 0xf8203000, "LDORH": 0x78203000, "LDORLB": 0x38603000, "LDORLD": 0xf8603000, "LDORLH": 0x78603000, "LDORLW": 0xb8603000, "LDORW": 0xb8203000, "SWPAB": 0x38a08000, "SWPAD": 0xf8a08000, "SWPAH": 0x78a08000, "SWPALH": 0x78e08000, "SWPAW": 0xb8a08000, "SWPB": 0x38208000, "SWPH": 0x78208000, "SWPLB": 0x38608000, "SWPLD": 0xf8608000, "SWPLH": 0x78608000, "SWPLW": 0xb8608000, "CASAD": 0xc8e07c00, "CASALB": 0x08e0fc00, "CASLW": 0x88a0fc00, } for m, op := range lseMore { a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} } // ---- carry-setting/carry-using arithmetic and widening multiply ---- // MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate // register preset to ZR (bits 14:10 = 11111). dpsrExtra := map[string]uint32{ "ADC": 0x9a000000, "ADCW": 0x1a000000, "ADCS": 0xba000000, "ADCSW": 0x3a000000, "SBC": 0xda000000, "SBCW": 0x5a000000, "SBCS": 0xfa000000, "SBCSW": 0x7a000000, // MNEG/MSUB and NGC/SBC with the complementing register preset to ZR. "MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00, "NGC": 0xda000000, "NGCW": 0x5a000000, "NGCS": 0xfa000000, "NGCSW": 0x7a000000, "NEGSW": 0x6b000000, "MUL": 0x9b007c00, "MULW": 0x1b007c00, "SMULH": 0x9b407c00, "UMULH": 0x9bc07c00, } for m, op := range dpsrExtra { a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op} } // ---- crypto, 2-register (Rn, Rd) and 3-register (Rm, Rn, Rd) forms ---- crypto2 := map[string]uint32{ "AESD": 0x4e285800, "AESE": 0x4e284800, "AESIMC": 0x4e287800, "AESMC": 0x4e286800, "SHA1H": 0x5e280800, "SHA1SU1": 0x5e281800, "SHA256SU0": 0x5e282800, "SHA512SU0": 0xcec08000, } for m, op := range crypto2 { a64InstrTable[m] = a64Enc{format: a64FCrypto2, op: op} } crypto3 := map[string]uint32{ "SHA1C": 0x5e000000, "SHA1P": 0x5e001000, "SHA1M": 0x5e002000, "SHA1SU0": 0x5e003000, "SHA256H": 0x5e004000, "SHA256H2": 0x5e005000, "SHA256SU1": 0x5e006000, "SHA512H": 0xce608000, "SHA512H2": 0xce608400, "SHA512SU1": 0xce608800, } for m, op := range crypto3 { a64InstrTable[m] = a64Enc{format: a64FCrypto3, op: op} } // ---- arrangement-aware SIMD, see a64SimdVTable and a64SimdV2Table ---- a64InstrTable["VEOR3"] = a64Enc{format: a64FSIMDV4, op: 0xce000000} a64InstrTable["VBCAX"] = a64Enc{format: a64FSIMDV4, op: 0xce200000} a64InstrTable["VXAR"] = a64Enc{format: a64FSIMDV4, op: 0xce800000} a64InstrTable["VEXT"] = a64Enc{format: a64FSIMDV4, op: 0x2e000000} a64InstrTable["VTBL"] = a64Enc{format: a64FVTBL} a64InstrTable["VDUP"] = a64Enc{format: a64FDUP} a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000} a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000} a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000} a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10} a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10} a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10} a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10} a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10} a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10} a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10} a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10} a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10} a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST} a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VST1"] = a64Enc{format: a64FVLDST} a64InstrTable["VST1.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VLD1R"] = a64Enc{format: a64FVLDST} a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST} a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1} } // a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word, // the set of arrangements it accepts as a bitmask over the a64Arr index and, // for instructions that exist at a single arrangement and carry that // arrangement's bits inside the base already, the fixed flag. type a64SimdVSpec struct { base uint32 arrs uint16 fixed bool } // a64Arr names the vector arrangements the encoders deal with, indexed by // a64Arr. The source spellings put the element letter first: B8, H4, S2, // D1 and the 128-bit halves B16, H8, S4, D2. const ( a64Arr8B = iota a64Arr16B a64Arr4H a64Arr8H a64Arr2S a64Arr4S a64Arr2D a64ArrD1 a64ArrQ1 a64ArrCount ) // a64ArrNames maps an arrangement to its source spelling (element letter // first, as the toolchain writes it). var a64ArrNames = [a64ArrCount]string{ a64Arr8B: "B8", a64Arr16B: "B16", a64Arr4H: "H4", a64Arr8H: "H8", a64Arr2S: "S2", a64Arr4S: "S4", a64Arr2D: "D2", a64ArrD1: "D1", a64ArrQ1: "Q1", } // a64ArrIndex resolves a source spelling to its a64Arr index, -1 when // unknown. func a64ArrIndex(s string) int { for i, n := range a64ArrNames { if n == s { return i } } return -1 } // a64ElemLetter reports whether s is a bare element spelling (B, H, S, D, Q) // as it appears in element operands such as V13.S[0]. func a64ElemLetter(s string) bool { switch s { case "B", "H", "S", "D", "Q": return true } return false } // fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms // accept: H, S and D widths for the pairwise data-processing, H and S for // the across-vector reductions. var fpSimdArrs = uint16(1< or // MSR Rn, ; the source register rides bits 4:0. var a64MSRRegOps = map[string]uint32{ "NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420, "ELR_EL1": 0xd5184020, } // a64MSROps maps the system register names GOROOT writes to their fixed // word; the immediate rides CRm at bits 11:8 and Rt is the fixed 11111. var a64MSROps = map[string]uint32{ "SPSel": 0xd50040a0, "DAIFSet": 0xd50340c0, "DAIFClr": 0xd50340e0, "DIT": 0xd5034040, } // a64PRFOps maps the prefetch operation names to their prfop immediate // (word = 0xf9800000 | Rn<<5 | prfop). var a64PRFOps = map[string]int{ "PLDL1KEEP": 0x00, "PLDL1STRM": 0x01, "PLDL2KEEP": 0x02, "PLDL2STRM": 0x03, "PLDL3KEEP": 0x04, "PLDL3STRM": 0x05, "PLIL1KEEP": 0x08, "PLIL1STRM": 0x09, "PLIL2KEEP": 0x0a, "PLIL2STRM": 0x0b, "PLIL3KEEP": 0x0c, "PLIL3STRM": 0x0d, "PSTL1KEEP": 0x10, "PSTL1STRM": 0x11, "PSTL2KEEP": 0x12, "PSTL2STRM": 0x13, "PSTL3KEEP": 0x14, "PSTL3STRM": 0x15, } // a64VLD1Base holds the fixed words of the multi-register structure // accesses, indexed by register count 1..4, before the Q and size bits. // Post-index spellings add 0x9f0000 (post bit and Rm = 11111). var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000} var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000} // a64Vec is a parsed vector operand: the register number, the arrangement // ("" when the operand spells none) and, for element forms, the lane index. type a64Vec struct { reg int arr string idx int hasIdx bool } // a64VecReg parses a vector register operand: V0..V31 (F0..F31 as an alias, // the same architectural registers the scalar floating-point spellings use), // optionally with an arrangement suffix such as V0.B16 and, for element // forms, a lane index such as V13.S[0]. It reports ok=false for anything // else, including X/W and R spellings, which the toolchain's vector // operands reject as well. func a64VecReg(name string) (v a64Vec, ok bool) { s := strings.TrimSpace(name) if i := strings.IndexByte(s, '.'); i >= 0 { v.arr = strings.TrimSpace(s[i+1:]) s = s[:i] } if v.arr != "" { // Element form: B[3], S[2] and friends. if j := strings.IndexByte(v.arr, '['); j >= 0 { k := strings.LastIndexByte(v.arr, ']') if k < j { return v, false } n, err := strconv.Atoi(strings.TrimSpace(v.arr[j+1 : k])) if err != nil || n < 0 { return v, false } v.idx, v.hasIdx = n, true v.arr = strings.TrimSpace(v.arr[:j]) } if a64ArrIndex(v.arr) < 0 && !a64ElemLetter(v.arr) { return v, false } } if len(s) < 2 || (s[0] != 'V' && s[0] != 'F') { return v, false } n := 0 for i := 1; i < len(s); i++ { if s[i] < '0' || s[i] > '9' { return v, false } n = n*10 + int(s[i]-'0') } if n > 31 { return v, false } v.reg = n return v, true } // a64ElemField encodes a lane index for the copy/insert group: imm5 = the // index shifted by the element scale, with the scale's own bit set. B gets // shift 1 (the Q bit rides elsewhere), H shift 2, S shift 3 and D shift 4. func a64ElemField(arr string, idx int) (uint32, bool) { var shift, low uint32 switch arr { case "B8", "B16", "B": shift, low = 1, 1 case "H4", "H8", "H": shift, low = 2, 2 case "S2", "S4", "S": shift, low = 3, 4 case "D1", "D2", "D": shift, low = 4, 8 default: return 0, false } if idx < 0 || idx >= 1<<(5-shift) { return 0, false } return uint32(idx)<= len(ops) || !strings.HasPrefix(strings.TrimSpace(ops[start].Raw), "[") { return nil, 0, false } end = start for end < len(ops) { if strings.HasSuffix(strings.TrimSpace(ops[end].Raw), "]") { break } end++ } if end >= len(ops) { return nil, 0, false } for i := start; i <= end; i++ { s := strings.TrimSpace(ops[i].Raw) s = strings.TrimPrefix(s, "[") s = strings.TrimSuffix(s, "]") if s == "" && len(ops) > start+1 { return nil, 0, false } for part := range strings.SplitSeq(s, ",") { v, ok := a64VecReg(part) if !ok { return nil, 0, false } vs = append(vs, v) } } return vs, end, true } // ---- load/store helper tables ---- // a64LSType describes the load/store parameters for a MOV width mnemonic. type a64LSType struct { size int // 0=byte, 1=half, 2=word, 3=dword V int // 0=integer, 1=FP opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP } // a64LoadTable maps MOV width mnemonics to their load/store encoding parameters. // For loads, opc selects signed vs unsigned; for stores, we flip the opc. var a64LoadTable = map[string]a64LSType{ "MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset) "MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned) "MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit) "MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned) "MOVH": {1, 0, 2}, // LDRSH (16-bit signed) "MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned) "MOVB": {0, 0, 2}, // LDRSB (8-bit signed) "FMOVS": {2, 1, 1}, // LDR S (32-bit FP) "FMOVD": {3, 1, 1}, // LDR D (64-bit FP) } // a64StoreOpc returns the store opc for a given load type: integer and FP // stores both encode opc=00 (the load's signedness bit sits in opc[1], which // the store form clears; FP registers are selected by V, not opc). func a64StoreOpc(t a64LSType) int { return 0 } // arm64RegClass discriminates integer (R), floating-point (F) registers for // the MOV pseudo-instruction. type arm64RegClass int const ( arm64ClsNone arm64RegClass = iota arm64ClsGR arm64ClsFP ) // arm64RegClassOf reports the register class of a register operand name. func arm64RegClassOf(name string) arm64RegClass { switch { case name == "": return arm64ClsNone case len(name) >= 1 && name[0] == 'F': return arm64ClsFP default: return arm64ClsGR } } // arm64Movcon returns the shift (in units of 16 bits) at which a non-zero // 16-bit chunk of v sits, or -1 if v cannot be represented as a single // MOVZ/MOVN immediate. This is the Go toolchain's movcon function. func arm64Movcon(v int64) int { for s := 0; s < 64; s += 16 { if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 { return s } } return -1 }