// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) // SPDX-License-Identifier: BSD-3-Clause package asm // arm64 (AArch64) instruction encoding. // // The encoder is data-driven: each mnemonic maps to an instruction format and // an opcode constant, and the format selects the bit layout. The opcode // constants and formats are transcribed from the Go toolchain's own arm64 // backend (cmd/internal/obj/arm64), so the emitted bytes match `go tool asm` // exactly, the ground-truth oracle for the verify suite. // // All AArch64 instructions are 32 bits, little-endian. The formats used here // (per the ARM Architecture Reference Manual): // // DP-shifted-reg sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | 0<<21 | Rm<<16 | imm6<<10 | Rn<<5 | Rd // DP-immediate sf<<31 | op<<30 | S<<29 | 0x11<<24 | imm12<<10 | Rn<<5 | Rd // Logical-imm sf<<31 | opc<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | Rn<<5 | Rd // Move-wide sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd // Load/store size<<30 | 0x7<<27 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt // LDST-unscaled size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<5 | Rt (actually imm9<<12 | Rn<<5 | Rt) // LDST-pair opc<<30 | 0x5<<27 | V<<26 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt // Branch-imm 0<<31 | 0x5<<26 | imm26 (B) // Branch-imm 1<<31 | 0x5<<26 | imm26 (BL) // Branch-cond 0x2A<<25 | imm19<<5 | cond (B.cond) // Uncond-branch 0x6B<<25 | opc<<21 | Rn<<5 | Rd (BR/BLR/RET) // ADR/ADRP p<<31 | 0x10<<24 | immlo<<29 | immhi<<5 | Rd import "maps" // arm64RegNum returns the 5-bit register number for an AArch64 register name: // R0-R30 (integer), F0-F31 (floating point), and the ABI aliases the // runtime's assembly uses. Returns -1 for an unrecognised name. func arm64RegNum(name string) int { switch name { case "R0": return 0 case "R1": return 1 case "R2": return 2 case "R3": return 3 case "R4": return 4 case "R5": return 5 case "R6": return 6 case "R7": return 7 case "R8": return 8 case "R9": return 9 case "R10": return 10 case "R11": return 11 case "R12": return 12 case "R13": return 13 case "R14": return 14 case "R15": return 15 case "R16": return 16 case "R17": return 17 case "R18": return 18 case "R19": return 19 case "R20": return 20 case "R21": return 21 case "R22": return 22 case "R23": return 23 case "R24": return 24 case "R25": return 25 case "R26", "REGCTXT", "CTXT": return 26 case "R27", "REGTMP", "TMP": return 27 case "R28", "REGG", "g": return 28 case "R29", "FP": return 29 case "R30", "LR", "LINK": return 30 case "R31", "ZR": return 31 case "SP": return 31 // SP and ZR share encoding 31; context determines meaning } // F0-F31. if len(name) >= 1 && name[0] == 'F' { n := 0 for i := 1; i < len(name); i++ { if name[i] < '0' || name[i] > '9' { return -1 } n = n*10 + int(name[i]-'0') } if n <= 31 { return n } } return -1 } // ---- format helpers ---- // a64wordLE encodes a uint32 as 4 little-endian bytes. func a64wordLE(w uint32) []byte { return []byte{byte(w), byte(w >> 8), byte(w >> 16), byte(w >> 24)} } // a64WordsLE concatenates one or more instruction words as little-endian bytes. func a64WordsLE(ws ...uint32) []byte { var out []byte for _, w := range ws { out = append(out, a64wordLE(w)...) } return out } // ---- data-processing (immediate) ---- // a64AddSub encodes an ADD/SUB (immediate) instruction: // sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | Rn<<5 | Rd. func a64AddSub(sf, op, S, sh, imm12, rn, rd uint32) uint32 { return sf<<31 | op<<30 | S<<29 | 0x11<<24 | sh<<22 | imm12<<10 | rn<<5 | rd } // ---- move wide ---- // a64MoveWide encodes a MOVZ/MOVK/MOVN instruction: // sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | Rd. func a64MoveWide(sf, opc, hw, imm16, rd uint32) uint32 { return sf<<31 | opc<<29 | 0x25<<23 | hw<<21 | imm16<<5 | rd } // ---- load/store (unsigned immediate, scaled) ---- // a64LSU encodes a load/store register (unsigned immediate, scaled): // size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | Rn<<5 | Rt. // (0x39<<24 encodes bits 29:24 = 111001, the scaled unsigned offset form.) func a64LSU(size, V, opc, imm12, rn, rt uint32) uint32 { return size<<30 | 0x39<<24 | V<<26 | opc<<22 | imm12<<10 | rn<<5 | rt } // ---- load/store (unscaled immediate) ---- // a64LSUnscaled encodes a load/store register (unscaled immediate, 9-bit signed): // size<<30 | 0x7<<27 | V<<26 | opc<<22 | 0<<12 | imm9<<12 | Rn<<5 | Rt. // Note: the 0<<24 distinguishes unscaled from the pre/post-index forms. func a64LSUnscaled(size, V, opc int, imm9 int32, rn, rt int) uint32 { return uint32(size)<<30 | 7<<27 | uint32(V)<<26 | uint32(opc)<<22 | (uint32(imm9)&0x1FF)<<12 | uint32(rn&31)<<5 | uint32(rt&31) } // ---- load/store pair ---- // a64LSP encodes a load/store pair instruction (signed offset): // opc<<30 | 0x5<<27 | V<<26 | 2<<23 | L<<22 | imm7<<15 | Rt2<<10 | Rn<<5 | Rt. // opc: 0=32-bit, 1=reserved, 2=64-bit. V: 0=integer, 1=FP/SIMD. // L: 0=store, 1=load. imm7 is the signed scaled offset (÷8 for 64-bit pairs). func a64LSP(opc, V, L uint32, imm7 int32, rt2, rn, rt uint32) uint32 { return opc<<30 | 5<<27 | V<<26 | 2<<23 | L<<22 | (uint32(imm7)&0x7F)<<15 | rt2<<10 | rn<<5 | rt } // ---- branches ---- // a64Branch encodes an unconditional branch (B/BL): // op<<31 | 0x5<<26 | imm26. func a64Branch(op uint32, imm26 int32) uint32 { return op<<31 | 5<<26 | (uint32(imm26) & 0x03FFFFFF) } // a64BranchCond encodes a conditional branch (B.cond): // 0x2A<<25 | imm19<<5 | cond. func a64BranchCond(imm19 int32, cond uint32) uint32 { return 0x2A<<25 | (uint32(imm19)&0x7FFFF)<<5 | cond&0xF } // a64UncondBranch encodes an unconditional branch register (BR/BLR/RET): // 0x6B<<25 | opc<<21 | 0x1F<<16 | Rn<<5 | Rd. // opc: 0=BR, 1=BLR, 2=RET. For RET, Rn defaults to LR(30). func a64UncondBranch(opc, rn, rd uint32) uint32 { return 0x6B<<25 | opc<<21 | 0x1F<<16 | rn<<5 | rd } // ---- ADR/ADRP ---- // a64ADR encodes an ADR instruction (p=0) or ADRP instruction (p=1): // p<<31 | immlo<<29 | 0x10<<24 | immhi<<5 | Rd. func a64ADR(p uint32, immhi int32, immlo uint32, rd uint32) uint32 { return p<<31 | immlo<<29 | 0x10<<24 | (uint32(immhi)&0x7FFFF)<<5 | rd } // ---- system ---- // a64NOP encodes a NOP: 0xd503201f. const a64NOP uint32 = 0xd503201f // a64BRK encodes a BRK instruction: 0xd4200000 | imm16<<5. func a64BRK(imm16 uint32) uint32 { return 0xd4200000 | imm16<<5 } // ---- condition codes ---- const ( a64CondEQ = 0x0 a64CondNE = 0x1 a64CondCS = 0x2 a64CondHS = 0x2 a64CondCC = 0x3 a64CondLO = 0x3 a64CondMI = 0x4 a64CondPL = 0x5 a64CondVS = 0x6 a64CondVC = 0x7 a64CondHI = 0x8 a64CondLS = 0x9 a64CondGE = 0xa a64CondLT = 0xb a64CondGT = 0xc a64CondLE = 0xd ) // arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes. var arm64CondMap = map[string]uint32{ "EQ": a64CondEQ, "NE": a64CondNE, "CS": a64CondCS, "HS": a64CondHS, "CC": a64CondCC, "LO": a64CondLO, "MI": a64CondMI, "PL": a64CondPL, "VS": a64CondVS, "VC": a64CondVC, "HI": a64CondHI, "LS": a64CondLS, "GE": a64CondGE, "LT": a64CondLT, "GT": a64CondGT, "LE": a64CondLE, } // ---- instruction format tags ---- type a64Format uint8 const ( a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc. a64FMovWide // move wide: MOVZ, MOVN, MOVK a64FBranch // unconditional branch (B/BL) a64FBranchCond // conditional branch (B.cond) a64FUncondBranch // unconditional branch register (BR/BLR/RET) a64FADR // ADR/ADRP a64FEXTR // EXTR a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10 a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT* a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc. a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc. a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL a64FCRC32 // CRC32 a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR a64FLSE // LSE atomics: LDADD, CAS, SWP a64FSIMD3 // SIMD 3-operand: VADD, VSUB, VMUL ) // a64Enc is one instruction's encoding: its bit layout (format) and the // opcode constant, positioned at its exact bit range. type a64Enc struct { format a64Format op uint32 // the pre-positioned opcode bits } // a64InstrTable maps AArch64 mnemonics (as the Go assembler spells them) to // their encoding. The base integer, memory, floating-point and SIMD // instruction sets are covered. var a64InstrTable = map[string]a64Enc{} func init() { // ---- data-processing (shifted register) ---- // Format: sf<<31 | op<<30 | S<<29 | 0x0b<<24 | shift<<22 | Rm<<16 | imm6<<10 | Rn<<5 | Rd dpsr := map[string]uint32{ // Add/Sub "ADD": 1<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=1, op=0, S=0 (64-bit default) "ADDW": 0<<31 | 0<<30 | 0<<29 | 0x0b<<24, // sf=0 "ADDS": 1<<31 | 0<<30 | 1<<29 | 0x0b<<24, "ADDSW": 0<<31 | 0<<30 | 1<<29 | 0x0b<<24, "SUB": 1<<31 | 1<<30 | 0<<29 | 0x0b<<24, "SUBW": 0<<31 | 1<<30 | 0<<29 | 0x0b<<24, "SUBS": 1<<31 | 1<<30 | 1<<29 | 0x0b<<24, "SUBSW": 0<<31 | 1<<30 | 1<<29 | 0x0b<<24, // Logical (shifted register) "AND": 1<<31 | 0<<29 | 0x0a<<24, "ANDW": 0<<31 | 0<<29 | 0x0a<<24, "BIC": 1<<31 | 0<<29 | 0x0a<<24 | 1<<21, "BICW": 0<<31 | 0<<29 | 0x0a<<24 | 1<<21, "ORR": 1<<31 | 1<<29 | 0x0a<<24, "ORRW": 0<<31 | 1<<29 | 0x0a<<24, "ORN": 1<<31 | 1<<29 | 0x0a<<24 | 1<<21, "ORNW": 0<<31 | 1<<29 | 0x0a<<24 | 1<<21, "EOR": 1<<31 | 2<<29 | 0x0a<<24, "EORW": 0<<31 | 2<<29 | 0x0a<<24, "EON": 1<<31 | 2<<29 | 0x0a<<24 | 1<<21, "EONW": 0<<31 | 2<<29 | 0x0a<<24 | 1<<21, "ANDS": 1<<31 | 3<<29 | 0x0a<<24, "ANDSW": 0<<31 | 3<<29 | 0x0a<<24, "BICS": 1<<31 | 3<<29 | 0x0a<<24 | 1<<21, "BICSW": 0<<31 | 3<<29 | 0x0a<<24 | 1<<21, // Divide (data-processing 2 source): the opcode occupies bits 15:10 // of the 0xd6<<21 fixed field, UDIV=0b0010 and SDIV=0b0011 (ARM ARM // "Data-processing (2 source)"; the toolchain spells them OPDP2(2) // and OPDP2(3)). sf=1 selects the X forms. "SDIV": 1<<31 | 0xd6<<21 | 3<<10, "SDIVW": 0<<31 | 0xd6<<21 | 3<<10, "UDIV": 1<<31 | 0xd6<<21 | 2<<10, "UDIVW": 0<<31 | 0xd6<<21 | 2<<10, // Conditional select "CSEL": 1<<31 | 0<<29 | 0x1d<<24 | 0<<10, "CSELW": 0<<31 | 0<<29 | 0x1d<<24 | 0<<10, "CSINC": 1<<31 | 0<<29 | 0x1d<<24 | 1<<10, "CSINCW": 0<<31 | 0<<29 | 0x1d<<24 | 1<<10, "CSINV": 1<<31 | 0<<29 | 0x1d<<24 | 2<<10, "CSINVW": 0<<31 | 0<<29 | 0x1d<<24 | 2<<10, "CSNEG": 1<<31 | 0<<29 | 0x1d<<24 | 3<<10, "CSNEGW": 0<<31 | 0<<29 | 0x1d<<24 | 3<<10, } for m, op := range dpsr { a64InstrTable[m] = a64Enc{format: a64FDPSR, op: op} } // Aliases that map to the same encoding as their target. a64InstrTable["CMP"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]} a64InstrTable["CMPW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBSW"]} a64InstrTable["CMN"] = a64Enc{format: a64FDPSR, op: dpsr["ADDS"]} a64InstrTable["CMNW"] = a64Enc{format: a64FDPSR, op: dpsr["ADDSW"]} a64InstrTable["TST"] = a64Enc{format: a64FDPSR, op: dpsr["ANDS"]} a64InstrTable["TSTW"] = a64Enc{format: a64FDPSR, op: dpsr["ANDSW"]} a64InstrTable["NEG"] = a64Enc{format: a64FDPSR, op: dpsr["SUB"]} a64InstrTable["NEGW"] = a64Enc{format: a64FDPSR, op: dpsr["SUBW"]} a64InstrTable["NEGS"] = a64Enc{format: a64FDPSR, op: dpsr["SUBS"]} a64InstrTable["MVN"] = a64Enc{format: a64FDPSR, op: dpsr["ORN"]} a64InstrTable["MVNW"] = a64Enc{format: a64FDPSR, op: dpsr["ORNW"]} a64InstrTable["MOV"] = a64Enc{format: a64FDPSR, op: dpsr["ORR"]} a64InstrTable["MOVW"] = a64Enc{format: a64FDPSR, op: dpsr["ORRW"]} // ---- shifts ---- // The mnemonic serves both forms: with an immediate the aliases of the // data-processing (immediate) group apply (ARM ARM "Shifts"), with a // register the data-processing (2 source) LSLV/LSRV/ASRV/RORV. The op // field carries the immediate-alias base; encodeARM64Shift derives both // it and the two-source opcode. Identities, W = 64 (X) or 32 (W): // // LSL $sh, Rn, Rd = UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh // LSR $sh, Rn, Rd = UBFM Rd, Rn, #sh, #(W-1) // ASR $sh, Rn, Rd = SBFM Rd, Rn, #sh, #(W-1) // ROR $sh, Rn, Rd = EXTR Rd, Rn, Rn, #sh shifts := map[string]a64Enc{ "LSL": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X "LSLW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W "LSR": {format: a64FShift, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}, // UBFM X "LSRW": {format: a64FShift, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22}, // UBFM W "ASR": {format: a64FShift, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}, // SBFM X "ASRW": {format: a64FShift, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}, // SBFM W "ROR": {format: a64FShift, op: 1<<31 | 0x27<<23 | 1<<22}, // EXTR X "RORW": {format: a64FShift, op: 0<<31 | 0x27<<23 | 0<<22}, // EXTR W } maps.Copy(a64InstrTable, shifts) // ---- multiply accumulate ---- // MADD/MSUB Rm, Ra, Rn, Rd: sf 00 11011 o0(15) Rm Ra Rn Rd. The // toolchain's optab has no shorter row, so all four operands are // mandatory, and Ra is the SECOND operand. a64InstrTable["MADD"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24} a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24} a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15} a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15} // ---- move wide ---- // MOVZ/MOVN/MOVK a64InstrTable["MOVZ"] = a64Enc{format: a64FMovWide, op: 1<<31 | 2<<29 | 0x25<<23} a64InstrTable["MOVZW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 2<<29 | 0x25<<23} a64InstrTable["MOVN"] = a64Enc{format: a64FMovWide, op: 1<<31 | 0<<29 | 0x25<<23} a64InstrTable["MOVNW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 0<<29 | 0x25<<23} a64InstrTable["MOVK"] = a64Enc{format: a64FMovWide, op: 1<<31 | 3<<29 | 0x25<<23} a64InstrTable["MOVKW"] = a64Enc{format: a64FMovWide, op: 0<<31 | 3<<29 | 0x25<<23} // ---- ADR/ADRP ---- a64InstrTable["ADR"] = a64Enc{format: a64FADR, op: 0} a64InstrTable["ADRP"] = a64Enc{format: a64FADR, op: 1} // Load/store mnemonics never enter this table: the MOV pseudo-instruction // dispatch handles them through a64LoadTable, which also carries the store // opcode (integer and FP stores both use opc=00, differing only in V). // ---- branches ---- a64InstrTable["B"] = a64Enc{format: a64FBranch, op: 0<<31 | 5<<26} a64InstrTable["BL"] = a64Enc{format: a64FBranch, op: 1<<31 | 5<<26} // Conditional branches. condBranches := map[string]uint32{ "BEQ": 0x0, "BNE": 0x1, "BCS": 0x2, "BHS": 0x2, "BCC": 0x3, "BLO": 0x3, "BMI": 0x4, "BPL": 0x5, "BVS": 0x6, "BVC": 0x7, "BHI": 0x8, "BLS": 0x9, "BGE": 0xa, "BLT": 0xb, "BGT": 0xc, "BLE": 0xd, } for name, cond := range condBranches { a64InstrTable[name] = a64Enc{format: a64FBranchCond, op: 0x2A<<25 | cond} } // Unconditional branch register (BR/BLR/RET). a64InstrTable["BR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 0<<21} a64InstrTable["BLR"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 1<<21} a64InstrTable["RET"] = a64Enc{format: a64FUncondBranch, op: 0x6B<<25 | 2<<21} // ---- system ---- // NOP/NOOP/UNDEF are spelled out in encodeARM64Instr's pseudo switch, // so they carry no table entry; a64NOP and a64BRK are the encoders. // ---- EXTR ---- a64InstrTable["EXTR"] = a64Enc{format: a64FEXTR, op: 1<<31 | 0x27<<23 | 1<<22} a64InstrTable["EXTRW"] = a64Enc{format: a64FEXTR, op: 0<<31 | 0x27<<23 | 0<<22} // ---- bitfield ---- a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22} a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22} a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} a64InstrTable["UBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} a64InstrTable["BFI"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFIW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 2<<29 | 0x26<<23 | 0<<22} a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} // ---- FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL ---- fp3 := map[string]uint32{ "FADDS": 0x1e202800, "FADDD": 0x1e602800, "FSUBS": 0x1e203800, "FSUBD": 0x1e603800, "FMULS": 0x1e200800, "FMULD": 0x1e600800, "FDIVS": 0x1e201800, "FDIVD": 0x1e601800, "FMAXS": 0x1e204800, "FMAXD": 0x1e604800, "FMINS": 0x1e205800, "FMIND": 0x1e605800, "FMAXNMS": 0x1e206800, "FMAXNMD": 0x1e606800, "FMINNMS": 0x1e207800, "FMINNMD": 0x1e607800, "FNMULS": 0x1e208800, "FNMULD": 0x1e608800, } for m, op := range fp3 { a64InstrTable[m] = a64Enc{format: a64FFP3, op: op} } // ---- FP unary (Rn, Rd): FMOV reg-reg, FABS, FNEG, FSQRT, FCVT, FRINT* ---- fp1 := map[string]uint32{ "FMOVS": 0x1e204000, "FMOVD": 0x1e604000, "FABSS": 0x1e20c000, "FABSD": 0x1e60c000, "FNEGS": 0x1e214000, "FNEGD": 0x1e614000, "FSQRTS": 0x1e21c000, "FSQRTD": 0x1e61c000, "FCVTSD": 0x1e22c000, "FCVTDS": 0x1e624000, "FRINTNS": 0x1e244000, "FRINTND": 0x1e644000, "FRINTPS": 0x1e24c000, "FRINTPD": 0x1e64c000, "FRINTMS": 0x1e254000, "FRINTMD": 0x1e654000, "FRINTZS": 0x1e25c000, "FRINTZD": 0x1e65c000, "FRINTAS": 0x1e264000, "FRINTAD": 0x1e664000, "FRINTXS": 0x1e274000, "FRINTXD": 0x1e674000, "FRINTIS": 0x1e27c000, "FRINTID": 0x1e67c000, } for m, op := range fp1 { a64InstrTable[m] = a64Enc{format: a64FFPUnary, op: op} } // ---- FP 4-operand FMA (Ra, Rm, Rn, Rd) ---- fp4 := map[string]uint32{ "FMADDS": 0x1f000000, "FMADDD": 0x1f400000, "FMSUBS": 0x1f008000, "FMSUBD": 0x1f408000, "FNMADDS": 0x1f200000, "FNMADDD": 0x1f600000, "FNMSUBS": 0x1f208000, "FNMSUBD": 0x1f608000, } for m, op := range fp4 { a64InstrTable[m] = a64Enc{format: a64FFP4, op: op} } // ---- FP compare (Rm, Rn or #0, Rn) ---- fpcmp := map[string]uint32{ "FCMPS": 0x1e202000, "FCMPD": 0x1e602000, "FCMPES": 0x1e202010, "FCMPED": 0x1e602010, } for m, op := range fpcmp { a64InstrTable[m] = a64Enc{format: a64FFPCmp, op: op} } // ---- FP conditional compare (Rm, Rn, #nzcv, cond) ---- fpccmp := map[string]uint32{ "FCCMPS": 0x1e200400, "FCCMPD": 0x1e600400, "FCCMPES": 0x1e200410, "FCCMPED": 0x1e600410, } for m, op := range fpccmp { a64InstrTable[m] = a64Enc{format: a64FFPCCmp, op: op} } // ---- FP conditional select (Rm, Rn, Rd, cond) ---- a64InstrTable["FCSELS"] = a64Enc{format: a64FFPSel, op: 0x1e200c00} a64InstrTable["FCSELD"] = a64Enc{format: a64FFPSel, op: 0x1e600c00} // ---- FP ↔ integer conversion ---- fpcvt := map[string]uint32{ "FCVTZSD": 0x9e780000, "FCVTZSDW": 0x1e780000, "FCVTZSS": 0x9e380000, "FCVTZSSW": 0x1e380000, "FCVTZUD": 0x9e790000, "FCVTZUDW": 0x1e790000, "FCVTZUS": 0x9e390000, "FCVTZUSW": 0x1e390000, "SCVTFD": 0x9e620000, "SCVTFS": 0x9e220000, "SCVTFWD": 0x1e620000, "SCVTFWS": 0x1e220000, "UCVTFD": 0x9e630000, "UCVTFS": 0x9e230000, "UCVTFWD": 0x1e630000, "UCVTFWS": 0x1e230000, } for m, op := range fpcvt { a64InstrTable[m] = a64Enc{format: a64FFPCvt, op: op} } // FMOV between GP and FP registers needs no table entry: the MOV // pseudo-instruction dispatches it by operand class (encodeARM64RegMove). // ---- conditional select: CSEL, CSINC, CSINV, CSNEG ---- csel := map[string]uint32{ "CSEL": 0x9a800000, "CSELW": 0x1a800000, "CSINC": 0x9a800400, "CSINCW": 0x1a800400, "CSINV": 0xda800000, "CSINVW": 0x5a800000, "CSNEG": 0xda800400, "CSNEGW": 0x5a800400, } for m, op := range csel { a64InstrTable[m] = a64Enc{format: a64FCSEL, op: op} } // Aliases a64InstrTable["CSET"] = a64Enc{format: a64FCSEL, op: 0x9a800400} a64InstrTable["CSETW"] = a64Enc{format: a64FCSEL, op: 0x1a800400} a64InstrTable["CSETM"] = a64Enc{format: a64FCSEL, op: 0xda800000} a64InstrTable["CSETMW"] = a64Enc{format: a64FCSEL, op: 0x5a800000} a64InstrTable["CINC"] = a64Enc{format: a64FCSEL, op: 0x9a800400} a64InstrTable["CINCW"] = a64Enc{format: a64FCSEL, op: 0x1a800400} a64InstrTable["CINV"] = a64Enc{format: a64FCSEL, op: 0xda800000} a64InstrTable["CINVW"] = a64Enc{format: a64FCSEL, op: 0x5a800000} a64InstrTable["CNEG"] = a64Enc{format: a64FCSEL, op: 0xda800400} a64InstrTable["CNEGW"] = a64Enc{format: a64FCSEL, op: 0x5a800400} // ---- CRC32 ---- crc32 := map[string]uint32{ "CRC32B": 0x1ac04000, "CRC32H": 0x1ac04400, "CRC32W": 0x1ac04800, "CRC32X": 0x9ac04c00, "CRC32CB": 0x1ac05000, "CRC32CH": 0x1ac05400, "CRC32CW": 0x1ac05800, "CRC32CX": 0x9ac05c00, } for m, op := range crc32 { a64InstrTable[m] = a64Enc{format: a64FCRC32, op: op} } // ---- exclusive load/store ---- a64InstrTable["LDXR"] = a64Enc{format: a64FExcl, op: 0xc85f7c00} a64InstrTable["LDXRB"] = a64Enc{format: a64FExcl, op: 0x085f7c00} a64InstrTable["LDXRH"] = a64Enc{format: a64FExcl, op: 0x485f7c00} a64InstrTable["LDXRW"] = a64Enc{format: a64FExcl, op: 0x885f7c00} a64InstrTable["LDAXR"] = a64Enc{format: a64FExcl, op: 0xc85ffc00} a64InstrTable["LDAXRB"] = a64Enc{format: a64FExcl, op: 0x085ffc00} a64InstrTable["LDAXRH"] = a64Enc{format: a64FExcl, op: 0x485ffc00} a64InstrTable["LDAXRW"] = a64Enc{format: a64FExcl, op: 0x885ffc00} a64InstrTable["STXR"] = a64Enc{format: a64FExcl, op: 0xc8007c00} a64InstrTable["STXRB"] = a64Enc{format: a64FExcl, op: 0x08007c00} a64InstrTable["STXRH"] = a64Enc{format: a64FExcl, op: 0x48007c00} a64InstrTable["STXRW"] = a64Enc{format: a64FExcl, op: 0x88007c00} a64InstrTable["STLXR"] = a64Enc{format: a64FExcl, op: 0xc800fc00} a64InstrTable["STLXRB"] = a64Enc{format: a64FExcl, op: 0x0800fc00} a64InstrTable["STLXRH"] = a64Enc{format: a64FExcl, op: 0x4800fc00} a64InstrTable["STLXRW"] = a64Enc{format: a64FExcl, op: 0x8800fc00} // ---- LSE atomics ---- a64InstrTable["LDADDD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDB"] = a64Enc{format: a64FLSE, op: 0<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["LDADDH"] = a64Enc{format: a64FLSE, op: 1<<30 | 0x1c1<<21 | 0x00<<10} a64InstrTable["CASD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x45<<21 | 0x1f<<10} a64InstrTable["CASW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x45<<21 | 0x1f<<10} a64InstrTable["SWPD"] = a64Enc{format: a64FLSE, op: 3<<30 | 0x1c1<<21 | 0x20<<10} a64InstrTable["SWPW"] = a64Enc{format: a64FLSE, op: 2<<30 | 0x1c1<<21 | 0x20<<10} // ---- SIMD basics ---- a64InstrTable["VADD"] = a64Enc{format: a64FSIMD3, op: 0x0e208400} a64InstrTable["VSUB"] = a64Enc{format: a64FSIMD3, op: 0x2e208400} a64InstrTable["VMUL"] = a64Enc{format: a64FSIMD3, op: 0x0e209c00} } // ---- load/store helper tables ---- // a64LSType describes the load/store parameters for a MOV width mnemonic. type a64LSType struct { size int // 0=byte, 1=half, 2=word, 3=dword V int // 0=integer, 1=FP opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP } // a64LoadTable maps MOV width mnemonics to their load/store encoding parameters. // For loads, opc selects signed vs unsigned; for stores, we flip the opc. var a64LoadTable = map[string]a64LSType{ "MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset) "MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned) "MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit) "MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned) "MOVH": {1, 0, 2}, // LDRSH (16-bit signed) "MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned) "MOVB": {0, 0, 2}, // LDRSB (8-bit signed) "FMOVS": {2, 1, 1}, // LDR S (32-bit FP) "FMOVD": {3, 1, 1}, // LDR D (64-bit FP) } // a64StoreOpc returns the store opc for a given load type: integer and FP // stores both encode opc=00 (the load's signedness bit sits in opc[1], which // the store form clears; FP registers are selected by V, not opc). func a64StoreOpc(t a64LSType) int { return 0 } // arm64RegClass discriminates integer (R), floating-point (F) registers for // the MOV pseudo-instruction. type arm64RegClass int const ( arm64ClsNone arm64RegClass = iota arm64ClsGR arm64ClsFP ) // arm64RegClassOf reports the register class of a register operand name. func arm64RegClassOf(name string) arm64RegClass { switch { case name == "": return arm64ClsNone case len(name) >= 1 && name[0] == 'F': return arm64ClsFP default: return arm64ClsGR } } // arm64Movcon returns the shift (in units of 16 bits) at which a non-zero // 16-bit chunk of v sits, or -1 if v cannot be represented as a single // MOVZ/MOVN immediate. This is the Go toolchain's movcon function. func arm64Movcon(v int64) int { for s := 0; s < 64; s += 16 { if (uint64(v) &^ (uint64(0xFFFF) << uint(s))) == 0 { return s } } return -1 }