From 2747fce7d328692bbe21947367ee2cd93e1f20e2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Fri, 2 Oct 2026 17:58:55 +0200 Subject: [PATCH] feat(asm): encode the arm64 system registers and structure loads --- asm/arm64_assemble.go | 542 ++++++++++++++++++++++++++++++----- asm/arm64_encode.go | 160 ++++++++++- asm/arm64_sysregs.go | 588 ++++++++++++++++++++++++++++++++++++++ asm/arm64_sysregs_test.go | 164 +++++++++++ 4 files changed, 1375 insertions(+), 79 deletions(-) create mode 100644 asm/arm64_sysregs.go create mode 100644 asm/arm64_sysregs_test.go diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index e9478de..24afc96 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -458,10 +458,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64Excl(mnem, enc.op, ops) } - // LSE atomics (LDADD, CAS, SWP). + // LSE atomics (LDADD, CAS, SWP) and the compare-and-swap pair. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE { return encodeARM64LSEAtom(mnem, enc.op, ops) } + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCASP { + return encodeARM64CASP(mnem, enc.op, ops) + } // Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield { @@ -555,9 +558,12 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 // VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so // this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist // only in the zero-immediate form (a64SimdVZero), so they route here with - // an empty register-form spec. + // an empty register-form spec. VSQSHL/VUQSHL keep their shift-by- + // immediate route when the first operand is an immediate. if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 { - return encodeARM64SimdV(mnem, spec, ops) + if !arm64SimdShiftImmRoute(mnem, ops) { + return encodeARM64SimdV(mnem, spec, ops) + } } // Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV). @@ -565,6 +571,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64SimdV2(mnem, spec, ops) } + // Narrow/long/wide SIMD families whose size and Q bits read off one + // designated operand (VXTN, VSXTL, VUADDW, VUMULL, VSHRN, VSSHLL, VFCVTN + // and friends). + if spec, ok := a64SimdNLTable[mnem]; ok { + return encodeARM64SimdNL(mnem, spec, ops) + } + // SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR, // VEXT). if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 { @@ -587,6 +600,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64ShiftImm(mnem, enc.op, ops) } + // SIMD move immediate: VMOVI $imm8, Vd.B8/B16. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVMoviImm { + return encodeARM64MoviImm(ops) + } + // VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit { return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits) @@ -2568,6 +2586,263 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil } +// encodeARM64CASP encodes the compare-and-swap pair: CASP (Rs, Rs+1), (Rn), +// (Rt, Rt+1). Both pairs must start on an even register and be contiguous; +// the second register of each pair rides no encoding field. +func encodeARM64CASP(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects (Rs, Rs+1), (Rn), (Rt, Rt+1), got %d operands", mnem, len(ops)) + } + rs, rs1, ok := arm64PairOf(ops[0]) + if !ok { + return nil, fmt.Errorf("%s expects a source register pair (Rs, Rs+1)", mnem) + } + rn, err := arm64ExclMem(mnem, ops[1]) + if err != nil { + return nil, err + } + rt, rt1, ok := arm64PairOf(ops[2]) + if !ok { + return nil, fmt.Errorf("%s expects a destination register pair (Rt, Rt+1)", mnem) + } + if rs&1 != 0 { + return nil, fmt.Errorf("%s: source register pair must start from an even register", mnem) + } + if rt&1 != 0 { + return nil, fmt.Errorf("%s: destination register pair must start from an even register", mnem) + } + if rs != rs1-1 { + return nil, fmt.Errorf("%s: source register pair must be contiguous", mnem) + } + if rt != rt1-1 { + return nil, fmt.Errorf("%s: destination register pair must be contiguous", mnem) + } + if rt == 31 { + return nil, fmt.Errorf("%s: illegal destination register", mnem) + } + return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil +} + +// encodeARM64MoviImm encodes VMOVI $imm8, Vd.B8/B16: the modified-immediate +// form of the SIMD move (asm7.go case 86). Only the byte arrangements exist +// and the immediate is one unsigned byte. +func encodeARM64MoviImm(ops []*ast.Operand) ([]byte, error) { + if len(ops) != 2 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("VMOVI expects $immediate, Vd.") + } + vd, ok := arm64VecOf(ops[1]) + if !ok || vd.hasIdx || (vd.arr != "B8" && vd.arr != "B16") { + return nil, fmt.Errorf("VMOVI: destination arrangement must be B8 or B16") + } + imm := arm64Imm64(ops[0]) + if imm < 0 || imm > 255 { + return nil, fmt.Errorf("VMOVI: immediate constant %d out of range (0..255)", imm) + } + q := uint32(0) + if vd.arr == "B16" { + q = 1 << 30 + } + w := 0x0f00e400 | q | uint32(imm>>5&7)<<16 | uint32(imm&0x1f)<<5 | uint32(vd.reg) + return a64wordLE(w), nil +} + +// arm64SimdShiftImmRoute reports whether a mnemonic carries both a shift-by- +// immediate and a register form and the operands spell the immediate one: the +// dedicated shift route keeps them. +func arm64SimdShiftImmRoute(mnem string, ops []*ast.Operand) bool { + if mnem != "VSQSHL" && mnem != "VUQSHL" { + return false + } + return len(ops) > 0 && isImmOperand(ops[0]) +} + +// arm64SimdNLArr describes one arrangement for the narrow/long/wide families: +// the element width in bytes and whether the spelling names the 128-bit form. +func arm64SimdNLArr(arr string) (esize int, wide bool, ok bool) { + switch arr { + case "B8", "B16": + return 1, arr == "B16", true + case "H4", "H8": + return 2, arr == "H8", true + case "S2", "S4": + return 4, arr == "S4", true + case "D1", "D2": + return 8, arr == "D2", true + } + return 0, false, false +} + +// arm64SimdLongPair validates a long pairing (source narrow, destination +// wide): the destination element is twice the source's, the destination is +// always spelled the wide way (H8/S4/D2) and the source carries the 128-bit +// flag exactly for the .2 spellings. +func arm64SimdLongPair(mnem, src, dst string, two bool) error { + se, sw, ok1 := arm64SimdNLArr(src) + de, dw, ok2 := arm64SimdNLArr(dst) + if !ok1 || !ok2 || de != 2*se { + return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst) + } + if !dw || sw != two { + return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem) + } + return nil +} + +// arm64SimdNarrowPair validates a narrow pairing (source wide, destination +// narrow): the mirror image of arm64SimdLongPair. +func arm64SimdNarrowPair(mnem, src, dst string, two bool) error { + se, sw, ok1 := arm64SimdNLArr(src) + de, dw, ok2 := arm64SimdNLArr(dst) + if !ok1 || !ok2 || se != 2*de { + return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst) + } + if !sw || dw != two { + return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem) + } + return nil +} + +// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide +// instruction contributes: the driving arrangement's size and Q bits, or for +// the FCVT family only the Q bit, whose size field is fixed in the base. +func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 { + if spec.qonly { + if two { + return 1 << 30 + } + return 0 + } + return a64ArrBits[a64ArrIndex(drive)] +} + +// encodeARM64SimdNL encodes the narrow/long/wide SIMD families +// (a64SimdNLTable): XTN and FCVTN narrow a wide source, SXTL and FCVTL +// lengthen, the MULL/MLAL/MLSL group multiplies long, UADDW widens, and the +// SSHLL/USHLL and SHRN shifts carry their immediate in the immh:immb field. +// The size and Q bits read off the designated driving operand, and the .2 +// spellings force the 128-bit side through their own arrangement. +func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]byte, error) { + two := strings.HasSuffix(mnem, "2") + + // vecAt parses operand i as a vector register with an arrangement. + vecAt := func(i int) (a64Vec, bool) { + if i >= len(ops) { + return a64Vec{}, false + } + v, ok := arm64VecOf(ops[i]) + if !ok || v.hasIdx { + return a64Vec{}, false + } + return v, true + } + + switch spec.form { + case a64NLTwoNarrow, a64NLTwoLong: + if len(ops) != 2 { + return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) + } + vn, ok1 := vecAt(0) + vd, ok2 := vecAt(1) + if !ok1 || !ok2 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + var drive string + var pairErr error + if spec.form == a64NLTwoNarrow { + // XTN/FCVTN: wide source into a narrow destination; the + // arrangement bits follow the destination. + drive, pairErr = vd.arr, arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two) + } else { + // SXTL/UXTL/FCVTL: narrow source into a wide destination; the + // arrangement bits follow the source. + drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two) + } + if pairErr != nil { + return nil, pairErr + } + arrBits := arm64SimdNLArrBits(spec, drive, two) + return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil + case a64NLThreeLongMul, a64NLThreeWide: + if len(ops) != 3 { + return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops)) + } + vm, ok1 := vecAt(0) + vn, ok2 := vecAt(1) + vd, ok3 := vecAt(2) + if !ok1 || !ok2 || !ok3 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + var drive string + var pairErr error + if spec.form == a64NLThreeWide { + // UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one; + // the arrangement bits follow the wide side. + drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two) + } else { + // MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the + // wide one; the arrangement bits follow the narrow source. + drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two) + } + if pairErr != nil { + return nil, pairErr + } + if spec.form == a64NLThreeWide && vd.arr != vn.arr { + return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vn.arr, vd.arr) + } + if spec.form == a64NLThreeLongMul && vm.arr != vn.arr { + return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr) + } + arrBits := arm64SimdNLArrBits(spec, drive, two) + return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil + case a64NLThreeLongShift, a64NLThreeNarrowShift: + if len(ops) != 3 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects ($shift, Vn., Vd.)", mnem) + } + sh := arm64Imm64(ops[0]) + vn, ok1 := vecAt(1) + vd, ok2 := vecAt(2) + if !ok1 || !ok2 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + if spec.form == a64NLThreeLongShift { + // SSHLL/USHLL: the narrow source drives the immediate's size + // (immh:immb = esize + shift), so the arrangement bits carry + // the Q bit alone: the size field belongs to immh, and ORing + // the source's size bits into it would collide with immb. + if err := arm64SimdLongPair(mnem, vn.arr, vd.arr, two); err != nil { + return nil, err + } + se, _, _ := arm64SimdNLArr(vn.arr) + esize := se * 8 + if sh < 0 || sh >= int64(esize) { + return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1) + } + var qBit uint32 + if two { + qBit = 1 << 30 + } + return a64wordLE(spec.base | uint32(esize+int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil + } + // SHRN: the narrow destination drives the immediate's size + // (immh:immb = esize - shift over the wide source element), so the + // arrangement bits carry the Q bit alone, exactly as above. + if err := arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two); err != nil { + return nil, err + } + se, _, _ := arm64SimdNLArr(vn.arr) + esize := se * 8 + if sh < 1 || sh >= int64(esize) { + return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize-1) + } + var qBit uint32 + if two { + qBit = 1 << 30 + } + return a64wordLE(spec.base | uint32(esize-int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil + } + return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) +} + // encodeARM64DP1 encodes a data-processing (1 source) instruction: // RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd. func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { @@ -2584,7 +2859,8 @@ func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, err // encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the // target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31 -// selecting the page form. +// selecting the page form. An n(PC) operand resolves to the instruction's +// own address: the toolchain rewrites it away and encodes displacement 0. func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) @@ -2593,14 +2869,17 @@ func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offset if rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - target := resolve(arm64Label(ops[0])) - targetOff, ok := offsets[target] - if !ok { - return nil, fmt.Errorf("undefined label %q", target) - } - rel := int64(targetOff - pc) - if rel < -(1<<20) || rel >= 1<<20 { - return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target) + var rel int64 + if _, pcRel := arm64PCRelOffset(ops[0]); !pcRel { + target := resolve(arm64Label(ops[0])) + targetOff, ok := offsets[target] + if !ok { + return nil, fmt.Errorf("undefined label %q", target) + } + rel = int64(targetOff - pc) + if rel < -(1<<20) || rel >= 1<<20 { + return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target) + } } return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil } @@ -2879,11 +3158,16 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, // BRK [$imm16] SVC $imm16 // DMB|DSB|ISB $imm4 DC , Rn // MRS , Rd MSR $imm4, -// PRFM (Rn), $imm| +// PRFM (Rn), $imm| RPRFM (Rn), Rm, +// SYS $imm[, Rn] SYSL $imm, Rd +// TLBI [, Rn] SB, PACIASP, PACIBSP +// +// The system registers, TLBI and DC aliases and the range-prefetch operations +// come from the toolchain's own data tables in arm64_sysregs.go. func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { // Operand-less returns and pointer-authentication hints. if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0, - "AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, + "AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "PACIASP": 0xd503233f, "PACIBSP": 0xd503237f, "AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f, "YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f, "SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok { @@ -2919,6 +3203,12 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { } base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem] return a64wordLE(base | uint32(v)<<8), nil + case "SB": + // Speculation barrier: DSB with a fixed barrier domain. + if len(ops) != 0 { + return nil, fmt.Errorf("%s expects no operand", mnem) + } + return a64wordLE(0xd50330ff), nil case "HINT": if len(ops) != 1 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $immediate", mnem) @@ -2957,7 +3247,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("DC expects , Rn") } - base, ok := a64DCOps[operandRegName(ops[0])] + inst, ok := a64DCOps2[operandRegName(ops[0])] if !ok { return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0])) } @@ -2965,51 +3255,115 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { if rn < 0 { return nil, fmt.Errorf("DC: invalid register operand") } - return a64wordLE(base | uint32(rn)&31), nil + w := 0xd5080000 | inst.op1<<16 | 7<<12 | inst.cm<<8 | inst.op2<<5 + return a64wordLE(w | uint32(rn)&31), nil + case "TLBI": + // The register operand is optional: TLBI VMALLE1IS alone means ZR. + if len(ops) != 1 && len(ops) != 2 { + return nil, fmt.Errorf("TLBI expects [, Rn]") + } + inst, ok := a64TLBIOps[operandRegName(ops[0])] + if !ok { + return nil, fmt.Errorf("TLBI: unknown operation %q", operandRegName(ops[0])) + } + rt := 31 + if len(ops) == 2 { + if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 { + return nil, fmt.Errorf("TLBI: invalid register operand") + } + } + w := 0xd5080000 | inst.op1<<16 | 8<<12 | inst.cm<<8 | inst.op2<<5 + return a64wordLE(w | uint32(rt)&31), nil + case "SYS", "SYSL": + // SYS $imm[, Rn] / SYSL $imm, Rd: the immediate packs + // op1<<16 | CRn<<12 | CRm<<8 | op2<<5, the register defaults to ZR. + if len(ops) != 1 && len(ops) != 2 { + return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem) + } + if len(ops) == 1 && mnem == "SYSL" { + return nil, fmt.Errorf("SYSL expects $immediate, Rd") + } + if !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem) + } + imm := arm64Imm64(ops[0]) + if imm < 0 || imm&^0x7FFE0 != 0 { + return nil, fmt.Errorf("%s: illegal SYS argument %d", mnem, imm) + } + rt := 31 + if len(ops) == 2 { + if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 { + return nil, fmt.Errorf("%s: invalid register operand", mnem) + } + } + base := uint32(0xd5080000) + if mnem == "SYSL" { + base = 0xd5280000 + } + return a64wordLE(base | uint32(imm) | uint32(rt)&31), nil case "MRS": if len(ops) != 2 { return nil, fmt.Errorf("MRS expects , Rd") } - base, ok := a64MRSOps[operandRegName(ops[0])] + reg, ok := a64SysRegs[operandRegName(ops[0])] if !ok { return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0])) } + if !reg.read { + return nil, fmt.Errorf("MRS: system register is not readable: %q", operandRegName(ops[0])) + } rd := arm64RegNum(operandRegName(ops[1])) if rd < 0 { return nil, fmt.Errorf("MRS: invalid register operand") } - return a64wordLE(base | uint32(rd)&31), nil + return a64wordLE(0xd5300000 | reg.v | uint32(rd)&31), nil case "MSR": if len(ops) != 2 { return nil, fmt.Errorf("MSR expects $immediate, or Rn, ") } - if !isImmOperand(ops[0]) { - // Register form: MSR Rn, (the a64MSRRegOps words). - base, ok := a64MSRRegOps[operandRegName(ops[1])] + if isImmOperand(ops[0]) { + v := arm64Imm64(ops[0]) + // The PSTATE fields keep their dedicated immediate form. + if base, ok := a64MSROps[operandRegName(ops[1])]; ok { + if v < 0 || v > 15 { + return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v) + } + return a64wordLE(base | uint32(v)<<8 | 31), nil + } + // A $0 against a full system register writes it from ZR, exactly + // the way the toolchain preprocesses the constant away; any other + // immediate is the PSTATE-form error. + if v != 0 { + return nil, fmt.Errorf("MSR: illegal PSTATE field for immediate move: %q", operandRegName(ops[1])) + } + reg, ok := a64SysRegs[operandRegName(ops[1])] if !ok { return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1])) } - rs := arm64RegNum(operandRegName(ops[0])) - if rs < 0 { - return nil, fmt.Errorf("MSR: invalid source register") + if !reg.write { + return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1])) } - return a64wordLE(base | uint32(rs)&31), nil + return a64wordLE(0xd5100000 | reg.v | 31), nil } - base, ok := a64MSROps[operandRegName(ops[1])] + // Register form: MSR Rn, . + reg, ok := a64SysRegs[operandRegName(ops[1])] if !ok { return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1])) } - v := arm64Imm64(ops[0]) - if v < 0 || v > 15 { - return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v) + if !reg.write { + return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1])) } - return a64wordLE(base | uint32(v)<<8 | 31), nil + rs := arm64RegNum(operandRegName(ops[0])) + if rs < 0 { + return nil, fmt.Errorf("MSR: invalid source register") + } + return a64wordLE(0xd5100000 | reg.v | uint32(rs)&31), nil case "PRFM": if len(ops) != 2 { return nil, fmt.Errorf("PRFM expects (Rn), $immediate|") } rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{}) - if rn < 0 || off != 0 { + if rn < 0 || off < 0 || off%8 != 0 || off/8 >= 4096 { return nil, fmt.Errorf("PRFM: invalid memory operand") } var prfop int64 @@ -3025,7 +3379,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { } prfop = int64(p) } - return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil + return a64wordLE(0xf9800000 | uint32(off/8)<<10 | uint32(rn)<<5 | uint32(prfop)), nil + case "RPRFM": + // RPRFM (Rn), Rm, : the 6-bit operation scatters across + // bits 15, 13, 12 and 2:0 (asm7.go case 110). + if len(ops) != 3 { + return nil, fmt.Errorf("RPRFM expects (Rn), Rm, ") + } + rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{}) + if rn < 0 || off != 0 { + return nil, fmt.Errorf("RPRFM: invalid memory operand") + } + rm := arm64RegNum(operandRegName(ops[1])) + if rm < 0 { + return nil, fmt.Errorf("RPRFM: invalid register operand") + } + var op uint64 + if isImmOperand(ops[2]) { + op = uint64(arm64Imm64(ops[2])) + if op > 63 { + return nil, fmt.Errorf("RPRFM: range prefetch immediate %d out of range (0..63)", op) + } + } else { + v, ok := a64RPRFOps[operandRegName(ops[2])] + if !ok { + return nil, fmt.Errorf("RPRFM: unknown range prefetch operation %q", operandRegName(ops[2])) + } + op = uint64(v) + } + scatter := (op&(1<<5))<<10 | (op&(1<<4))<<9 | (op&(1<<3))<<9 | op&7 + return a64wordLE(0xf8a04818 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(scatter)), nil } return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem) } @@ -3448,7 +3831,7 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) { return nil, fmt.Errorf("invalid destination register in VTBL") } for i, t := range ts { - if t.hasIdx || t.reg != ts[0].reg+i { + if t.hasIdx || (ts[0].reg+i)&31 != t.reg { return nil, fmt.Errorf("VTBL table registers must be consecutive") } } @@ -3605,15 +3988,17 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) { // encodeARM64VLDST encodes the SIMD structure loads and stores: // -// VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn) -// VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn) -// VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane) -// VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3] +// VLD1|2|3|4 (Rn), [Vt.arr, ...] VST1|2|3|4 [Vt.arr, ...], (Rn) +// VLD1|2|3|4.P off(Rn), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], off(Rn) +// VLD1|2|3|4.P (Rn)(Rm), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], (Rn)(Rm) +// VLD1|2|3|4R (Rn), [Vt.arr, ...] (replicating loads) +// VLD1 off(Rn), Vt.T[i] VST1 Vt.T[i], off(Rn) (one lane) // -// The post-index forms set the post bit and Rm = 11111. A spelled offset -// rides along (the encoding ignores it; the toolchain only checks that it -// matches the access size), and a multi-register post-index list takes no -// offset at all, the increment following from the list. +// The post-index forms set the post bit and carry Rm: 11111 for an immediate +// increment, the spelled register for (Rn)(Rm). A register list may wrap +// around V31: the toolchain checks only (first+i) mod 32. A spelled offset +// rides along on the one-lane forms (the toolchain only checks that it +// matches the access size). func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) { load := strings.HasPrefix(mnem, "VLD") @@ -3651,24 +4036,31 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err if off != 0 && post == 0 { return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off) } - - // VLD1R loads one register and replicates; VLD4R loads four. - if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") { - want := 1 - base := uint32(0x0d40c000) - if strings.HasPrefix(mnem, "VLD4R") { - want, base = 4, 0x0d60e000 + // The post-index increment: 11111 for an immediate offset, else the + // spelled (Rn)(Rm) register. + rm := 31 + if post != 0 { + if idx := ops[memIdx].Addr.Index; idx != "" { + if rm = arm64RegNum(idx); rm < 0 { + return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx) + } } - if len(vs) != want { - return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want) + } + + // The replicating loads: VLD1R through VLD4R load one register and + // replicate it across the whole list. + if base := strings.TrimSuffix(mnem, ".P"); load && strings.HasSuffix(base, "R") && len(base) == 5 { + n := int(base[3] - '0') + if len(vs) != n { + return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n) } size, q, ok := a64ArrSizeQ(vs[0].arr) if !ok { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) } - w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg) + w := a64VLDNReplicate[n] | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg) if post != 0 { - w |= 1<<23 | 0x1f<<16 + w |= 1<<23 | uint32(rm)<<16 } return a64wordLE(w), nil } @@ -3677,7 +4069,7 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err return nil, fmt.Errorf("%s expects a list of one to four registers", mnem) } for i, v := range vs { - if v.hasIdx || v.reg != vs[0].reg+i { + if v.hasIdx || (vs[0].reg+i)&31 != v.reg { return nil, fmt.Errorf("%s: register list must be consecutive", mnem) } _, _, okArr := a64ArrSizeQ(v.arr) @@ -3689,21 +4081,36 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err if !ok { return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr) } - base := a64VLD1Base[len(vs)] + n := len(vs) + base := a64VLD1Base[n] if !load { - base = a64VST1Base[len(vs)] + base = a64VST1Base[n] + } + // VLD2/VLD3/VLD4 and VST2/VST3/VST4 name the register count in the + // mnemonic and carry their own opcode fields. The count digit sits at + // index 3 of the mnemonic (VLD2, VST3.P, ...), before any .P suffix. + if stem := strings.TrimSuffix(mnem, ".P"); len(stem) >= 4 && stem[3] >= '2' && stem[3] <= '4' { + n := int(stem[3] - '0') + if n != len(vs) { + return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n) + } + if load { + base = a64VLDNBase[n] + } else { + base = a64VSTNBase[n] + } } postBits := uint32(0) if post != 0 { - postBits = 0x9f0000 + postBits = 1<<23 | uint32(rm)<<16 } return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil } // encodeARM64VLDSTLane encodes the one-lane structure forms: -// VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and -// VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the -// toolchain's table and is rejected. +// VLD1 off(Rn), Vt.T[i] and VST1 Vt.T[i], off(Rn); the post-index spellings +// add the post bit and Rm: 11111 for an immediate increment, the spelled +// register for (Rn)(Rm). func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) { if len(ops) != 2 { return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem) @@ -3713,14 +4120,19 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo if rn < 0 { return nil, fmt.Errorf("%s: invalid memory operand", mnem) } - if !load && post == 0 { - return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem) + rm := 31 + if post != 0 { + if idx := ops[memIdx].Addr.Index; idx != "" { + if rm = arm64RegNum(idx); rm < 0 { + return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx) + } + } } w := uint32(0x0d400000) switch strings.ToUpper(v.arr) { case "B": - // Index at bits 12:10 (the size field doubles as the low index bits). - w |= uint32(v.idx) << 10 + // Index<3> rides bit 30, index<2:0> the size field at bits 12:10. + w |= uint32(v.idx&7)<<10 | uint32(v.idx>>3&1)<<30 case "H": // Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11. w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30 @@ -3734,12 +4146,12 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr) } // The base carries bit 22 (L) set; a store clears it. The post-index - // forms add bit 23 and Rm = 11111. + // forms add bit 23 and Rm. if !load { w &^= 1 << 22 } if post != 0 { - w |= 1<<23 | 0x1f<<16 + w |= 1<<23 | uint32(rm)<<16 } return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil } diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index 8b44aa2..150bd9b 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -374,6 +374,7 @@ const ( a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD a64FAcqRel // acquire/release: LDAR family, STLR family a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM + a64FCASP // compare and swap pair: CASP a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ... a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ... a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ... @@ -384,6 +385,7 @@ const ( a64FDUP // SIMD element moves: VDUP, VMOV with element indices a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI + a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16 a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool) ) @@ -770,7 +772,7 @@ func init() { a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000} // ---- system operations ---- - for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} { + for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} { a64InstrTable[m] = a64Enc{format: a64FSys} } @@ -783,12 +785,20 @@ func init() { a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000} // ---- load/store pair (signed offset) ---- + // The scale column of a64LoadTable does not reach the pair forms, so each + // entry states its own access width through the imm7 divisor the pair + // encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q). a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000} a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000} + a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000} a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000} a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000} a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000} a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000} + a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000} + a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000} + a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000} + a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000} // ---- acquire/release loads and stores ---- a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00} @@ -806,11 +816,23 @@ func init() { lse := map[string]uint32{ "CASALD": 0xc8e0fc00, "CASALW": 0x88e0fc00, + "CASB": 0x08a07c00, + "CASAB": 0x08e07c00, + "CASH": 0x48a07c00, + "CASLD": 0xc8a0fc00, + "CASLH": 0x48a0fc00, + "CASAW": 0x88e07c00, + "CASAD": 0xc8e07c00, + "CASALH": 0x48e07c00, "LDADDALD": 0xf8e00000, "LDADDALW": 0xb8e00000, + "LDADDAD": 0xf8a00000, + "LDADDAW": 0xb8a00000, "LDCLRALB": 0x38e01000, "LDCLRALW": 0xb8e01000, "LDCLRALD": 0xf8e01000, + "LDCLRAD": 0xf8a01000, + "LDCLRAW": 0xb8a01000, "LDORALB": 0x38e03000, "LDORALW": 0xb8e03000, "LDORALD": 0xf8e03000, @@ -889,6 +911,12 @@ func init() { a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} } + // Compare and swap pair: the second register of each pair is implicit + // (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset + // fixed field (asm7.go atomicCASP). + a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10} + a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10} + // ---- carry-setting/carry-using arithmetic and widening multiply ---- // MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate // register preset to ZR (bits 14:10 = 11111). @@ -940,11 +968,13 @@ func init() { a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000} a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000} a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000} + a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm} a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10} a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10} a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10} a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10} - a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10} + a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10} + a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10} a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10} a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10} a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10} @@ -957,6 +987,24 @@ func init() { a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST} a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1} + // Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and + // the replicate loads VLD2R/VLD3R, each with the post-index spelling. + a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST} + a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VST2"] = a64Enc{format: a64FVLDST} + a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VST3"] = a64Enc{format: a64FVLDST} + a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1} + a64InstrTable["VST4"] = a64Enc{format: a64FVLDST} + a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1} } // a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word, @@ -1120,6 +1168,78 @@ var a64SimdVTable = map[string]a64SimdVSpec{ "VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only "VPMULL": {0x0e20e000, 1< (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +// arm64 system registers and system-instruction aliases. +// +// The tables are transcribed from the data the Go toolchain itself carries +// (cmd/internal/obj/arm64/sysRegEnc.go and the sysInstFields map of asm7.go), +// which the ARM ARM defines: every system register is the packed field set +// op0<<19 | op1<<16 | CRn<<12 | CRm<<8 | op2<<5, and the read/write flags are +// the toolchain's own access classification. The encoding tables live here so +// the encoder stays testable against the GOROOT testdata word for word. + +// a64SysReg is one system register: the packed encoding fields and the +// directions the register supports. +type a64SysReg struct { + v uint32 + read bool + write bool +} + +// a64SysRegs maps the system register names the toolchain knows to their +// encodings. MRS reads 0xd5300000 | v | Rd and MSR writes +// 0xd5100000 | v | Rt. +var a64SysRegs = map[string]a64SysReg{ + "ACTLR_EL1": a64SysReg{0x181020, true, true}, + "AFSR0_EL1": a64SysReg{0x185100, true, true}, + "AFSR1_EL1": a64SysReg{0x185120, true, true}, + "AIDR_EL1": a64SysReg{0x1900e0, true, false}, + "AMAIR_EL1": a64SysReg{0x18a300, true, true}, + "AMCFGR_EL0": a64SysReg{0x1bd220, true, false}, + "AMCGCR_EL0": a64SysReg{0x1bd240, true, false}, + "AMCNTENCLR0_EL0": a64SysReg{0x1bd280, true, true}, + "AMCNTENCLR1_EL0": a64SysReg{0x1bd300, true, true}, + "AMCNTENSET0_EL0": a64SysReg{0x1bd2a0, true, true}, + "AMCNTENSET1_EL0": a64SysReg{0x1bd320, true, true}, + "AMCR_EL0": a64SysReg{0x1bd200, true, true}, + "AMEVCNTR00_EL0": a64SysReg{0x1bd400, true, true}, + "AMEVCNTR01_EL0": a64SysReg{0x1bd420, true, true}, + "AMEVCNTR02_EL0": a64SysReg{0x1bd440, true, true}, + "AMEVCNTR03_EL0": a64SysReg{0x1bd460, true, true}, + "AMEVCNTR04_EL0": a64SysReg{0x1bd480, true, true}, + "AMEVCNTR05_EL0": a64SysReg{0x1bd4a0, true, true}, + "AMEVCNTR06_EL0": a64SysReg{0x1bd4c0, true, true}, + "AMEVCNTR07_EL0": a64SysReg{0x1bd4e0, true, true}, + "AMEVCNTR08_EL0": a64SysReg{0x1bd500, true, true}, + "AMEVCNTR09_EL0": a64SysReg{0x1bd520, true, true}, + "AMEVCNTR010_EL0": a64SysReg{0x1bd540, true, true}, + "AMEVCNTR011_EL0": a64SysReg{0x1bd560, true, true}, + "AMEVCNTR012_EL0": a64SysReg{0x1bd580, true, true}, + "AMEVCNTR013_EL0": a64SysReg{0x1bd5a0, true, true}, + "AMEVCNTR014_EL0": a64SysReg{0x1bd5c0, true, true}, + "AMEVCNTR015_EL0": a64SysReg{0x1bd5e0, true, true}, + "AMEVCNTR10_EL0": a64SysReg{0x1bdc00, true, true}, + "AMEVCNTR11_EL0": a64SysReg{0x1bdc20, true, true}, + "AMEVCNTR12_EL0": a64SysReg{0x1bdc40, true, true}, + "AMEVCNTR13_EL0": a64SysReg{0x1bdc60, true, true}, + "AMEVCNTR14_EL0": a64SysReg{0x1bdc80, true, true}, + "AMEVCNTR15_EL0": a64SysReg{0x1bdca0, true, true}, + "AMEVCNTR16_EL0": a64SysReg{0x1bdcc0, true, true}, + "AMEVCNTR17_EL0": a64SysReg{0x1bdce0, true, true}, + "AMEVCNTR18_EL0": a64SysReg{0x1bdd00, true, true}, + "AMEVCNTR19_EL0": a64SysReg{0x1bdd20, true, true}, + "AMEVCNTR110_EL0": a64SysReg{0x1bdd40, true, true}, + "AMEVCNTR111_EL0": a64SysReg{0x1bdd60, true, true}, + "AMEVCNTR112_EL0": a64SysReg{0x1bdd80, true, true}, + "AMEVCNTR113_EL0": a64SysReg{0x1bdda0, true, true}, + "AMEVCNTR114_EL0": a64SysReg{0x1bddc0, true, true}, + "AMEVCNTR115_EL0": a64SysReg{0x1bdde0, true, true}, + "AMEVTYPER00_EL0": a64SysReg{0x1bd600, true, false}, + "AMEVTYPER01_EL0": a64SysReg{0x1bd620, true, false}, + "AMEVTYPER02_EL0": a64SysReg{0x1bd640, true, false}, + "AMEVTYPER03_EL0": a64SysReg{0x1bd660, true, false}, + "AMEVTYPER04_EL0": a64SysReg{0x1bd680, true, false}, + "AMEVTYPER05_EL0": a64SysReg{0x1bd6a0, true, false}, + "AMEVTYPER06_EL0": a64SysReg{0x1bd6c0, true, false}, + "AMEVTYPER07_EL0": a64SysReg{0x1bd6e0, true, false}, + "AMEVTYPER08_EL0": a64SysReg{0x1bd700, true, false}, + "AMEVTYPER09_EL0": a64SysReg{0x1bd720, true, false}, + "AMEVTYPER010_EL0": a64SysReg{0x1bd740, true, false}, + "AMEVTYPER011_EL0": a64SysReg{0x1bd760, true, false}, + "AMEVTYPER012_EL0": a64SysReg{0x1bd780, true, false}, + "AMEVTYPER013_EL0": a64SysReg{0x1bd7a0, true, false}, + "AMEVTYPER014_EL0": a64SysReg{0x1bd7c0, true, false}, + "AMEVTYPER015_EL0": a64SysReg{0x1bd7e0, true, false}, + "AMEVTYPER10_EL0": a64SysReg{0x1bde00, true, true}, + "AMEVTYPER11_EL0": a64SysReg{0x1bde20, true, true}, + "AMEVTYPER12_EL0": a64SysReg{0x1bde40, true, true}, + "AMEVTYPER13_EL0": a64SysReg{0x1bde60, true, true}, + "AMEVTYPER14_EL0": a64SysReg{0x1bde80, true, true}, + "AMEVTYPER15_EL0": a64SysReg{0x1bdea0, true, true}, + "AMEVTYPER16_EL0": a64SysReg{0x1bdec0, true, true}, + "AMEVTYPER17_EL0": a64SysReg{0x1bdee0, true, true}, + "AMEVTYPER18_EL0": a64SysReg{0x1bdf00, true, true}, + "AMEVTYPER19_EL0": a64SysReg{0x1bdf20, true, true}, + "AMEVTYPER110_EL0": a64SysReg{0x1bdf40, true, true}, + "AMEVTYPER111_EL0": a64SysReg{0x1bdf60, true, true}, + "AMEVTYPER112_EL0": a64SysReg{0x1bdf80, true, true}, + "AMEVTYPER113_EL0": a64SysReg{0x1bdfa0, true, true}, + "AMEVTYPER114_EL0": a64SysReg{0x1bdfc0, true, true}, + "AMEVTYPER115_EL0": a64SysReg{0x1bdfe0, true, true}, + "AMUSERENR_EL0": a64SysReg{0x1bd260, true, true}, + "APDAKeyHi_EL1": a64SysReg{0x182220, true, true}, + "APDAKeyLo_EL1": a64SysReg{0x182200, true, true}, + "APDBKeyHi_EL1": a64SysReg{0x182260, true, true}, + "APDBKeyLo_EL1": a64SysReg{0x182240, true, true}, + "APGAKeyHi_EL1": a64SysReg{0x182320, true, true}, + "APGAKeyLo_EL1": a64SysReg{0x182300, true, true}, + "APIAKeyHi_EL1": a64SysReg{0x182120, true, true}, + "APIAKeyLo_EL1": a64SysReg{0x182100, true, true}, + "APIBKeyHi_EL1": a64SysReg{0x182160, true, true}, + "APIBKeyLo_EL1": a64SysReg{0x182140, true, true}, + "CCSIDR2_EL1": a64SysReg{0x190040, true, false}, + "CCSIDR_EL1": a64SysReg{0x190000, true, false}, + "CLIDR_EL1": a64SysReg{0x190020, true, false}, + "CNTFRQ_EL0": a64SysReg{0x1be000, true, true}, + "CNTKCTL_EL1": a64SysReg{0x18e100, true, true}, + "CNTP_CTL_EL0": a64SysReg{0x1be220, true, true}, + "CNTP_CVAL_EL0": a64SysReg{0x1be240, true, true}, + "CNTP_TVAL_EL0": a64SysReg{0x1be200, true, true}, + "CNTPCT_EL0": a64SysReg{0x1be020, true, false}, + "CNTPS_CTL_EL1": a64SysReg{0x1fe220, true, true}, + "CNTPS_CVAL_EL1": a64SysReg{0x1fe240, true, true}, + "CNTPS_TVAL_EL1": a64SysReg{0x1fe200, true, true}, + "CNTV_CTL_EL0": a64SysReg{0x1be320, true, true}, + "CNTV_CVAL_EL0": a64SysReg{0x1be340, true, true}, + "CNTV_TVAL_EL0": a64SysReg{0x1be300, true, true}, + "CNTVCT_EL0": a64SysReg{0x1be040, true, false}, + "CONTEXTIDR_EL1": a64SysReg{0x18d020, true, true}, + "CPACR_EL1": a64SysReg{0x181040, true, true}, + "CSSELR_EL1": a64SysReg{0x1a0000, true, true}, + "CTR_EL0": a64SysReg{0x1b0020, true, false}, + "CurrentEL": a64SysReg{0x184240, true, false}, + "DAIF": a64SysReg{0x1b4220, true, true}, + "DBGAUTHSTATUS_EL1": a64SysReg{0x107ec0, true, false}, + "DBGBCR0_EL1": a64SysReg{0x1000a0, true, true}, + "DBGBCR1_EL1": a64SysReg{0x1001a0, true, true}, + "DBGBCR2_EL1": a64SysReg{0x1002a0, true, true}, + "DBGBCR3_EL1": a64SysReg{0x1003a0, true, true}, + "DBGBCR4_EL1": a64SysReg{0x1004a0, true, true}, + "DBGBCR5_EL1": a64SysReg{0x1005a0, true, true}, + "DBGBCR6_EL1": a64SysReg{0x1006a0, true, true}, + "DBGBCR7_EL1": a64SysReg{0x1007a0, true, true}, + "DBGBCR8_EL1": a64SysReg{0x1008a0, true, true}, + "DBGBCR9_EL1": a64SysReg{0x1009a0, true, true}, + "DBGBCR10_EL1": a64SysReg{0x100aa0, true, true}, + "DBGBCR11_EL1": a64SysReg{0x100ba0, true, true}, + "DBGBCR12_EL1": a64SysReg{0x100ca0, true, true}, + "DBGBCR13_EL1": a64SysReg{0x100da0, true, true}, + "DBGBCR14_EL1": a64SysReg{0x100ea0, true, true}, + "DBGBCR15_EL1": a64SysReg{0x100fa0, true, true}, + "DBGBVR0_EL1": a64SysReg{0x100080, true, true}, + "DBGBVR1_EL1": a64SysReg{0x100180, true, true}, + "DBGBVR2_EL1": a64SysReg{0x100280, true, true}, + "DBGBVR3_EL1": a64SysReg{0x100380, true, true}, + "DBGBVR4_EL1": a64SysReg{0x100480, true, true}, + "DBGBVR5_EL1": a64SysReg{0x100580, true, true}, + "DBGBVR6_EL1": a64SysReg{0x100680, true, true}, + "DBGBVR7_EL1": a64SysReg{0x100780, true, true}, + "DBGBVR8_EL1": a64SysReg{0x100880, true, true}, + "DBGBVR9_EL1": a64SysReg{0x100980, true, true}, + "DBGBVR10_EL1": a64SysReg{0x100a80, true, true}, + "DBGBVR11_EL1": a64SysReg{0x100b80, true, true}, + "DBGBVR12_EL1": a64SysReg{0x100c80, true, true}, + "DBGBVR13_EL1": a64SysReg{0x100d80, true, true}, + "DBGBVR14_EL1": a64SysReg{0x100e80, true, true}, + "DBGBVR15_EL1": a64SysReg{0x100f80, true, true}, + "DBGCLAIMCLR_EL1": a64SysReg{0x1079c0, true, true}, + "DBGCLAIMSET_EL1": a64SysReg{0x1078c0, true, true}, + "DBGDTR_EL0": a64SysReg{0x130400, true, true}, + "DBGDTRRX_EL0": a64SysReg{0x130500, true, false}, + "DBGDTRTX_EL0": a64SysReg{0x130500, false, true}, + "DBGPRCR_EL1": a64SysReg{0x101480, true, true}, + "DBGWCR0_EL1": a64SysReg{0x1000e0, true, true}, + "DBGWCR1_EL1": a64SysReg{0x1001e0, true, true}, + "DBGWCR2_EL1": a64SysReg{0x1002e0, true, true}, + "DBGWCR3_EL1": a64SysReg{0x1003e0, true, true}, + "DBGWCR4_EL1": a64SysReg{0x1004e0, true, true}, + "DBGWCR5_EL1": a64SysReg{0x1005e0, true, true}, + "DBGWCR6_EL1": a64SysReg{0x1006e0, true, true}, + "DBGWCR7_EL1": a64SysReg{0x1007e0, true, true}, + "DBGWCR8_EL1": a64SysReg{0x1008e0, true, true}, + "DBGWCR9_EL1": a64SysReg{0x1009e0, true, true}, + "DBGWCR10_EL1": a64SysReg{0x100ae0, true, true}, + "DBGWCR11_EL1": a64SysReg{0x100be0, true, true}, + "DBGWCR12_EL1": a64SysReg{0x100ce0, true, true}, + "DBGWCR13_EL1": a64SysReg{0x100de0, true, true}, + "DBGWCR14_EL1": a64SysReg{0x100ee0, true, true}, + "DBGWCR15_EL1": a64SysReg{0x100fe0, true, true}, + "DBGWVR0_EL1": a64SysReg{0x1000c0, true, true}, + "DBGWVR1_EL1": a64SysReg{0x1001c0, true, true}, + "DBGWVR2_EL1": a64SysReg{0x1002c0, true, true}, + "DBGWVR3_EL1": a64SysReg{0x1003c0, true, true}, + "DBGWVR4_EL1": a64SysReg{0x1004c0, true, true}, + "DBGWVR5_EL1": a64SysReg{0x1005c0, true, true}, + "DBGWVR6_EL1": a64SysReg{0x1006c0, true, true}, + "DBGWVR7_EL1": a64SysReg{0x1007c0, true, true}, + "DBGWVR8_EL1": a64SysReg{0x1008c0, true, true}, + "DBGWVR9_EL1": a64SysReg{0x1009c0, true, true}, + "DBGWVR10_EL1": a64SysReg{0x100ac0, true, true}, + "DBGWVR11_EL1": a64SysReg{0x100bc0, true, true}, + "DBGWVR12_EL1": a64SysReg{0x100cc0, true, true}, + "DBGWVR13_EL1": a64SysReg{0x100dc0, true, true}, + "DBGWVR14_EL1": a64SysReg{0x100ec0, true, true}, + "DBGWVR15_EL1": a64SysReg{0x100fc0, true, true}, + "DCZID_EL0": a64SysReg{0x1b00e0, true, false}, + "DISR_EL1": a64SysReg{0x18c120, true, true}, + "DIT": a64SysReg{0x1b42a0, true, true}, + "DLR_EL0": a64SysReg{0x1b4520, true, true}, + "DSPSR_EL0": a64SysReg{0x1b4500, true, true}, + "ELR_EL1": a64SysReg{0x184020, true, true}, + "ERRIDR_EL1": a64SysReg{0x185300, true, false}, + "ERRSELR_EL1": a64SysReg{0x185320, true, true}, + "ERXADDR_EL1": a64SysReg{0x185460, true, true}, + "ERXCTLR_EL1": a64SysReg{0x185420, true, true}, + "ERXFR_EL1": a64SysReg{0x185400, true, false}, + "ERXMISC0_EL1": a64SysReg{0x185500, true, true}, + "ERXMISC1_EL1": a64SysReg{0x185520, true, true}, + "ERXMISC2_EL1": a64SysReg{0x185540, true, true}, + "ERXMISC3_EL1": a64SysReg{0x185560, true, true}, + "ERXPFGCDN_EL1": a64SysReg{0x1854c0, true, true}, + "ERXPFGCTL_EL1": a64SysReg{0x1854a0, true, true}, + "ERXPFGF_EL1": a64SysReg{0x185480, true, false}, + "ERXSTATUS_EL1": a64SysReg{0x185440, true, true}, + "ESR_EL1": a64SysReg{0x185200, true, true}, + "FAR_EL1": a64SysReg{0x186000, true, true}, + "FPCR": a64SysReg{0x1b4400, true, true}, + "FPSR": a64SysReg{0x1b4420, true, true}, + "GCR_EL1": a64SysReg{0x1810c0, true, true}, + "GMID_EL1": a64SysReg{0x31400, true, false}, + "ICC_AP0R0_EL1": a64SysReg{0x18c880, true, true}, + "ICC_AP0R1_EL1": a64SysReg{0x18c8a0, true, true}, + "ICC_AP0R2_EL1": a64SysReg{0x18c8c0, true, true}, + "ICC_AP0R3_EL1": a64SysReg{0x18c8e0, true, true}, + "ICC_AP1R0_EL1": a64SysReg{0x18c900, true, true}, + "ICC_AP1R1_EL1": a64SysReg{0x18c920, true, true}, + "ICC_AP1R2_EL1": a64SysReg{0x18c940, true, true}, + "ICC_AP1R3_EL1": a64SysReg{0x18c960, true, true}, + "ICC_ASGI1R_EL1": a64SysReg{0x18cbc0, false, true}, + "ICC_BPR0_EL1": a64SysReg{0x18c860, true, true}, + "ICC_BPR1_EL1": a64SysReg{0x18cc60, true, true}, + "ICC_CTLR_EL1": a64SysReg{0x18cc80, true, true}, + "ICC_DIR_EL1": a64SysReg{0x18cb20, false, true}, + "ICC_EOIR0_EL1": a64SysReg{0x18c820, false, true}, + "ICC_EOIR1_EL1": a64SysReg{0x18cc20, false, true}, + "ICC_HPPIR0_EL1": a64SysReg{0x18c840, true, false}, + "ICC_HPPIR1_EL1": a64SysReg{0x18cc40, true, false}, + "ICC_IAR0_EL1": a64SysReg{0x18c800, true, false}, + "ICC_IAR1_EL1": a64SysReg{0x18cc00, true, false}, + "ICC_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true}, + "ICC_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true}, + "ICC_PMR_EL1": a64SysReg{0x184600, true, true}, + "ICC_RPR_EL1": a64SysReg{0x18cb60, true, false}, + "ICC_SGI0R_EL1": a64SysReg{0x18cbe0, false, true}, + "ICC_SGI1R_EL1": a64SysReg{0x18cba0, false, true}, + "ICC_SRE_EL1": a64SysReg{0x18cca0, true, true}, + "ICV_AP0R0_EL1": a64SysReg{0x18c880, true, true}, + "ICV_AP0R1_EL1": a64SysReg{0x18c8a0, true, true}, + "ICV_AP0R2_EL1": a64SysReg{0x18c8c0, true, true}, + "ICV_AP0R3_EL1": a64SysReg{0x18c8e0, true, true}, + "ICV_AP1R0_EL1": a64SysReg{0x18c900, true, true}, + "ICV_AP1R1_EL1": a64SysReg{0x18c920, true, true}, + "ICV_AP1R2_EL1": a64SysReg{0x18c940, true, true}, + "ICV_AP1R3_EL1": a64SysReg{0x18c960, true, true}, + "ICV_BPR0_EL1": a64SysReg{0x18c860, true, true}, + "ICV_BPR1_EL1": a64SysReg{0x18cc60, true, true}, + "ICV_CTLR_EL1": a64SysReg{0x18cc80, true, true}, + "ICV_DIR_EL1": a64SysReg{0x18cb20, false, true}, + "ICV_EOIR0_EL1": a64SysReg{0x18c820, false, true}, + "ICV_EOIR1_EL1": a64SysReg{0x18cc20, false, true}, + "ICV_HPPIR0_EL1": a64SysReg{0x18c840, true, false}, + "ICV_HPPIR1_EL1": a64SysReg{0x18cc40, true, false}, + "ICV_IAR0_EL1": a64SysReg{0x18c800, true, false}, + "ICV_IAR1_EL1": a64SysReg{0x18cc00, true, false}, + "ICV_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true}, + "ICV_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true}, + "ICV_PMR_EL1": a64SysReg{0x184600, true, true}, + "ICV_RPR_EL1": a64SysReg{0x18cb60, true, false}, + "ID_AA64AFR0_EL1": a64SysReg{0x180580, true, false}, + "ID_AA64AFR1_EL1": a64SysReg{0x1805a0, true, false}, + "ID_AA64DFR0_EL1": a64SysReg{0x180500, true, false}, + "ID_AA64DFR1_EL1": a64SysReg{0x180520, true, false}, + "ID_AA64ISAR0_EL1": a64SysReg{0x180600, true, false}, + "ID_AA64ISAR1_EL1": a64SysReg{0x180620, true, false}, + "ID_AA64MMFR0_EL1": a64SysReg{0x180700, true, false}, + "ID_AA64MMFR1_EL1": a64SysReg{0x180720, true, false}, + "ID_AA64MMFR2_EL1": a64SysReg{0x180740, true, false}, + "ID_AA64PFR0_EL1": a64SysReg{0x180400, true, false}, + "ID_AA64PFR1_EL1": a64SysReg{0x180420, true, false}, + "ID_AA64ZFR0_EL1": a64SysReg{0x180480, true, false}, + "ID_AFR0_EL1": a64SysReg{0x180160, true, false}, + "ID_DFR0_EL1": a64SysReg{0x180140, true, false}, + "ID_ISAR0_EL1": a64SysReg{0x180200, true, false}, + "ID_ISAR1_EL1": a64SysReg{0x180220, true, false}, + "ID_ISAR2_EL1": a64SysReg{0x180240, true, false}, + "ID_ISAR3_EL1": a64SysReg{0x180260, true, false}, + "ID_ISAR4_EL1": a64SysReg{0x180280, true, false}, + "ID_ISAR5_EL1": a64SysReg{0x1802a0, true, false}, + "ID_ISAR6_EL1": a64SysReg{0x1802e0, true, false}, + "ID_MMFR0_EL1": a64SysReg{0x180180, true, false}, + "ID_MMFR1_EL1": a64SysReg{0x1801a0, true, false}, + "ID_MMFR2_EL1": a64SysReg{0x1801c0, true, false}, + "ID_MMFR3_EL1": a64SysReg{0x1801e0, true, false}, + "ID_MMFR4_EL1": a64SysReg{0x1802c0, true, false}, + "ID_PFR0_EL1": a64SysReg{0x180100, true, false}, + "ID_PFR1_EL1": a64SysReg{0x180120, true, false}, + "ID_PFR2_EL1": a64SysReg{0x180380, true, false}, + "ISR_EL1": a64SysReg{0x18c100, true, false}, + "LORC_EL1": a64SysReg{0x18a460, true, true}, + "LOREA_EL1": a64SysReg{0x18a420, true, true}, + "LORID_EL1": a64SysReg{0x18a4e0, true, false}, + "LORN_EL1": a64SysReg{0x18a440, true, true}, + "LORSA_EL1": a64SysReg{0x18a400, true, true}, + "MAIR_EL1": a64SysReg{0x18a200, true, true}, + "MDCCINT_EL1": a64SysReg{0x100200, true, true}, + "MDCCSR_EL0": a64SysReg{0x130100, true, false}, + "MDRAR_EL1": a64SysReg{0x101000, true, false}, + "MDSCR_EL1": a64SysReg{0x100240, true, true}, + "MIDR_EL1": a64SysReg{0x180000, true, false}, + "MPAM0_EL1": a64SysReg{0x18a520, true, true}, + "MPAM1_EL1": a64SysReg{0x18a500, true, true}, + "MPAMIDR_EL1": a64SysReg{0x18a480, true, false}, + "MPIDR_EL1": a64SysReg{0x1800a0, true, false}, + "MVFR0_EL1": a64SysReg{0x180300, true, false}, + "MVFR1_EL1": a64SysReg{0x180320, true, false}, + "MVFR2_EL1": a64SysReg{0x180340, true, false}, + "NZCV": a64SysReg{0x1b4200, true, true}, + "OSDLR_EL1": a64SysReg{0x101380, true, true}, + "OSDTRRX_EL1": a64SysReg{0x100040, true, true}, + "OSDTRTX_EL1": a64SysReg{0x100340, true, true}, + "OSECCR_EL1": a64SysReg{0x100640, true, true}, + "OSLAR_EL1": a64SysReg{0x101080, false, true}, + "OSLSR_EL1": a64SysReg{0x101180, true, false}, + "PAN": a64SysReg{0x184260, true, true}, + "PAR_EL1": a64SysReg{0x187400, true, true}, + "PMBIDR_EL1": a64SysReg{0x189ae0, true, false}, + "PMBLIMITR_EL1": a64SysReg{0x189a00, true, true}, + "PMBPTR_EL1": a64SysReg{0x189a20, true, true}, + "PMBSR_EL1": a64SysReg{0x189a60, true, true}, + "PMCCFILTR_EL0": a64SysReg{0x1befe0, true, true}, + "PMCCNTR_EL0": a64SysReg{0x1b9d00, true, true}, + "PMCEID0_EL0": a64SysReg{0x1b9cc0, true, false}, + "PMCEID1_EL0": a64SysReg{0x1b9ce0, true, false}, + "PMCNTENCLR_EL0": a64SysReg{0x1b9c40, true, true}, + "PMCNTENSET_EL0": a64SysReg{0x1b9c20, true, true}, + "PMCR_EL0": a64SysReg{0x1b9c00, true, true}, + "PMEVCNTR0_EL0": a64SysReg{0x1be800, true, true}, + "PMEVCNTR1_EL0": a64SysReg{0x1be820, true, true}, + "PMEVCNTR2_EL0": a64SysReg{0x1be840, true, true}, + "PMEVCNTR3_EL0": a64SysReg{0x1be860, true, true}, + "PMEVCNTR4_EL0": a64SysReg{0x1be880, true, true}, + "PMEVCNTR5_EL0": a64SysReg{0x1be8a0, true, true}, + "PMEVCNTR6_EL0": a64SysReg{0x1be8c0, true, true}, + "PMEVCNTR7_EL0": a64SysReg{0x1be8e0, true, true}, + "PMEVCNTR8_EL0": a64SysReg{0x1be900, true, true}, + "PMEVCNTR9_EL0": a64SysReg{0x1be920, true, true}, + "PMEVCNTR10_EL0": a64SysReg{0x1be940, true, true}, + "PMEVCNTR11_EL0": a64SysReg{0x1be960, true, true}, + "PMEVCNTR12_EL0": a64SysReg{0x1be980, true, true}, + "PMEVCNTR13_EL0": a64SysReg{0x1be9a0, true, true}, + "PMEVCNTR14_EL0": a64SysReg{0x1be9c0, true, true}, + "PMEVCNTR15_EL0": a64SysReg{0x1be9e0, true, true}, + "PMEVCNTR16_EL0": a64SysReg{0x1bea00, true, true}, + "PMEVCNTR17_EL0": a64SysReg{0x1bea20, true, true}, + "PMEVCNTR18_EL0": a64SysReg{0x1bea40, true, true}, + "PMEVCNTR19_EL0": a64SysReg{0x1bea60, true, true}, + "PMEVCNTR20_EL0": a64SysReg{0x1bea80, true, true}, + "PMEVCNTR21_EL0": a64SysReg{0x1beaa0, true, true}, + "PMEVCNTR22_EL0": a64SysReg{0x1beac0, true, true}, + "PMEVCNTR23_EL0": a64SysReg{0x1beae0, true, true}, + "PMEVCNTR24_EL0": a64SysReg{0x1beb00, true, true}, + "PMEVCNTR25_EL0": a64SysReg{0x1beb20, true, true}, + "PMEVCNTR26_EL0": a64SysReg{0x1beb40, true, true}, + "PMEVCNTR27_EL0": a64SysReg{0x1beb60, true, true}, + "PMEVCNTR28_EL0": a64SysReg{0x1beb80, true, true}, + "PMEVCNTR29_EL0": a64SysReg{0x1beba0, true, true}, + "PMEVCNTR30_EL0": a64SysReg{0x1bebc0, true, true}, + "PMEVTYPER0_EL0": a64SysReg{0x1bec00, true, true}, + "PMEVTYPER1_EL0": a64SysReg{0x1bec20, true, true}, + "PMEVTYPER2_EL0": a64SysReg{0x1bec40, true, true}, + "PMEVTYPER3_EL0": a64SysReg{0x1bec60, true, true}, + "PMEVTYPER4_EL0": a64SysReg{0x1bec80, true, true}, + "PMEVTYPER5_EL0": a64SysReg{0x1beca0, true, true}, + "PMEVTYPER6_EL0": a64SysReg{0x1becc0, true, true}, + "PMEVTYPER7_EL0": a64SysReg{0x1bece0, true, true}, + "PMEVTYPER8_EL0": a64SysReg{0x1bed00, true, true}, + "PMEVTYPER9_EL0": a64SysReg{0x1bed20, true, true}, + "PMEVTYPER10_EL0": a64SysReg{0x1bed40, true, true}, + "PMEVTYPER11_EL0": a64SysReg{0x1bed60, true, true}, + "PMEVTYPER12_EL0": a64SysReg{0x1bed80, true, true}, + "PMEVTYPER13_EL0": a64SysReg{0x1beda0, true, true}, + "PMEVTYPER14_EL0": a64SysReg{0x1bedc0, true, true}, + "PMEVTYPER15_EL0": a64SysReg{0x1bede0, true, true}, + "PMEVTYPER16_EL0": a64SysReg{0x1bee00, true, true}, + "PMEVTYPER17_EL0": a64SysReg{0x1bee20, true, true}, + "PMEVTYPER18_EL0": a64SysReg{0x1bee40, true, true}, + "PMEVTYPER19_EL0": a64SysReg{0x1bee60, true, true}, + "PMEVTYPER20_EL0": a64SysReg{0x1bee80, true, true}, + "PMEVTYPER21_EL0": a64SysReg{0x1beea0, true, true}, + "PMEVTYPER22_EL0": a64SysReg{0x1beec0, true, true}, + "PMEVTYPER23_EL0": a64SysReg{0x1beee0, true, true}, + "PMEVTYPER24_EL0": a64SysReg{0x1bef00, true, true}, + "PMEVTYPER25_EL0": a64SysReg{0x1bef20, true, true}, + "PMEVTYPER26_EL0": a64SysReg{0x1bef40, true, true}, + "PMEVTYPER27_EL0": a64SysReg{0x1bef60, true, true}, + "PMEVTYPER28_EL0": a64SysReg{0x1bef80, true, true}, + "PMEVTYPER29_EL0": a64SysReg{0x1befa0, true, true}, + "PMEVTYPER30_EL0": a64SysReg{0x1befc0, true, true}, + "PMINTENCLR_EL1": a64SysReg{0x189e40, true, true}, + "PMINTENSET_EL1": a64SysReg{0x189e20, true, true}, + "PMMIR_EL1": a64SysReg{0x189ec0, true, false}, + "PMOVSCLR_EL0": a64SysReg{0x1b9c60, true, true}, + "PMOVSSET_EL0": a64SysReg{0x1b9e60, true, true}, + "PMSCR_EL1": a64SysReg{0x189900, true, true}, + "PMSELR_EL0": a64SysReg{0x1b9ca0, true, true}, + "PMSEVFR_EL1": a64SysReg{0x1899a0, true, true}, + "PMSFCR_EL1": a64SysReg{0x189980, true, true}, + "PMSICR_EL1": a64SysReg{0x189940, true, true}, + "PMSIDR_EL1": a64SysReg{0x1899e0, true, false}, + "PMSIRR_EL1": a64SysReg{0x189960, true, true}, + "PMSLATFR_EL1": a64SysReg{0x1899c0, true, true}, + "PMSWINC_EL0": a64SysReg{0x1b9c80, false, true}, + "PMUSERENR_EL0": a64SysReg{0x1b9e00, true, true}, + "PMXEVCNTR_EL0": a64SysReg{0x1b9d40, true, true}, + "PMXEVTYPER_EL0": a64SysReg{0x1b9d20, true, true}, + "REVIDR_EL1": a64SysReg{0x1800c0, true, false}, + "RGSR_EL1": a64SysReg{0x1810a0, true, true}, + "RMR_EL1": a64SysReg{0x18c040, true, true}, + "RNDR": a64SysReg{0x1b2400, true, false}, + "RNDRRS": a64SysReg{0x1b2420, true, false}, + "RVBAR_EL1": a64SysReg{0x18c020, true, false}, + "SCTLR_EL1": a64SysReg{0x181000, true, true}, + "SCXTNUM_EL0": a64SysReg{0x1bd0e0, true, true}, + "SCXTNUM_EL1": a64SysReg{0x18d0e0, true, true}, + "SP_EL0": a64SysReg{0x184100, true, true}, + "SP_EL1": a64SysReg{0x1c4100, true, true}, + "SPSel": a64SysReg{0x184200, true, true}, + "SPSR_abt": a64SysReg{0x1c4320, true, true}, + "SPSR_EL1": a64SysReg{0x184000, true, true}, + "SPSR_fiq": a64SysReg{0x1c4360, true, true}, + "SPSR_irq": a64SysReg{0x1c4300, true, true}, + "SPSR_und": a64SysReg{0x1c4340, true, true}, + "SSBS": a64SysReg{0x1b42c0, true, true}, + "TCO": a64SysReg{0x1b42e0, true, true}, + "TCR_EL1": a64SysReg{0x182040, true, true}, + "TFSR_EL1": a64SysReg{0x185600, true, true}, + "TFSRE0_EL1": a64SysReg{0x185620, true, true}, + "TPIDR_EL0": a64SysReg{0x1bd040, true, true}, + "TPIDR_EL1": a64SysReg{0x18d080, true, true}, + "TPIDRRO_EL0": a64SysReg{0x1bd060, true, true}, + "TRFCR_EL1": a64SysReg{0x181220, true, true}, + "TTBR0_EL1": a64SysReg{0x182000, true, true}, + "TTBR1_EL1": a64SysReg{0x182020, true, true}, + "UAO": a64SysReg{0x184280, true, true}, + "VBAR_EL1": a64SysReg{0x18c000, true, true}, + "ZCR_EL1": a64SysReg{0x181200, true, true}, +} + +// a64SysInst is one TLBI alias: the fields the SYS encoding carries beside +// the fixed op0 = 01 and CRn = 8. +type a64SysInst struct { + op1, cm, op2 uint32 +} + +// a64TLBIOps maps the TLBI operation names to their fields; the register +// operand is optional and defaults to ZR. +var a64TLBIOps = map[string]a64SysInst{ + "ALLE1": {0x4, 0x7, 0x4}, + "ALLE1IS": {0x4, 0x3, 0x4}, + "ALLE1OS": {0x4, 0x1, 0x4}, + "ALLE2": {0x4, 0x7, 0x0}, + "ALLE2IS": {0x4, 0x3, 0x0}, + "ALLE2OS": {0x4, 0x1, 0x0}, + "ALLE3": {0x6, 0x7, 0x0}, + "ALLE3IS": {0x6, 0x3, 0x0}, + "ALLE3OS": {0x6, 0x1, 0x0}, + "ASIDE1": {0x0, 0x7, 0x2}, + "ASIDE1IS": {0x0, 0x3, 0x2}, + "ASIDE1OS": {0x0, 0x1, 0x2}, + "IPAS2E1": {0x4, 0x4, 0x1}, + "IPAS2E1IS": {0x4, 0x0, 0x1}, + "IPAS2E1OS": {0x4, 0x4, 0x0}, + "IPAS2LE1": {0x4, 0x4, 0x5}, + "IPAS2LE1IS": {0x4, 0x0, 0x5}, + "IPAS2LE1OS": {0x4, 0x4, 0x4}, + "RIPAS2E1": {0x4, 0x4, 0x2}, + "RIPAS2E1IS": {0x4, 0x0, 0x2}, + "RIPAS2E1OS": {0x4, 0x4, 0x3}, + "RIPAS2LE1": {0x4, 0x4, 0x6}, + "RIPAS2LE1IS": {0x4, 0x0, 0x6}, + "RIPAS2LE1OS": {0x4, 0x4, 0x7}, + "RVAAE1": {0x0, 0x6, 0x3}, + "RVAAE1IS": {0x0, 0x2, 0x3}, + "RVAAE1OS": {0x0, 0x5, 0x3}, + "RVAALE1": {0x0, 0x6, 0x7}, + "RVAALE1IS": {0x0, 0x2, 0x7}, + "RVAALE1OS": {0x0, 0x5, 0x7}, + "RVAE1": {0x0, 0x6, 0x1}, + "RVAE1IS": {0x0, 0x2, 0x1}, + "RVAE1OS": {0x0, 0x5, 0x1}, + "RVAE2": {0x4, 0x6, 0x1}, + "RVAE2IS": {0x4, 0x2, 0x1}, + "RVAE2OS": {0x4, 0x5, 0x1}, + "RVAE3": {0x6, 0x6, 0x1}, + "RVAE3IS": {0x6, 0x2, 0x1}, + "RVAE3OS": {0x6, 0x5, 0x1}, + "RVALE1": {0x0, 0x6, 0x5}, + "RVALE1IS": {0x0, 0x2, 0x5}, + "RVALE1OS": {0x0, 0x5, 0x5}, + "RVALE2": {0x4, 0x6, 0x5}, + "RVALE2IS": {0x4, 0x2, 0x5}, + "RVALE2OS": {0x4, 0x5, 0x5}, + "RVALE3": {0x6, 0x6, 0x5}, + "RVALE3IS": {0x6, 0x2, 0x5}, + "RVALE3OS": {0x6, 0x5, 0x5}, + "VAAE1": {0x0, 0x7, 0x3}, + "VAAE1IS": {0x0, 0x3, 0x3}, + "VAAE1OS": {0x0, 0x1, 0x3}, + "VAALE1": {0x0, 0x7, 0x7}, + "VAALE1IS": {0x0, 0x3, 0x7}, + "VAALE1OS": {0x0, 0x1, 0x7}, + "VAE1": {0x0, 0x7, 0x1}, + "VAE1IS": {0x0, 0x3, 0x1}, + "VAE1OS": {0x0, 0x1, 0x1}, + "VAE2": {0x4, 0x7, 0x1}, + "VAE2IS": {0x4, 0x3, 0x1}, + "VAE2OS": {0x4, 0x1, 0x1}, + "VAE3": {0x6, 0x7, 0x1}, + "VAE3IS": {0x6, 0x3, 0x1}, + "VAE3OS": {0x6, 0x1, 0x1}, + "VALE1": {0x0, 0x7, 0x5}, + "VALE1IS": {0x0, 0x3, 0x5}, + "VALE1OS": {0x0, 0x1, 0x5}, + "VALE2": {0x4, 0x7, 0x5}, + "VALE2IS": {0x4, 0x3, 0x5}, + "VALE2OS": {0x4, 0x1, 0x5}, + "VALE3": {0x6, 0x7, 0x5}, + "VALE3IS": {0x6, 0x3, 0x5}, + "VALE3OS": {0x6, 0x1, 0x5}, + "VMALLE1": {0x0, 0x7, 0x0}, + "VMALLE1IS": {0x0, 0x3, 0x0}, + "VMALLE1OS": {0x0, 0x1, 0x0}, + "VMALLS12E1": {0x4, 0x7, 0x6}, + "VMALLS12E1IS": {0x4, 0x3, 0x6}, + "VMALLS12E1OS": {0x4, 0x1, 0x6}, +} + +// a64DCOps2 maps the DC operation names to their fields; the register +// operand is mandatory. +var a64DCOps2 = map[string]a64SysInst{ + "CGDSW": {0x0, 0xa, 0x6}, + "CGDVAC": {0x3, 0xa, 0x5}, + "CGDVADP": {0x3, 0xd, 0x5}, + "CGDVAP": {0x3, 0xc, 0x5}, + "CGSW": {0x0, 0xa, 0x4}, + "CGVAC": {0x3, 0xa, 0x3}, + "CGVADP": {0x3, 0xd, 0x3}, + "CGVAP": {0x3, 0xc, 0x3}, + "CIGDSW": {0x0, 0xe, 0x6}, + "CIGDVAC": {0x3, 0xe, 0x5}, + "CIGSW": {0x0, 0xe, 0x4}, + "CIGVAC": {0x3, 0xe, 0x3}, + "CISW": {0x0, 0xe, 0x2}, + "CIVAC": {0x3, 0xe, 0x1}, + "CSW": {0x0, 0xa, 0x2}, + "CVAC": {0x3, 0xa, 0x1}, + "CVADP": {0x3, 0xd, 0x1}, + "CVAP": {0x3, 0xc, 0x1}, + "CVAU": {0x3, 0xb, 0x1}, + "GVA": {0x3, 0x4, 0x3}, + "GZVA": {0x3, 0x4, 0x4}, + "IGDSW": {0x0, 0x6, 0x6}, + "IGDVAC": {0x0, 0x6, 0x5}, + "IGSW": {0x0, 0x6, 0x4}, + "IGVAC": {0x0, 0x6, 0x3}, + "ISW": {0x0, 0x6, 0x2}, + "IVAC": {0x0, 0x6, 0x1}, + "ZVA": {0x3, 0x4, 0x1}, +} + +// a64RPRFOps maps the range-prefetch operation names to their 6-bit values. +var a64RPRFOps = map[string]uint32{ + "PLDKEEP": 0, + "PLDSTRM": 4, + "PSTKEEP": 1, + "PSTSTRM": 5, +} diff --git a/asm/arm64_sysregs_test.go b/asm/arm64_sysregs_test.go new file mode 100644 index 0000000..8897759 --- /dev/null +++ b/asm/arm64_sysregs_test.go @@ -0,0 +1,164 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +package asm + +import ( + "encoding/binary" + "fmt" + "os" + "path/filepath" + "slices" + "strings" + "testing" + + "sourcedock.dev/petrbalvin/gasm-sdk/parser" +) + +// TestARM64SysRegsDifferential proves the whole system-register table against +// the toolchain at once: one TEXT whose body reads every register the table +// carries (and writes every writable one), assembled by gasm and by +// go tool asm, must agree byte for byte. A single wrong op0/op1/CRn/CRm/op2 +// packing names its register through the first differing word. +func TestARM64SysRegsDifferential(t *testing.T) { + names := make([]string, 0, len(a64SysRegs)) + for name := range a64SysRegs { + names = append(names, name) + } + slices.Sort(names) + + var body strings.Builder + for i, name := range names { + // R18 is the arm64 platform register and R29-R31 carry dedicated + // meanings; a plain read/write destination keeps to R0-R17. + reg := fmt.Sprintf("R%d", i%18) + if a64SysRegs[name].read { + body.WriteString(fmt.Sprintf("\tMRS %s, %s\n", name, reg)) + } + if a64SysRegs[name].write { + body.WriteString(fmt.Sprintf("\tMSR %s, %s\n", reg, name)) + } + } + src := "#include \"textflag.h\"\n\nTEXT ·sysregs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n" + + dir := t.TempDir() + path := filepath.Join(dir, "sysregs_arm64.s") + if err := os.WriteFile(path, []byte(src), 0o644); err != nil { + t.Fatal(err) + } + assertARM64Differential(t, path, src, "sysregs") +} + +// TestARM64FamiliesDifferential pins the non-sysreg families the arm64 +// campaign added: the LSE compare-and-swap pairs, the VMOVI immediate, the +// SIMD narrow/long shift pairs, the VLD2/VLD3/VLD4 and VST2/VST3/VST4 +// structure accesses with their post-index and replicate forms, LDPSW, the +// pointer-authentication hint and the DC maintenance operation. Every +// spelling is the toolchain's own, taken from its arm64 testdata, and the +// bytes must agree word for word. +func TestARM64FamiliesDifferential(t *testing.T) { + src := `#include "textflag.h" + +TEXT ·families(SB), NOSPLIT, $0 + CASPD (R2, R3), (R2), (R8, R9) + CASPW (R6, R7), (R8), (R4, R5) + VMOVI $82, V0.B16 + VMOVI $146, V22.B16 + VSSHLL $0, V1.B8, V2.H8 + VSSHLL $7, V1.B8, V2.H8 + VSSHLL2 $0, V1.B16, V2.H8 + VSHRN $7, V1.H8, V0.B8 + VSHRN2 $31, V1.D2, V0.S4 + VLD2 (R29), [V23.H8, V24.H8] + VLD2.P 16(R0), [V18.B8, V19.B8] + VLD2.P (R1)(R2), [V15.S2, V16.S2] + VLD3 (R27), [V11.S4, V12.S4, V13.S4] + VLD3.P 48(RSP), [V11.S4, V12.S4, V13.S4] + VLD4 (R15), [V10.H4, V11.H4, V12.H4, V13.H4] + VLD4.P 32(R24), [V31.B8, V0.B8, V1.B8, V2.B8] + VLD1R (R1), [V9.B8] + VLD1R.P (R0), [V0.B16] + VLD1R.P 2(R1), [V2.H4] + VLD2R (R15), [V15.H4, V16.H4] + VLD2R.P 16(R0), [V0.D2, V1.D2] + VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8] + VLD4R.P 16(RSP), [V31.S4, V0.S4, V1.S4, V2.S4] + VST2 [V22.H8, V23.H8], (R23) + VST2.P [V14.H4, V15.H4], 16(R17) + VST2.P [V14.H4, V15.H4], (R3)(R17) + VST3 [V1.D2, V2.D2, V3.D2], (R11) + VST3.P [V18.S4, V19.S4, V20.S4], 48(R25) + VST4 [V22.D2, V23.D2, V24.D2, V25.D2], (R3) + VST4.P [V14.D2, V15.D2, V16.D2, V17.D2], 64(R15) + LDPSW (R0), (R1, R2) + LDPSW 4(R0), (R1, R2) + LDPSW -4(R0), (R1, R2) + PACIASP + DC IVAC, R1 + RET +` + dir := t.TempDir() + path := filepath.Join(dir, "families_arm64.s") + if err := os.WriteFile(path, []byte(src), 0o644); err != nil { + t.Fatal(err) + } + assertARM64Differential(t, path, src, "families") +} + +// assertARM64Differential assembles the same source with gasm and with the +// toolchain for arm64 and requires the named function's code bytes to agree. +// The live oracle is a deliberate-run comparison, so -short skips it (the +// push pipeline's mode); the golden bytes of the individual encoders are +// pinned separately in every mode. +func assertARM64Differential(t *testing.T, path, src, fn string) { + t.Helper() + oracle := oracleFuncCode(t, toolAsmObject(t, path, "arm64")) + + // The oracle keys its functions by the qualified object name + // (pkg.name); match on the local part. + want := map[string][]byte{} + for name, code := range oracle { + if _, after, ok := strings.Cut(name, "."); ok { + want[after] = code + } else { + want[name] = code + } + } + if want[fn] == nil { + t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want)) + } + + f, perrs := parser.Parse(path, src) + if len(perrs) > 0 { + t.Fatalf("parse: %v", perrs[0]) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + got := trimTrailingZeroWords(img.Code) + wantB := trimTrailingZeroWords(want[fn]) + if len(got) != len(wantB) { + t.Fatalf("gasm %d bytes, oracle %d bytes", len(got), len(wantB)) + } + for i := range wantB { + if got[i] != wantB[i] { + t.Fatalf("word %d differs: gasm %08x, oracle %08x", i/4, + binary.LittleEndian.Uint32(got[i:i+4]), binary.LittleEndian.Uint32(wantB[i:i+4])) + } + } +} + +// trimTrailingZeroWords drops whole zero words off the end of a code span: +// an object pads a function to its alignment, and the raw image does not. +// A difference in the middle survives the trim untouched. +func trimTrailingZeroWords(b []byte) []byte { + for len(b) >= 4 { + last := b[len(b)-4:] + if last[0]|last[1]|last[2]|last[3] != 0 { + break + } + b = b[:len(b)-4] + } + return b +}