feat(asm): encode the arm64 system registers and structure loads

This commit is contained in:
2026-10-02 20:39:26 +02:00
parent e02918c17b
commit 2747fce7d3
4 changed files with 1375 additions and 79 deletions
+477 -65
View File
@@ -458,10 +458,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Excl(mnem, enc.op, ops)
}
// LSE atomics (LDADD, CAS, SWP).
// LSE atomics (LDADD, CAS, SWP) and the compare-and-swap pair.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
return encodeARM64LSEAtom(mnem, enc.op, ops)
}
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCASP {
return encodeARM64CASP(mnem, enc.op, ops)
}
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
@@ -555,9 +558,12 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
// only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
// an empty register-form spec. VSQSHL/VUQSHL keep their shift-by-
// immediate route when the first operand is an immediate.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
return encodeARM64SimdV(mnem, spec, ops)
if !arm64SimdShiftImmRoute(mnem, ops) {
return encodeARM64SimdV(mnem, spec, ops)
}
}
// Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV).
@@ -565,6 +571,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64SimdV2(mnem, spec, ops)
}
// Narrow/long/wide SIMD families whose size and Q bits read off one
// designated operand (VXTN, VSXTL, VUADDW, VUMULL, VSHRN, VSSHLL, VFCVTN
// and friends).
if spec, ok := a64SimdNLTable[mnem]; ok {
return encodeARM64SimdNL(mnem, spec, ops)
}
// SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR,
// VEXT).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 {
@@ -587,6 +600,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64ShiftImm(mnem, enc.op, ops)
}
// SIMD move immediate: VMOVI $imm8, Vd.B8/B16.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVMoviImm {
return encodeARM64MoviImm(ops)
}
// VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit {
return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits)
@@ -2568,6 +2586,263 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64CASP encodes the compare-and-swap pair: CASP (Rs, Rs+1), (Rn),
// (Rt, Rt+1). Both pairs must start on an even register and be contiguous;
// the second register of each pair rides no encoding field.
func encodeARM64CASP(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects (Rs, Rs+1), (Rn), (Rt, Rt+1), got %d operands", mnem, len(ops))
}
rs, rs1, ok := arm64PairOf(ops[0])
if !ok {
return nil, fmt.Errorf("%s expects a source register pair (Rs, Rs+1)", mnem)
}
rn, err := arm64ExclMem(mnem, ops[1])
if err != nil {
return nil, err
}
rt, rt1, ok := arm64PairOf(ops[2])
if !ok {
return nil, fmt.Errorf("%s expects a destination register pair (Rt, Rt+1)", mnem)
}
if rs&1 != 0 {
return nil, fmt.Errorf("%s: source register pair must start from an even register", mnem)
}
if rt&1 != 0 {
return nil, fmt.Errorf("%s: destination register pair must start from an even register", mnem)
}
if rs != rs1-1 {
return nil, fmt.Errorf("%s: source register pair must be contiguous", mnem)
}
if rt != rt1-1 {
return nil, fmt.Errorf("%s: destination register pair must be contiguous", mnem)
}
if rt == 31 {
return nil, fmt.Errorf("%s: illegal destination register", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64MoviImm encodes VMOVI $imm8, Vd.B8/B16: the modified-immediate
// form of the SIMD move (asm7.go case 86). Only the byte arrangements exist
// and the immediate is one unsigned byte.
func encodeARM64MoviImm(ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("VMOVI expects $immediate, Vd.<T>")
}
vd, ok := arm64VecOf(ops[1])
if !ok || vd.hasIdx || (vd.arr != "B8" && vd.arr != "B16") {
return nil, fmt.Errorf("VMOVI: destination arrangement must be B8 or B16")
}
imm := arm64Imm64(ops[0])
if imm < 0 || imm > 255 {
return nil, fmt.Errorf("VMOVI: immediate constant %d out of range (0..255)", imm)
}
q := uint32(0)
if vd.arr == "B16" {
q = 1 << 30
}
w := 0x0f00e400 | q | uint32(imm>>5&7)<<16 | uint32(imm&0x1f)<<5 | uint32(vd.reg)
return a64wordLE(w), nil
}
// arm64SimdShiftImmRoute reports whether a mnemonic carries both a shift-by-
// immediate and a register form and the operands spell the immediate one: the
// dedicated shift route keeps them.
func arm64SimdShiftImmRoute(mnem string, ops []*ast.Operand) bool {
if mnem != "VSQSHL" && mnem != "VUQSHL" {
return false
}
return len(ops) > 0 && isImmOperand(ops[0])
}
// arm64SimdNLArr describes one arrangement for the narrow/long/wide families:
// the element width in bytes and whether the spelling names the 128-bit form.
func arm64SimdNLArr(arr string) (esize int, wide bool, ok bool) {
switch arr {
case "B8", "B16":
return 1, arr == "B16", true
case "H4", "H8":
return 2, arr == "H8", true
case "S2", "S4":
return 4, arr == "S4", true
case "D1", "D2":
return 8, arr == "D2", true
}
return 0, false, false
}
// arm64SimdLongPair validates a long pairing (source narrow, destination
// wide): the destination element is twice the source's, the destination is
// always spelled the wide way (H8/S4/D2) and the source carries the 128-bit
// flag exactly for the .2 spellings.
func arm64SimdLongPair(mnem, src, dst string, two bool) error {
se, sw, ok1 := arm64SimdNLArr(src)
de, dw, ok2 := arm64SimdNLArr(dst)
if !ok1 || !ok2 || de != 2*se {
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
}
if !dw || sw != two {
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
}
return nil
}
// arm64SimdNarrowPair validates a narrow pairing (source wide, destination
// narrow): the mirror image of arm64SimdLongPair.
func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
se, sw, ok1 := arm64SimdNLArr(src)
de, dw, ok2 := arm64SimdNLArr(dst)
if !ok1 || !ok2 || se != 2*de {
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
}
if !sw || dw != two {
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
}
return nil
}
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
// instruction contributes: the driving arrangement's size and Q bits, or for
// the FCVT family only the Q bit, whose size field is fixed in the base.
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
if spec.qonly {
if two {
return 1 << 30
}
return 0
}
return a64ArrBits[a64ArrIndex(drive)]
}
// encodeARM64SimdNL encodes the narrow/long/wide SIMD families
// (a64SimdNLTable): XTN and FCVTN narrow a wide source, SXTL and FCVTL
// lengthen, the MULL/MLAL/MLSL group multiplies long, UADDW widens, and the
// SSHLL/USHLL and SHRN shifts carry their immediate in the immh:immb field.
// The size and Q bits read off the designated driving operand, and the .2
// spellings force the 128-bit side through their own arrangement.
func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]byte, error) {
two := strings.HasSuffix(mnem, "2")
// vecAt parses operand i as a vector register with an arrangement.
vecAt := func(i int) (a64Vec, bool) {
if i >= len(ops) {
return a64Vec{}, false
}
v, ok := arm64VecOf(ops[i])
if !ok || v.hasIdx {
return a64Vec{}, false
}
return v, true
}
switch spec.form {
case a64NLTwoNarrow, a64NLTwoLong:
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vn, ok1 := vecAt(0)
vd, ok2 := vecAt(1)
if !ok1 || !ok2 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var drive string
var pairErr error
if spec.form == a64NLTwoNarrow {
// XTN/FCVTN: wide source into a narrow destination; the
// arrangement bits follow the destination.
drive, pairErr = vd.arr, arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two)
} else {
// SXTL/UXTL/FCVTL: narrow source into a wide destination; the
// arrangement bits follow the source.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
}
if pairErr != nil {
return nil, pairErr
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongMul, a64NLThreeWide:
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
vm, ok1 := vecAt(0)
vn, ok2 := vecAt(1)
vd, ok3 := vecAt(2)
if !ok1 || !ok2 || !ok3 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var drive string
var pairErr error
if spec.form == a64NLThreeWide {
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
// the arrangement bits follow the wide side.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
} else {
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
// wide one; the arrangement bits follow the narrow source.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
}
if pairErr != nil {
return nil, pairErr
}
if spec.form == a64NLThreeWide && vd.arr != vn.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vn.arr, vd.arr)
}
if spec.form == a64NLThreeLongMul && vm.arr != vn.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongShift, a64NLThreeNarrowShift:
if len(ops) != 3 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects ($shift, Vn.<T>, Vd.<T>)", mnem)
}
sh := arm64Imm64(ops[0])
vn, ok1 := vecAt(1)
vd, ok2 := vecAt(2)
if !ok1 || !ok2 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
if spec.form == a64NLThreeLongShift {
// SSHLL/USHLL: the narrow source drives the immediate's size
// (immh:immb = esize + shift), so the arrangement bits carry
// the Q bit alone: the size field belongs to immh, and ORing
// the source's size bits into it would collide with immb.
if err := arm64SimdLongPair(mnem, vn.arr, vd.arr, two); err != nil {
return nil, err
}
se, _, _ := arm64SimdNLArr(vn.arr)
esize := se * 8
if sh < 0 || sh >= int64(esize) {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
}
var qBit uint32
if two {
qBit = 1 << 30
}
return a64wordLE(spec.base | uint32(esize+int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// SHRN: the narrow destination drives the immediate's size
// (immh:immb = esize - shift over the wide source element), so the
// arrangement bits carry the Q bit alone, exactly as above.
if err := arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two); err != nil {
return nil, err
}
se, _, _ := arm64SimdNLArr(vn.arr)
esize := se * 8
if sh < 1 || sh >= int64(esize) {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize-1)
}
var qBit uint32
if two {
qBit = 1 << 30
}
return a64wordLE(spec.base | uint32(esize-int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
// encodeARM64DP1 encodes a data-processing (1 source) instruction:
// RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd.
func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
@@ -2584,7 +2859,8 @@ func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, err
// encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the
// target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31
// selecting the page form.
// selecting the page form. An n(PC) operand resolves to the instruction's
// own address: the toolchain rewrites it away and encodes displacement 0.
func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
@@ -2593,14 +2869,17 @@ func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offset
if rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
target := resolve(arm64Label(ops[0]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel := int64(targetOff - pc)
if rel < -(1<<20) || rel >= 1<<20 {
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
var rel int64
if _, pcRel := arm64PCRelOffset(ops[0]); !pcRel {
target := resolve(arm64Label(ops[0]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel = int64(targetOff - pc)
if rel < -(1<<20) || rel >= 1<<20 {
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
}
}
return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil
}
@@ -2879,11 +3158,16 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// BRK [$imm16] SVC $imm16
// DMB|DSB|ISB $imm4 DC <op>, Rn
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op>
// PRFM (Rn), $imm|<op> RPRFM (Rn), Rm, <op|$imm6>
// SYS $imm[, Rn] SYSL $imm, Rd
// TLBI <op>[, Rn] SB, PACIASP, PACIBSP
//
// The system registers, TLBI and DC aliases and the range-prefetch operations
// come from the toolchain's own data tables in arm64_sysregs.go.
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "PACIASP": 0xd503233f, "PACIBSP": 0xd503237f,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
@@ -2919,6 +3203,12 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil
case "SB":
// Speculation barrier: DSB with a fixed barrier domain.
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(0xd50330ff), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
@@ -2957,7 +3247,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn")
}
base, ok := a64DCOps[operandRegName(ops[0])]
inst, ok := a64DCOps2[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0]))
}
@@ -2965,51 +3255,115 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 {
return nil, fmt.Errorf("DC: invalid register operand")
}
return a64wordLE(base | uint32(rn)&31), nil
w := 0xd5080000 | inst.op1<<16 | 7<<12 | inst.cm<<8 | inst.op2<<5
return a64wordLE(w | uint32(rn)&31), nil
case "TLBI":
// The register operand is optional: TLBI VMALLE1IS alone means ZR.
if len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("TLBI expects <op>[, Rn]")
}
inst, ok := a64TLBIOps[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("TLBI: unknown operation %q", operandRegName(ops[0]))
}
rt := 31
if len(ops) == 2 {
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
return nil, fmt.Errorf("TLBI: invalid register operand")
}
}
w := 0xd5080000 | inst.op1<<16 | 8<<12 | inst.cm<<8 | inst.op2<<5
return a64wordLE(w | uint32(rt)&31), nil
case "SYS", "SYSL":
// SYS $imm[, Rn] / SYSL $imm, Rd: the immediate packs
// op1<<16 | CRn<<12 | CRm<<8 | op2<<5, the register defaults to ZR.
if len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
}
if len(ops) == 1 && mnem == "SYSL" {
return nil, fmt.Errorf("SYSL expects $immediate, Rd")
}
if !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
}
imm := arm64Imm64(ops[0])
if imm < 0 || imm&^0x7FFE0 != 0 {
return nil, fmt.Errorf("%s: illegal SYS argument %d", mnem, imm)
}
rt := 31
if len(ops) == 2 {
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
}
base := uint32(0xd5080000)
if mnem == "SYSL" {
base = 0xd5280000
}
return a64wordLE(base | uint32(imm) | uint32(rt)&31), nil
case "MRS":
if len(ops) != 2 {
return nil, fmt.Errorf("MRS expects <sysreg>, Rd")
}
base, ok := a64MRSOps[operandRegName(ops[0])]
reg, ok := a64SysRegs[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0]))
}
if !reg.read {
return nil, fmt.Errorf("MRS: system register is not readable: %q", operandRegName(ops[0]))
}
rd := arm64RegNum(operandRegName(ops[1]))
if rd < 0 {
return nil, fmt.Errorf("MRS: invalid register operand")
}
return a64wordLE(base | uint32(rd)&31), nil
return a64wordLE(0xd5300000 | reg.v | uint32(rd)&31), nil
case "MSR":
if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if isImmOperand(ops[0]) {
v := arm64Imm64(ops[0])
// The PSTATE fields keep their dedicated immediate form.
if base, ok := a64MSROps[operandRegName(ops[1])]; ok {
if v < 0 || v > 15 {
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
}
return a64wordLE(base | uint32(v)<<8 | 31), nil
}
// A $0 against a full system register writes it from ZR, exactly
// the way the toolchain preprocesses the constant away; any other
// immediate is the PSTATE-form error.
if v != 0 {
return nil, fmt.Errorf("MSR: illegal PSTATE field for immediate move: %q", operandRegName(ops[1]))
}
reg, ok := a64SysRegs[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
if !reg.write {
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
}
return a64wordLE(base | uint32(rs)&31), nil
return a64wordLE(0xd5100000 | reg.v | 31), nil
}
base, ok := a64MSROps[operandRegName(ops[1])]
// Register form: MSR Rn, <sysreg>.
reg, ok := a64SysRegs[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
v := arm64Imm64(ops[0])
if v < 0 || v > 15 {
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
if !reg.write {
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
}
return a64wordLE(base | uint32(v)<<8 | 31), nil
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(0xd5100000 | reg.v | uint32(rs)&31), nil
case "PRFM":
if len(ops) != 2 {
return nil, fmt.Errorf("PRFM expects (Rn), $immediate|<op>")
}
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
if rn < 0 || off != 0 {
if rn < 0 || off < 0 || off%8 != 0 || off/8 >= 4096 {
return nil, fmt.Errorf("PRFM: invalid memory operand")
}
var prfop int64
@@ -3025,7 +3379,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
prfop = int64(p)
}
return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil
return a64wordLE(0xf9800000 | uint32(off/8)<<10 | uint32(rn)<<5 | uint32(prfop)), nil
case "RPRFM":
// RPRFM (Rn), Rm, <op|$imm6>: the 6-bit operation scatters across
// bits 15, 13, 12 and 2:0 (asm7.go case 110).
if len(ops) != 3 {
return nil, fmt.Errorf("RPRFM expects (Rn), Rm, <op|$immediate>")
}
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
if rn < 0 || off != 0 {
return nil, fmt.Errorf("RPRFM: invalid memory operand")
}
rm := arm64RegNum(operandRegName(ops[1]))
if rm < 0 {
return nil, fmt.Errorf("RPRFM: invalid register operand")
}
var op uint64
if isImmOperand(ops[2]) {
op = uint64(arm64Imm64(ops[2]))
if op > 63 {
return nil, fmt.Errorf("RPRFM: range prefetch immediate %d out of range (0..63)", op)
}
} else {
v, ok := a64RPRFOps[operandRegName(ops[2])]
if !ok {
return nil, fmt.Errorf("RPRFM: unknown range prefetch operation %q", operandRegName(ops[2]))
}
op = uint64(v)
}
scatter := (op&(1<<5))<<10 | (op&(1<<4))<<9 | (op&(1<<3))<<9 | op&7
return a64wordLE(0xf8a04818 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(scatter)), nil
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
@@ -3448,7 +3831,7 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("invalid destination register in VTBL")
}
for i, t := range ts {
if t.hasIdx || t.reg != ts[0].reg+i {
if t.hasIdx || (ts[0].reg+i)&31 != t.reg {
return nil, fmt.Errorf("VTBL table registers must be consecutive")
}
}
@@ -3605,15 +3988,17 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
// encodeARM64VLDST encodes the SIMD structure loads and stores:
//
// VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn)
// VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn)
// VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane)
// VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3]
// VLD1|2|3|4 (Rn), [Vt.arr, ...] VST1|2|3|4 [Vt.arr, ...], (Rn)
// VLD1|2|3|4.P off(Rn), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], off(Rn)
// VLD1|2|3|4.P (Rn)(Rm), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], (Rn)(Rm)
// VLD1|2|3|4R (Rn), [Vt.arr, ...] (replicating loads)
// VLD1 off(Rn), Vt.T[i] VST1 Vt.T[i], off(Rn) (one lane)
//
// The post-index forms set the post bit and Rm = 11111. A spelled offset
// rides along (the encoding ignores it; the toolchain only checks that it
// matches the access size), and a multi-register post-index list takes no
// offset at all, the increment following from the list.
// The post-index forms set the post bit and carry Rm: 11111 for an immediate
// increment, the spelled register for (Rn)(Rm). A register list may wrap
// around V31: the toolchain checks only (first+i) mod 32. A spelled offset
// rides along on the one-lane forms (the toolchain only checks that it
// matches the access size).
func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) {
load := strings.HasPrefix(mnem, "VLD")
@@ -3651,24 +4036,31 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
if off != 0 && post == 0 {
return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off)
}
// VLD1R loads one register and replicates; VLD4R loads four.
if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") {
want := 1
base := uint32(0x0d40c000)
if strings.HasPrefix(mnem, "VLD4R") {
want, base = 4, 0x0d60e000
// The post-index increment: 11111 for an immediate offset, else the
// spelled (Rn)(Rm) register.
rm := 31
if post != 0 {
if idx := ops[memIdx].Addr.Index; idx != "" {
if rm = arm64RegNum(idx); rm < 0 {
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
}
}
if len(vs) != want {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want)
}
// The replicating loads: VLD1R through VLD4R load one register and
// replicate it across the whole list.
if base := strings.TrimSuffix(mnem, ".P"); load && strings.HasSuffix(base, "R") && len(base) == 5 {
n := int(base[3] - '0')
if len(vs) != n {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
}
size, q, ok := a64ArrSizeQ(vs[0].arr)
if !ok {
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
}
w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
w := a64VLDNReplicate[n] | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
if post != 0 {
w |= 1<<23 | 0x1f<<16
w |= 1<<23 | uint32(rm)<<16
}
return a64wordLE(w), nil
}
@@ -3677,7 +4069,7 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
return nil, fmt.Errorf("%s expects a list of one to four registers", mnem)
}
for i, v := range vs {
if v.hasIdx || v.reg != vs[0].reg+i {
if v.hasIdx || (vs[0].reg+i)&31 != v.reg {
return nil, fmt.Errorf("%s: register list must be consecutive", mnem)
}
_, _, okArr := a64ArrSizeQ(v.arr)
@@ -3689,21 +4081,36 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
if !ok {
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
}
base := a64VLD1Base[len(vs)]
n := len(vs)
base := a64VLD1Base[n]
if !load {
base = a64VST1Base[len(vs)]
base = a64VST1Base[n]
}
// VLD2/VLD3/VLD4 and VST2/VST3/VST4 name the register count in the
// mnemonic and carry their own opcode fields. The count digit sits at
// index 3 of the mnemonic (VLD2, VST3.P, ...), before any .P suffix.
if stem := strings.TrimSuffix(mnem, ".P"); len(stem) >= 4 && stem[3] >= '2' && stem[3] <= '4' {
n := int(stem[3] - '0')
if n != len(vs) {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
}
if load {
base = a64VLDNBase[n]
} else {
base = a64VSTNBase[n]
}
}
postBits := uint32(0)
if post != 0 {
postBits = 0x9f0000
postBits = 1<<23 | uint32(rm)<<16
}
return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil
}
// encodeARM64VLDSTLane encodes the one-lane structure forms:
// VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and
// VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the
// toolchain's table and is rejected.
// VLD1 off(Rn), Vt.T[i] and VST1 Vt.T[i], off(Rn); the post-index spellings
// add the post bit and Rm: 11111 for an immediate increment, the spelled
// register for (Rn)(Rm).
func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem)
@@ -3713,14 +4120,19 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
if rn < 0 {
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
}
if !load && post == 0 {
return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem)
rm := 31
if post != 0 {
if idx := ops[memIdx].Addr.Index; idx != "" {
if rm = arm64RegNum(idx); rm < 0 {
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
}
}
}
w := uint32(0x0d400000)
switch strings.ToUpper(v.arr) {
case "B":
// Index at bits 12:10 (the size field doubles as the low index bits).
w |= uint32(v.idx) << 10
// Index<3> rides bit 30, index<2:0> the size field at bits 12:10.
w |= uint32(v.idx&7)<<10 | uint32(v.idx>>3&1)<<30
case "H":
// Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11.
w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30
@@ -3734,12 +4146,12 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr)
}
// The base carries bit 22 (L) set; a store clears it. The post-index
// forms add bit 23 and Rm = 11111.
// forms add bit 23 and Rm.
if !load {
w &^= 1 << 22
}
if post != 0 {
w |= 1<<23 | 0x1f<<16
w |= 1<<23 | uint32(rm)<<16
}
return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil
}
+146 -14
View File
@@ -374,6 +374,7 @@ const (
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCASP // compare and swap pair: CASP
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
@@ -384,6 +385,7 @@ const (
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FVMoviImm // SIMD move immediate: VMOVI $imm8, Vd.B8/B16
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
@@ -770,7 +772,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM", "RPRFM", "SYS", "SYSL", "TLBI", "SB", "PACIASP", "PACIBSP"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -783,12 +785,20 @@ func init() {
a64InstrTable["TBNZ"] = a64Enc{format: a64FTestBranch, op: 0x37000000}
// ---- load/store pair (signed offset) ----
// The scale column of a64LoadTable does not reach the pair forms, so each
// entry states its own access width through the imm7 divisor the pair
// encoder derives from the opc field (8 for D, 4 for W and SW, 16 for Q).
a64InstrTable["LDP"] = a64Enc{format: a64FPair, op: 0xa9400000}
a64InstrTable["LDPW"] = a64Enc{format: a64FPair, op: 0x29400000}
a64InstrTable["LDPSW"] = a64Enc{format: a64FPair, op: 0x69400000}
a64InstrTable["STP"] = a64Enc{format: a64FPair, op: 0xa9000000}
a64InstrTable["STPW"] = a64Enc{format: a64FPair, op: 0x29000000}
a64InstrTable["FLDPD"] = a64Enc{format: a64FPair, op: 0x6d400000}
a64InstrTable["FSTPD"] = a64Enc{format: a64FPair, op: 0x6d000000}
a64InstrTable["FLDPS"] = a64Enc{format: a64FPair, op: 0x2d400000}
a64InstrTable["FSTPS"] = a64Enc{format: a64FPair, op: 0x2d000000}
a64InstrTable["FLDPQ"] = a64Enc{format: a64FPair, op: 0xad400000}
a64InstrTable["FSTPQ"] = a64Enc{format: a64FPair, op: 0xad000000}
// ---- acquire/release loads and stores ----
a64InstrTable["LDAR"] = a64Enc{format: a64FAcqRel, op: 0xc8dffc00}
@@ -806,11 +816,23 @@ func init() {
lse := map[string]uint32{
"CASALD": 0xc8e0fc00,
"CASALW": 0x88e0fc00,
"CASB": 0x08a07c00,
"CASAB": 0x08e07c00,
"CASH": 0x48a07c00,
"CASLD": 0xc8a0fc00,
"CASLH": 0x48a0fc00,
"CASAW": 0x88e07c00,
"CASAD": 0xc8e07c00,
"CASALH": 0x48e07c00,
"LDADDALD": 0xf8e00000,
"LDADDALW": 0xb8e00000,
"LDADDAD": 0xf8a00000,
"LDADDAW": 0xb8a00000,
"LDCLRALB": 0x38e01000,
"LDCLRALW": 0xb8e01000,
"LDCLRALD": 0xf8e01000,
"LDCLRAD": 0xf8a01000,
"LDCLRAW": 0xb8a01000,
"LDORALB": 0x38e03000,
"LDORALW": 0xb8e03000,
"LDORALD": 0xf8e03000,
@@ -889,6 +911,12 @@ func init() {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// Compare and swap pair: the second register of each pair is implicit
// (Rs+1 and Rt+1), so the encoding carries Rs and Rt alone over a preset
// fixed field (asm7.go atomicCASP).
a64InstrTable["CASPD"] = a64Enc{format: a64FCASP, op: 1<<30 | 0x41<<21 | 0x1f<<10}
a64InstrTable["CASPW"] = a64Enc{format: a64FCASP, op: 0x41<<21 | 0x1f<<10}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
// register preset to ZR (bits 14:10 = 11111).
@@ -940,11 +968,13 @@ func init() {
a64InstrTable["VMOVS"] = a64Enc{format: a64FMoviLit, op: 0xbd400000}
a64InstrTable["VMOVD"] = a64Enc{format: a64FMoviLit, op: 0xfd400000}
a64InstrTable["VMOVQ"] = a64Enc{format: a64FMoviLit, op: 0x3dc00000}
a64InstrTable["VMOVI"] = a64Enc{format: a64FVMoviImm}
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 7<<10}
a64InstrTable["VUSRA"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 5<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
@@ -957,6 +987,24 @@ func init() {
a64InstrTable["VLD1R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4R.P"] = a64Enc{format: a64FVLDST, op: 1}
// Multi-register structure accesses beyond VLD1/VST1: VLD2/VLD3/VLD4 and
// the replicate loads VLD2R/VLD3R, each with the post-index spelling.
a64InstrTable["VLD2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD4.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD2R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD2R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VLD3R"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD3R.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST2"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST2.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST3"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST3.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST4"] = a64Enc{format: a64FVLDST}
a64InstrTable["VST4.P"] = a64Enc{format: a64FVLDST, op: 1}
}
// a64SimdVSpec is one arrangement-aware SIMD instruction: the 8B base word,
@@ -1120,6 +1168,78 @@ var a64SimdVTable = map[string]a64SimdVSpec{
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
// Saturating shifts, register forms (the immediate spellings route to
// a64FShiftImm).
"VSQSHL": {0x0e204c00, 0x7f, false},
"VUQSHL": {0x2e204c00, 0x7f, false},
}
// a64SimdNLForm classifies the narrow/long/wide SIMD families whose
// arrangement does not travel on every operand: the encoding's size and Q
// bits read off one designated operand and the element widths pair up across
// the operands.
type a64SimdNLForm uint8
const (
a64NLTwoNarrow a64SimdNLForm = iota // (Vn.wide, Vd.narrow): size/Q from Vd
a64NLTwoLong // (Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeLongMul // (Vm.narrow, Vn.narrow, Vd.long): size/Q from Vn
a64NLThreeWide // (Vm.narrow, Vn.wide, Vd.wide): size/Q from Vn
a64NLThreeLongShift // ($sh, Vn.narrow, Vd.long): size/Q from Vn, immh = esize+sh
a64NLThreeNarrowShift // ($sh, Vn.wide, Vd.narrow): size/Q from Vd, immh = esize-sh
)
// a64SimdNLSpec is one narrow/long/wide instruction: the base word (U, opcode
// and fixed bits positioned) and the arrangement form. qonly marks the FCVT
// family, whose size field is fixed in the base and only the Q bit follows
// the driving arrangement.
type a64SimdNLSpec struct {
base uint32
form a64SimdNLForm
qonly bool
}
// a64SimdNLTable holds the families the arrangement-driven three-register and
// two-register encoders cannot express. The .2 spellings force the 128-bit
// side of the pair through their operand arrangements, so the base carries no
// arrangement bits of its own.
var a64SimdNLTable = map[string]a64SimdNLSpec{
"VSHRN": {0x0f008400, a64NLThreeNarrowShift, false},
"VSHRN2": {0x0f008400, a64NLThreeNarrowShift, false},
"VSXTL": {0x0f00a400, a64NLTwoLong, false},
"VSXTL2": {0x0f00a400, a64NLTwoLong, false},
"VUXTL": {0x2f00a400, a64NLTwoLong, false},
"VUXTL2": {0x2f00a400, a64NLTwoLong, false},
"VXTN": {0x0e202800, a64NLTwoNarrow, false},
"VXTN2": {0x0e202800, a64NLTwoNarrow, false},
"VSQXTN": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTN2": {0x0e204800, a64NLTwoNarrow, false},
"VSQXTUN": {0x2e202800, a64NLTwoNarrow, false},
"VSQXTUN2": {0x2e202800, a64NLTwoNarrow, false},
"VUQXTN": {0x2e204800, a64NLTwoNarrow, false},
"VUQXTN2": {0x2e204800, a64NLTwoNarrow, false},
"VFCVTN": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTN2": {0x0e206800, a64NLTwoNarrow, true},
"VFCVTL": {0x0e217800, a64NLTwoLong, true},
"VFCVTL2": {0x0e217800, a64NLTwoLong, true},
"VSSHLL": {0x0f00a400, a64NLThreeLongShift, false},
"VSSHLL2": {0x0f00a400, a64NLThreeLongShift, false},
"VUSHLL": {0x2f00a400, a64NLThreeLongShift, false},
"VUSHLL2": {0x2f00a400, a64NLThreeLongShift, false},
"VUADDW": {0x2e201000, a64NLThreeWide, false},
"VUADDW2": {0x2e201000, a64NLThreeWide, false},
"VUMULL": {0x2e20c000, a64NLThreeLongMul, false},
"VUMULL2": {0x2e20c000, a64NLThreeLongMul, false},
"VSMULL": {0x0e20c000, a64NLThreeLongMul, false},
"VSMULL2": {0x0e20c000, a64NLThreeLongMul, false},
"VUMLAL": {0x2e208000, a64NLThreeLongMul, false},
"VUMLAL2": {0x2e208000, a64NLThreeLongMul, false},
"VSMLAL": {0x0e208000, a64NLThreeLongMul, false},
"VSMLAL2": {0x0e208000, a64NLThreeLongMul, false},
"VUMLSL": {0x2e20a000, a64NLThreeLongMul, false},
"VUMLSL2": {0x2e20a000, a64NLThreeLongMul, false},
"VSMLSL": {0x0e20a000, a64NLThreeLongMul, false},
"VSMLSL2": {0x0e20a000, a64NLThreeLongMul, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
@@ -1243,6 +1363,16 @@ var a64PRFOps = map[string]int{
var a64VLD1Base = [5]uint32{0, 0x0c407000, 0x0c40a000, 0x0c406000, 0x0c402000}
var a64VST1Base = [5]uint32{0, 0x0c007000, 0x0c00a000, 0x0c006000, 0x0c002000}
// a64VLDNBase and a64VSTNBase hold the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// fixed words (indexed by register count 2..4): the opcode field at bits
// 15:12 carries the access kind.
var a64VLDNBase = [5]uint32{0, 0, 0x0c408000, 0x0c404000, 0x0c400000}
var a64VSTNBase = [5]uint32{0, 0, 0x0c008000, 0x0c004000, 0x0c000000}
// a64VLDNReplicate holds the VLD2R/VLD3R fixed words beside the existing
// VLD1R (0x0d40c000) and VLD4R (0x0d60e000) bases.
var a64VLDNReplicate = [5]uint32{0, 0x0d40c000, 0x0d60c000, 0x0d40e000, 0x0d60e000}
// a64Vec is a parsed vector operand: the register number, the arrangement
// ("" when the operand spells none) and, for element forms, the lane index.
type a64Vec struct {
@@ -1363,23 +1493,25 @@ func a64VecListOf(ops []*ast.Operand, start int) (vs []a64Vec, end int, ok bool)
// a64LSType describes the load/store parameters for a MOV width mnemonic.
type a64LSType struct {
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
size int // 0=byte, 1=half, 2=word, 3=dword
V int // 0=integer, 1=FP
opc int // 00=store/unsigned load, 01=store FP, 10=signed load, 11=load FP
scale int // access width in bytes; the unsigned offset divides by it
}
// a64LoadTable maps MOV width mnemonics to their load/store encoding parameters.
// For loads, opc selects signed vs unsigned; for stores, we flip the opc.
var a64LoadTable = map[string]a64LSType{
"MOVD": {3, 0, 1}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1}, // LDR D (64-bit FP)
"MOVD": {3, 0, 1, 8}, // LDR X (64-bit, unsigned offset)
"MOVWU": {2, 0, 1, 4}, // LDR W (32-bit unsigned)
"MOVW": {2, 0, 2, 4}, // LDRSW (32-bit signed → 64-bit)
"MOVHU": {1, 0, 1, 2}, // LDRH (16-bit unsigned)
"MOVH": {1, 0, 2, 2}, // LDRSH (16-bit signed)
"MOVBU": {0, 0, 1, 1}, // LDRB (8-bit unsigned)
"MOVB": {0, 0, 2, 1}, // LDRSB (8-bit signed)
"FMOVS": {2, 1, 1, 4}, // LDR S (32-bit FP)
"FMOVD": {3, 1, 1, 8}, // LDR D (64-bit FP)
"FMOVQ": {0, 1, 3, 16}, // LDR/STR Q (128-bit FP): opc=11 selects it
}
// a64StoreOpc returns the store opc for a given load type: integer and FP
+588
View File
@@ -0,0 +1,588 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
// arm64 system registers and system-instruction aliases.
//
// The tables are transcribed from the data the Go toolchain itself carries
// (cmd/internal/obj/arm64/sysRegEnc.go and the sysInstFields map of asm7.go),
// which the ARM ARM defines: every system register is the packed field set
// op0<<19 | op1<<16 | CRn<<12 | CRm<<8 | op2<<5, and the read/write flags are
// the toolchain's own access classification. The encoding tables live here so
// the encoder stays testable against the GOROOT testdata word for word.
// a64SysReg is one system register: the packed encoding fields and the
// directions the register supports.
type a64SysReg struct {
v uint32
read bool
write bool
}
// a64SysRegs maps the system register names the toolchain knows to their
// encodings. MRS reads 0xd5300000 | v | Rd and MSR writes
// 0xd5100000 | v | Rt.
var a64SysRegs = map[string]a64SysReg{
"ACTLR_EL1": a64SysReg{0x181020, true, true},
"AFSR0_EL1": a64SysReg{0x185100, true, true},
"AFSR1_EL1": a64SysReg{0x185120, true, true},
"AIDR_EL1": a64SysReg{0x1900e0, true, false},
"AMAIR_EL1": a64SysReg{0x18a300, true, true},
"AMCFGR_EL0": a64SysReg{0x1bd220, true, false},
"AMCGCR_EL0": a64SysReg{0x1bd240, true, false},
"AMCNTENCLR0_EL0": a64SysReg{0x1bd280, true, true},
"AMCNTENCLR1_EL0": a64SysReg{0x1bd300, true, true},
"AMCNTENSET0_EL0": a64SysReg{0x1bd2a0, true, true},
"AMCNTENSET1_EL0": a64SysReg{0x1bd320, true, true},
"AMCR_EL0": a64SysReg{0x1bd200, true, true},
"AMEVCNTR00_EL0": a64SysReg{0x1bd400, true, true},
"AMEVCNTR01_EL0": a64SysReg{0x1bd420, true, true},
"AMEVCNTR02_EL0": a64SysReg{0x1bd440, true, true},
"AMEVCNTR03_EL0": a64SysReg{0x1bd460, true, true},
"AMEVCNTR04_EL0": a64SysReg{0x1bd480, true, true},
"AMEVCNTR05_EL0": a64SysReg{0x1bd4a0, true, true},
"AMEVCNTR06_EL0": a64SysReg{0x1bd4c0, true, true},
"AMEVCNTR07_EL0": a64SysReg{0x1bd4e0, true, true},
"AMEVCNTR08_EL0": a64SysReg{0x1bd500, true, true},
"AMEVCNTR09_EL0": a64SysReg{0x1bd520, true, true},
"AMEVCNTR010_EL0": a64SysReg{0x1bd540, true, true},
"AMEVCNTR011_EL0": a64SysReg{0x1bd560, true, true},
"AMEVCNTR012_EL0": a64SysReg{0x1bd580, true, true},
"AMEVCNTR013_EL0": a64SysReg{0x1bd5a0, true, true},
"AMEVCNTR014_EL0": a64SysReg{0x1bd5c0, true, true},
"AMEVCNTR015_EL0": a64SysReg{0x1bd5e0, true, true},
"AMEVCNTR10_EL0": a64SysReg{0x1bdc00, true, true},
"AMEVCNTR11_EL0": a64SysReg{0x1bdc20, true, true},
"AMEVCNTR12_EL0": a64SysReg{0x1bdc40, true, true},
"AMEVCNTR13_EL0": a64SysReg{0x1bdc60, true, true},
"AMEVCNTR14_EL0": a64SysReg{0x1bdc80, true, true},
"AMEVCNTR15_EL0": a64SysReg{0x1bdca0, true, true},
"AMEVCNTR16_EL0": a64SysReg{0x1bdcc0, true, true},
"AMEVCNTR17_EL0": a64SysReg{0x1bdce0, true, true},
"AMEVCNTR18_EL0": a64SysReg{0x1bdd00, true, true},
"AMEVCNTR19_EL0": a64SysReg{0x1bdd20, true, true},
"AMEVCNTR110_EL0": a64SysReg{0x1bdd40, true, true},
"AMEVCNTR111_EL0": a64SysReg{0x1bdd60, true, true},
"AMEVCNTR112_EL0": a64SysReg{0x1bdd80, true, true},
"AMEVCNTR113_EL0": a64SysReg{0x1bdda0, true, true},
"AMEVCNTR114_EL0": a64SysReg{0x1bddc0, true, true},
"AMEVCNTR115_EL0": a64SysReg{0x1bdde0, true, true},
"AMEVTYPER00_EL0": a64SysReg{0x1bd600, true, false},
"AMEVTYPER01_EL0": a64SysReg{0x1bd620, true, false},
"AMEVTYPER02_EL0": a64SysReg{0x1bd640, true, false},
"AMEVTYPER03_EL0": a64SysReg{0x1bd660, true, false},
"AMEVTYPER04_EL0": a64SysReg{0x1bd680, true, false},
"AMEVTYPER05_EL0": a64SysReg{0x1bd6a0, true, false},
"AMEVTYPER06_EL0": a64SysReg{0x1bd6c0, true, false},
"AMEVTYPER07_EL0": a64SysReg{0x1bd6e0, true, false},
"AMEVTYPER08_EL0": a64SysReg{0x1bd700, true, false},
"AMEVTYPER09_EL0": a64SysReg{0x1bd720, true, false},
"AMEVTYPER010_EL0": a64SysReg{0x1bd740, true, false},
"AMEVTYPER011_EL0": a64SysReg{0x1bd760, true, false},
"AMEVTYPER012_EL0": a64SysReg{0x1bd780, true, false},
"AMEVTYPER013_EL0": a64SysReg{0x1bd7a0, true, false},
"AMEVTYPER014_EL0": a64SysReg{0x1bd7c0, true, false},
"AMEVTYPER015_EL0": a64SysReg{0x1bd7e0, true, false},
"AMEVTYPER10_EL0": a64SysReg{0x1bde00, true, true},
"AMEVTYPER11_EL0": a64SysReg{0x1bde20, true, true},
"AMEVTYPER12_EL0": a64SysReg{0x1bde40, true, true},
"AMEVTYPER13_EL0": a64SysReg{0x1bde60, true, true},
"AMEVTYPER14_EL0": a64SysReg{0x1bde80, true, true},
"AMEVTYPER15_EL0": a64SysReg{0x1bdea0, true, true},
"AMEVTYPER16_EL0": a64SysReg{0x1bdec0, true, true},
"AMEVTYPER17_EL0": a64SysReg{0x1bdee0, true, true},
"AMEVTYPER18_EL0": a64SysReg{0x1bdf00, true, true},
"AMEVTYPER19_EL0": a64SysReg{0x1bdf20, true, true},
"AMEVTYPER110_EL0": a64SysReg{0x1bdf40, true, true},
"AMEVTYPER111_EL0": a64SysReg{0x1bdf60, true, true},
"AMEVTYPER112_EL0": a64SysReg{0x1bdf80, true, true},
"AMEVTYPER113_EL0": a64SysReg{0x1bdfa0, true, true},
"AMEVTYPER114_EL0": a64SysReg{0x1bdfc0, true, true},
"AMEVTYPER115_EL0": a64SysReg{0x1bdfe0, true, true},
"AMUSERENR_EL0": a64SysReg{0x1bd260, true, true},
"APDAKeyHi_EL1": a64SysReg{0x182220, true, true},
"APDAKeyLo_EL1": a64SysReg{0x182200, true, true},
"APDBKeyHi_EL1": a64SysReg{0x182260, true, true},
"APDBKeyLo_EL1": a64SysReg{0x182240, true, true},
"APGAKeyHi_EL1": a64SysReg{0x182320, true, true},
"APGAKeyLo_EL1": a64SysReg{0x182300, true, true},
"APIAKeyHi_EL1": a64SysReg{0x182120, true, true},
"APIAKeyLo_EL1": a64SysReg{0x182100, true, true},
"APIBKeyHi_EL1": a64SysReg{0x182160, true, true},
"APIBKeyLo_EL1": a64SysReg{0x182140, true, true},
"CCSIDR2_EL1": a64SysReg{0x190040, true, false},
"CCSIDR_EL1": a64SysReg{0x190000, true, false},
"CLIDR_EL1": a64SysReg{0x190020, true, false},
"CNTFRQ_EL0": a64SysReg{0x1be000, true, true},
"CNTKCTL_EL1": a64SysReg{0x18e100, true, true},
"CNTP_CTL_EL0": a64SysReg{0x1be220, true, true},
"CNTP_CVAL_EL0": a64SysReg{0x1be240, true, true},
"CNTP_TVAL_EL0": a64SysReg{0x1be200, true, true},
"CNTPCT_EL0": a64SysReg{0x1be020, true, false},
"CNTPS_CTL_EL1": a64SysReg{0x1fe220, true, true},
"CNTPS_CVAL_EL1": a64SysReg{0x1fe240, true, true},
"CNTPS_TVAL_EL1": a64SysReg{0x1fe200, true, true},
"CNTV_CTL_EL0": a64SysReg{0x1be320, true, true},
"CNTV_CVAL_EL0": a64SysReg{0x1be340, true, true},
"CNTV_TVAL_EL0": a64SysReg{0x1be300, true, true},
"CNTVCT_EL0": a64SysReg{0x1be040, true, false},
"CONTEXTIDR_EL1": a64SysReg{0x18d020, true, true},
"CPACR_EL1": a64SysReg{0x181040, true, true},
"CSSELR_EL1": a64SysReg{0x1a0000, true, true},
"CTR_EL0": a64SysReg{0x1b0020, true, false},
"CurrentEL": a64SysReg{0x184240, true, false},
"DAIF": a64SysReg{0x1b4220, true, true},
"DBGAUTHSTATUS_EL1": a64SysReg{0x107ec0, true, false},
"DBGBCR0_EL1": a64SysReg{0x1000a0, true, true},
"DBGBCR1_EL1": a64SysReg{0x1001a0, true, true},
"DBGBCR2_EL1": a64SysReg{0x1002a0, true, true},
"DBGBCR3_EL1": a64SysReg{0x1003a0, true, true},
"DBGBCR4_EL1": a64SysReg{0x1004a0, true, true},
"DBGBCR5_EL1": a64SysReg{0x1005a0, true, true},
"DBGBCR6_EL1": a64SysReg{0x1006a0, true, true},
"DBGBCR7_EL1": a64SysReg{0x1007a0, true, true},
"DBGBCR8_EL1": a64SysReg{0x1008a0, true, true},
"DBGBCR9_EL1": a64SysReg{0x1009a0, true, true},
"DBGBCR10_EL1": a64SysReg{0x100aa0, true, true},
"DBGBCR11_EL1": a64SysReg{0x100ba0, true, true},
"DBGBCR12_EL1": a64SysReg{0x100ca0, true, true},
"DBGBCR13_EL1": a64SysReg{0x100da0, true, true},
"DBGBCR14_EL1": a64SysReg{0x100ea0, true, true},
"DBGBCR15_EL1": a64SysReg{0x100fa0, true, true},
"DBGBVR0_EL1": a64SysReg{0x100080, true, true},
"DBGBVR1_EL1": a64SysReg{0x100180, true, true},
"DBGBVR2_EL1": a64SysReg{0x100280, true, true},
"DBGBVR3_EL1": a64SysReg{0x100380, true, true},
"DBGBVR4_EL1": a64SysReg{0x100480, true, true},
"DBGBVR5_EL1": a64SysReg{0x100580, true, true},
"DBGBVR6_EL1": a64SysReg{0x100680, true, true},
"DBGBVR7_EL1": a64SysReg{0x100780, true, true},
"DBGBVR8_EL1": a64SysReg{0x100880, true, true},
"DBGBVR9_EL1": a64SysReg{0x100980, true, true},
"DBGBVR10_EL1": a64SysReg{0x100a80, true, true},
"DBGBVR11_EL1": a64SysReg{0x100b80, true, true},
"DBGBVR12_EL1": a64SysReg{0x100c80, true, true},
"DBGBVR13_EL1": a64SysReg{0x100d80, true, true},
"DBGBVR14_EL1": a64SysReg{0x100e80, true, true},
"DBGBVR15_EL1": a64SysReg{0x100f80, true, true},
"DBGCLAIMCLR_EL1": a64SysReg{0x1079c0, true, true},
"DBGCLAIMSET_EL1": a64SysReg{0x1078c0, true, true},
"DBGDTR_EL0": a64SysReg{0x130400, true, true},
"DBGDTRRX_EL0": a64SysReg{0x130500, true, false},
"DBGDTRTX_EL0": a64SysReg{0x130500, false, true},
"DBGPRCR_EL1": a64SysReg{0x101480, true, true},
"DBGWCR0_EL1": a64SysReg{0x1000e0, true, true},
"DBGWCR1_EL1": a64SysReg{0x1001e0, true, true},
"DBGWCR2_EL1": a64SysReg{0x1002e0, true, true},
"DBGWCR3_EL1": a64SysReg{0x1003e0, true, true},
"DBGWCR4_EL1": a64SysReg{0x1004e0, true, true},
"DBGWCR5_EL1": a64SysReg{0x1005e0, true, true},
"DBGWCR6_EL1": a64SysReg{0x1006e0, true, true},
"DBGWCR7_EL1": a64SysReg{0x1007e0, true, true},
"DBGWCR8_EL1": a64SysReg{0x1008e0, true, true},
"DBGWCR9_EL1": a64SysReg{0x1009e0, true, true},
"DBGWCR10_EL1": a64SysReg{0x100ae0, true, true},
"DBGWCR11_EL1": a64SysReg{0x100be0, true, true},
"DBGWCR12_EL1": a64SysReg{0x100ce0, true, true},
"DBGWCR13_EL1": a64SysReg{0x100de0, true, true},
"DBGWCR14_EL1": a64SysReg{0x100ee0, true, true},
"DBGWCR15_EL1": a64SysReg{0x100fe0, true, true},
"DBGWVR0_EL1": a64SysReg{0x1000c0, true, true},
"DBGWVR1_EL1": a64SysReg{0x1001c0, true, true},
"DBGWVR2_EL1": a64SysReg{0x1002c0, true, true},
"DBGWVR3_EL1": a64SysReg{0x1003c0, true, true},
"DBGWVR4_EL1": a64SysReg{0x1004c0, true, true},
"DBGWVR5_EL1": a64SysReg{0x1005c0, true, true},
"DBGWVR6_EL1": a64SysReg{0x1006c0, true, true},
"DBGWVR7_EL1": a64SysReg{0x1007c0, true, true},
"DBGWVR8_EL1": a64SysReg{0x1008c0, true, true},
"DBGWVR9_EL1": a64SysReg{0x1009c0, true, true},
"DBGWVR10_EL1": a64SysReg{0x100ac0, true, true},
"DBGWVR11_EL1": a64SysReg{0x100bc0, true, true},
"DBGWVR12_EL1": a64SysReg{0x100cc0, true, true},
"DBGWVR13_EL1": a64SysReg{0x100dc0, true, true},
"DBGWVR14_EL1": a64SysReg{0x100ec0, true, true},
"DBGWVR15_EL1": a64SysReg{0x100fc0, true, true},
"DCZID_EL0": a64SysReg{0x1b00e0, true, false},
"DISR_EL1": a64SysReg{0x18c120, true, true},
"DIT": a64SysReg{0x1b42a0, true, true},
"DLR_EL0": a64SysReg{0x1b4520, true, true},
"DSPSR_EL0": a64SysReg{0x1b4500, true, true},
"ELR_EL1": a64SysReg{0x184020, true, true},
"ERRIDR_EL1": a64SysReg{0x185300, true, false},
"ERRSELR_EL1": a64SysReg{0x185320, true, true},
"ERXADDR_EL1": a64SysReg{0x185460, true, true},
"ERXCTLR_EL1": a64SysReg{0x185420, true, true},
"ERXFR_EL1": a64SysReg{0x185400, true, false},
"ERXMISC0_EL1": a64SysReg{0x185500, true, true},
"ERXMISC1_EL1": a64SysReg{0x185520, true, true},
"ERXMISC2_EL1": a64SysReg{0x185540, true, true},
"ERXMISC3_EL1": a64SysReg{0x185560, true, true},
"ERXPFGCDN_EL1": a64SysReg{0x1854c0, true, true},
"ERXPFGCTL_EL1": a64SysReg{0x1854a0, true, true},
"ERXPFGF_EL1": a64SysReg{0x185480, true, false},
"ERXSTATUS_EL1": a64SysReg{0x185440, true, true},
"ESR_EL1": a64SysReg{0x185200, true, true},
"FAR_EL1": a64SysReg{0x186000, true, true},
"FPCR": a64SysReg{0x1b4400, true, true},
"FPSR": a64SysReg{0x1b4420, true, true},
"GCR_EL1": a64SysReg{0x1810c0, true, true},
"GMID_EL1": a64SysReg{0x31400, true, false},
"ICC_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICC_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICC_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICC_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICC_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICC_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICC_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICC_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICC_ASGI1R_EL1": a64SysReg{0x18cbc0, false, true},
"ICC_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICC_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICC_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICC_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICC_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICC_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICC_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICC_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICC_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICC_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICC_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICC_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICC_PMR_EL1": a64SysReg{0x184600, true, true},
"ICC_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ICC_SGI0R_EL1": a64SysReg{0x18cbe0, false, true},
"ICC_SGI1R_EL1": a64SysReg{0x18cba0, false, true},
"ICC_SRE_EL1": a64SysReg{0x18cca0, true, true},
"ICV_AP0R0_EL1": a64SysReg{0x18c880, true, true},
"ICV_AP0R1_EL1": a64SysReg{0x18c8a0, true, true},
"ICV_AP0R2_EL1": a64SysReg{0x18c8c0, true, true},
"ICV_AP0R3_EL1": a64SysReg{0x18c8e0, true, true},
"ICV_AP1R0_EL1": a64SysReg{0x18c900, true, true},
"ICV_AP1R1_EL1": a64SysReg{0x18c920, true, true},
"ICV_AP1R2_EL1": a64SysReg{0x18c940, true, true},
"ICV_AP1R3_EL1": a64SysReg{0x18c960, true, true},
"ICV_BPR0_EL1": a64SysReg{0x18c860, true, true},
"ICV_BPR1_EL1": a64SysReg{0x18cc60, true, true},
"ICV_CTLR_EL1": a64SysReg{0x18cc80, true, true},
"ICV_DIR_EL1": a64SysReg{0x18cb20, false, true},
"ICV_EOIR0_EL1": a64SysReg{0x18c820, false, true},
"ICV_EOIR1_EL1": a64SysReg{0x18cc20, false, true},
"ICV_HPPIR0_EL1": a64SysReg{0x18c840, true, false},
"ICV_HPPIR1_EL1": a64SysReg{0x18cc40, true, false},
"ICV_IAR0_EL1": a64SysReg{0x18c800, true, false},
"ICV_IAR1_EL1": a64SysReg{0x18cc00, true, false},
"ICV_IGRPEN0_EL1": a64SysReg{0x18ccc0, true, true},
"ICV_IGRPEN1_EL1": a64SysReg{0x18cce0, true, true},
"ICV_PMR_EL1": a64SysReg{0x184600, true, true},
"ICV_RPR_EL1": a64SysReg{0x18cb60, true, false},
"ID_AA64AFR0_EL1": a64SysReg{0x180580, true, false},
"ID_AA64AFR1_EL1": a64SysReg{0x1805a0, true, false},
"ID_AA64DFR0_EL1": a64SysReg{0x180500, true, false},
"ID_AA64DFR1_EL1": a64SysReg{0x180520, true, false},
"ID_AA64ISAR0_EL1": a64SysReg{0x180600, true, false},
"ID_AA64ISAR1_EL1": a64SysReg{0x180620, true, false},
"ID_AA64MMFR0_EL1": a64SysReg{0x180700, true, false},
"ID_AA64MMFR1_EL1": a64SysReg{0x180720, true, false},
"ID_AA64MMFR2_EL1": a64SysReg{0x180740, true, false},
"ID_AA64PFR0_EL1": a64SysReg{0x180400, true, false},
"ID_AA64PFR1_EL1": a64SysReg{0x180420, true, false},
"ID_AA64ZFR0_EL1": a64SysReg{0x180480, true, false},
"ID_AFR0_EL1": a64SysReg{0x180160, true, false},
"ID_DFR0_EL1": a64SysReg{0x180140, true, false},
"ID_ISAR0_EL1": a64SysReg{0x180200, true, false},
"ID_ISAR1_EL1": a64SysReg{0x180220, true, false},
"ID_ISAR2_EL1": a64SysReg{0x180240, true, false},
"ID_ISAR3_EL1": a64SysReg{0x180260, true, false},
"ID_ISAR4_EL1": a64SysReg{0x180280, true, false},
"ID_ISAR5_EL1": a64SysReg{0x1802a0, true, false},
"ID_ISAR6_EL1": a64SysReg{0x1802e0, true, false},
"ID_MMFR0_EL1": a64SysReg{0x180180, true, false},
"ID_MMFR1_EL1": a64SysReg{0x1801a0, true, false},
"ID_MMFR2_EL1": a64SysReg{0x1801c0, true, false},
"ID_MMFR3_EL1": a64SysReg{0x1801e0, true, false},
"ID_MMFR4_EL1": a64SysReg{0x1802c0, true, false},
"ID_PFR0_EL1": a64SysReg{0x180100, true, false},
"ID_PFR1_EL1": a64SysReg{0x180120, true, false},
"ID_PFR2_EL1": a64SysReg{0x180380, true, false},
"ISR_EL1": a64SysReg{0x18c100, true, false},
"LORC_EL1": a64SysReg{0x18a460, true, true},
"LOREA_EL1": a64SysReg{0x18a420, true, true},
"LORID_EL1": a64SysReg{0x18a4e0, true, false},
"LORN_EL1": a64SysReg{0x18a440, true, true},
"LORSA_EL1": a64SysReg{0x18a400, true, true},
"MAIR_EL1": a64SysReg{0x18a200, true, true},
"MDCCINT_EL1": a64SysReg{0x100200, true, true},
"MDCCSR_EL0": a64SysReg{0x130100, true, false},
"MDRAR_EL1": a64SysReg{0x101000, true, false},
"MDSCR_EL1": a64SysReg{0x100240, true, true},
"MIDR_EL1": a64SysReg{0x180000, true, false},
"MPAM0_EL1": a64SysReg{0x18a520, true, true},
"MPAM1_EL1": a64SysReg{0x18a500, true, true},
"MPAMIDR_EL1": a64SysReg{0x18a480, true, false},
"MPIDR_EL1": a64SysReg{0x1800a0, true, false},
"MVFR0_EL1": a64SysReg{0x180300, true, false},
"MVFR1_EL1": a64SysReg{0x180320, true, false},
"MVFR2_EL1": a64SysReg{0x180340, true, false},
"NZCV": a64SysReg{0x1b4200, true, true},
"OSDLR_EL1": a64SysReg{0x101380, true, true},
"OSDTRRX_EL1": a64SysReg{0x100040, true, true},
"OSDTRTX_EL1": a64SysReg{0x100340, true, true},
"OSECCR_EL1": a64SysReg{0x100640, true, true},
"OSLAR_EL1": a64SysReg{0x101080, false, true},
"OSLSR_EL1": a64SysReg{0x101180, true, false},
"PAN": a64SysReg{0x184260, true, true},
"PAR_EL1": a64SysReg{0x187400, true, true},
"PMBIDR_EL1": a64SysReg{0x189ae0, true, false},
"PMBLIMITR_EL1": a64SysReg{0x189a00, true, true},
"PMBPTR_EL1": a64SysReg{0x189a20, true, true},
"PMBSR_EL1": a64SysReg{0x189a60, true, true},
"PMCCFILTR_EL0": a64SysReg{0x1befe0, true, true},
"PMCCNTR_EL0": a64SysReg{0x1b9d00, true, true},
"PMCEID0_EL0": a64SysReg{0x1b9cc0, true, false},
"PMCEID1_EL0": a64SysReg{0x1b9ce0, true, false},
"PMCNTENCLR_EL0": a64SysReg{0x1b9c40, true, true},
"PMCNTENSET_EL0": a64SysReg{0x1b9c20, true, true},
"PMCR_EL0": a64SysReg{0x1b9c00, true, true},
"PMEVCNTR0_EL0": a64SysReg{0x1be800, true, true},
"PMEVCNTR1_EL0": a64SysReg{0x1be820, true, true},
"PMEVCNTR2_EL0": a64SysReg{0x1be840, true, true},
"PMEVCNTR3_EL0": a64SysReg{0x1be860, true, true},
"PMEVCNTR4_EL0": a64SysReg{0x1be880, true, true},
"PMEVCNTR5_EL0": a64SysReg{0x1be8a0, true, true},
"PMEVCNTR6_EL0": a64SysReg{0x1be8c0, true, true},
"PMEVCNTR7_EL0": a64SysReg{0x1be8e0, true, true},
"PMEVCNTR8_EL0": a64SysReg{0x1be900, true, true},
"PMEVCNTR9_EL0": a64SysReg{0x1be920, true, true},
"PMEVCNTR10_EL0": a64SysReg{0x1be940, true, true},
"PMEVCNTR11_EL0": a64SysReg{0x1be960, true, true},
"PMEVCNTR12_EL0": a64SysReg{0x1be980, true, true},
"PMEVCNTR13_EL0": a64SysReg{0x1be9a0, true, true},
"PMEVCNTR14_EL0": a64SysReg{0x1be9c0, true, true},
"PMEVCNTR15_EL0": a64SysReg{0x1be9e0, true, true},
"PMEVCNTR16_EL0": a64SysReg{0x1bea00, true, true},
"PMEVCNTR17_EL0": a64SysReg{0x1bea20, true, true},
"PMEVCNTR18_EL0": a64SysReg{0x1bea40, true, true},
"PMEVCNTR19_EL0": a64SysReg{0x1bea60, true, true},
"PMEVCNTR20_EL0": a64SysReg{0x1bea80, true, true},
"PMEVCNTR21_EL0": a64SysReg{0x1beaa0, true, true},
"PMEVCNTR22_EL0": a64SysReg{0x1beac0, true, true},
"PMEVCNTR23_EL0": a64SysReg{0x1beae0, true, true},
"PMEVCNTR24_EL0": a64SysReg{0x1beb00, true, true},
"PMEVCNTR25_EL0": a64SysReg{0x1beb20, true, true},
"PMEVCNTR26_EL0": a64SysReg{0x1beb40, true, true},
"PMEVCNTR27_EL0": a64SysReg{0x1beb60, true, true},
"PMEVCNTR28_EL0": a64SysReg{0x1beb80, true, true},
"PMEVCNTR29_EL0": a64SysReg{0x1beba0, true, true},
"PMEVCNTR30_EL0": a64SysReg{0x1bebc0, true, true},
"PMEVTYPER0_EL0": a64SysReg{0x1bec00, true, true},
"PMEVTYPER1_EL0": a64SysReg{0x1bec20, true, true},
"PMEVTYPER2_EL0": a64SysReg{0x1bec40, true, true},
"PMEVTYPER3_EL0": a64SysReg{0x1bec60, true, true},
"PMEVTYPER4_EL0": a64SysReg{0x1bec80, true, true},
"PMEVTYPER5_EL0": a64SysReg{0x1beca0, true, true},
"PMEVTYPER6_EL0": a64SysReg{0x1becc0, true, true},
"PMEVTYPER7_EL0": a64SysReg{0x1bece0, true, true},
"PMEVTYPER8_EL0": a64SysReg{0x1bed00, true, true},
"PMEVTYPER9_EL0": a64SysReg{0x1bed20, true, true},
"PMEVTYPER10_EL0": a64SysReg{0x1bed40, true, true},
"PMEVTYPER11_EL0": a64SysReg{0x1bed60, true, true},
"PMEVTYPER12_EL0": a64SysReg{0x1bed80, true, true},
"PMEVTYPER13_EL0": a64SysReg{0x1beda0, true, true},
"PMEVTYPER14_EL0": a64SysReg{0x1bedc0, true, true},
"PMEVTYPER15_EL0": a64SysReg{0x1bede0, true, true},
"PMEVTYPER16_EL0": a64SysReg{0x1bee00, true, true},
"PMEVTYPER17_EL0": a64SysReg{0x1bee20, true, true},
"PMEVTYPER18_EL0": a64SysReg{0x1bee40, true, true},
"PMEVTYPER19_EL0": a64SysReg{0x1bee60, true, true},
"PMEVTYPER20_EL0": a64SysReg{0x1bee80, true, true},
"PMEVTYPER21_EL0": a64SysReg{0x1beea0, true, true},
"PMEVTYPER22_EL0": a64SysReg{0x1beec0, true, true},
"PMEVTYPER23_EL0": a64SysReg{0x1beee0, true, true},
"PMEVTYPER24_EL0": a64SysReg{0x1bef00, true, true},
"PMEVTYPER25_EL0": a64SysReg{0x1bef20, true, true},
"PMEVTYPER26_EL0": a64SysReg{0x1bef40, true, true},
"PMEVTYPER27_EL0": a64SysReg{0x1bef60, true, true},
"PMEVTYPER28_EL0": a64SysReg{0x1bef80, true, true},
"PMEVTYPER29_EL0": a64SysReg{0x1befa0, true, true},
"PMEVTYPER30_EL0": a64SysReg{0x1befc0, true, true},
"PMINTENCLR_EL1": a64SysReg{0x189e40, true, true},
"PMINTENSET_EL1": a64SysReg{0x189e20, true, true},
"PMMIR_EL1": a64SysReg{0x189ec0, true, false},
"PMOVSCLR_EL0": a64SysReg{0x1b9c60, true, true},
"PMOVSSET_EL0": a64SysReg{0x1b9e60, true, true},
"PMSCR_EL1": a64SysReg{0x189900, true, true},
"PMSELR_EL0": a64SysReg{0x1b9ca0, true, true},
"PMSEVFR_EL1": a64SysReg{0x1899a0, true, true},
"PMSFCR_EL1": a64SysReg{0x189980, true, true},
"PMSICR_EL1": a64SysReg{0x189940, true, true},
"PMSIDR_EL1": a64SysReg{0x1899e0, true, false},
"PMSIRR_EL1": a64SysReg{0x189960, true, true},
"PMSLATFR_EL1": a64SysReg{0x1899c0, true, true},
"PMSWINC_EL0": a64SysReg{0x1b9c80, false, true},
"PMUSERENR_EL0": a64SysReg{0x1b9e00, true, true},
"PMXEVCNTR_EL0": a64SysReg{0x1b9d40, true, true},
"PMXEVTYPER_EL0": a64SysReg{0x1b9d20, true, true},
"REVIDR_EL1": a64SysReg{0x1800c0, true, false},
"RGSR_EL1": a64SysReg{0x1810a0, true, true},
"RMR_EL1": a64SysReg{0x18c040, true, true},
"RNDR": a64SysReg{0x1b2400, true, false},
"RNDRRS": a64SysReg{0x1b2420, true, false},
"RVBAR_EL1": a64SysReg{0x18c020, true, false},
"SCTLR_EL1": a64SysReg{0x181000, true, true},
"SCXTNUM_EL0": a64SysReg{0x1bd0e0, true, true},
"SCXTNUM_EL1": a64SysReg{0x18d0e0, true, true},
"SP_EL0": a64SysReg{0x184100, true, true},
"SP_EL1": a64SysReg{0x1c4100, true, true},
"SPSel": a64SysReg{0x184200, true, true},
"SPSR_abt": a64SysReg{0x1c4320, true, true},
"SPSR_EL1": a64SysReg{0x184000, true, true},
"SPSR_fiq": a64SysReg{0x1c4360, true, true},
"SPSR_irq": a64SysReg{0x1c4300, true, true},
"SPSR_und": a64SysReg{0x1c4340, true, true},
"SSBS": a64SysReg{0x1b42c0, true, true},
"TCO": a64SysReg{0x1b42e0, true, true},
"TCR_EL1": a64SysReg{0x182040, true, true},
"TFSR_EL1": a64SysReg{0x185600, true, true},
"TFSRE0_EL1": a64SysReg{0x185620, true, true},
"TPIDR_EL0": a64SysReg{0x1bd040, true, true},
"TPIDR_EL1": a64SysReg{0x18d080, true, true},
"TPIDRRO_EL0": a64SysReg{0x1bd060, true, true},
"TRFCR_EL1": a64SysReg{0x181220, true, true},
"TTBR0_EL1": a64SysReg{0x182000, true, true},
"TTBR1_EL1": a64SysReg{0x182020, true, true},
"UAO": a64SysReg{0x184280, true, true},
"VBAR_EL1": a64SysReg{0x18c000, true, true},
"ZCR_EL1": a64SysReg{0x181200, true, true},
}
// a64SysInst is one TLBI alias: the fields the SYS encoding carries beside
// the fixed op0 = 01 and CRn = 8.
type a64SysInst struct {
op1, cm, op2 uint32
}
// a64TLBIOps maps the TLBI operation names to their fields; the register
// operand is optional and defaults to ZR.
var a64TLBIOps = map[string]a64SysInst{
"ALLE1": {0x4, 0x7, 0x4},
"ALLE1IS": {0x4, 0x3, 0x4},
"ALLE1OS": {0x4, 0x1, 0x4},
"ALLE2": {0x4, 0x7, 0x0},
"ALLE2IS": {0x4, 0x3, 0x0},
"ALLE2OS": {0x4, 0x1, 0x0},
"ALLE3": {0x6, 0x7, 0x0},
"ALLE3IS": {0x6, 0x3, 0x0},
"ALLE3OS": {0x6, 0x1, 0x0},
"ASIDE1": {0x0, 0x7, 0x2},
"ASIDE1IS": {0x0, 0x3, 0x2},
"ASIDE1OS": {0x0, 0x1, 0x2},
"IPAS2E1": {0x4, 0x4, 0x1},
"IPAS2E1IS": {0x4, 0x0, 0x1},
"IPAS2E1OS": {0x4, 0x4, 0x0},
"IPAS2LE1": {0x4, 0x4, 0x5},
"IPAS2LE1IS": {0x4, 0x0, 0x5},
"IPAS2LE1OS": {0x4, 0x4, 0x4},
"RIPAS2E1": {0x4, 0x4, 0x2},
"RIPAS2E1IS": {0x4, 0x0, 0x2},
"RIPAS2E1OS": {0x4, 0x4, 0x3},
"RIPAS2LE1": {0x4, 0x4, 0x6},
"RIPAS2LE1IS": {0x4, 0x0, 0x6},
"RIPAS2LE1OS": {0x4, 0x4, 0x7},
"RVAAE1": {0x0, 0x6, 0x3},
"RVAAE1IS": {0x0, 0x2, 0x3},
"RVAAE1OS": {0x0, 0x5, 0x3},
"RVAALE1": {0x0, 0x6, 0x7},
"RVAALE1IS": {0x0, 0x2, 0x7},
"RVAALE1OS": {0x0, 0x5, 0x7},
"RVAE1": {0x0, 0x6, 0x1},
"RVAE1IS": {0x0, 0x2, 0x1},
"RVAE1OS": {0x0, 0x5, 0x1},
"RVAE2": {0x4, 0x6, 0x1},
"RVAE2IS": {0x4, 0x2, 0x1},
"RVAE2OS": {0x4, 0x5, 0x1},
"RVAE3": {0x6, 0x6, 0x1},
"RVAE3IS": {0x6, 0x2, 0x1},
"RVAE3OS": {0x6, 0x5, 0x1},
"RVALE1": {0x0, 0x6, 0x5},
"RVALE1IS": {0x0, 0x2, 0x5},
"RVALE1OS": {0x0, 0x5, 0x5},
"RVALE2": {0x4, 0x6, 0x5},
"RVALE2IS": {0x4, 0x2, 0x5},
"RVALE2OS": {0x4, 0x5, 0x5},
"RVALE3": {0x6, 0x6, 0x5},
"RVALE3IS": {0x6, 0x2, 0x5},
"RVALE3OS": {0x6, 0x5, 0x5},
"VAAE1": {0x0, 0x7, 0x3},
"VAAE1IS": {0x0, 0x3, 0x3},
"VAAE1OS": {0x0, 0x1, 0x3},
"VAALE1": {0x0, 0x7, 0x7},
"VAALE1IS": {0x0, 0x3, 0x7},
"VAALE1OS": {0x0, 0x1, 0x7},
"VAE1": {0x0, 0x7, 0x1},
"VAE1IS": {0x0, 0x3, 0x1},
"VAE1OS": {0x0, 0x1, 0x1},
"VAE2": {0x4, 0x7, 0x1},
"VAE2IS": {0x4, 0x3, 0x1},
"VAE2OS": {0x4, 0x1, 0x1},
"VAE3": {0x6, 0x7, 0x1},
"VAE3IS": {0x6, 0x3, 0x1},
"VAE3OS": {0x6, 0x1, 0x1},
"VALE1": {0x0, 0x7, 0x5},
"VALE1IS": {0x0, 0x3, 0x5},
"VALE1OS": {0x0, 0x1, 0x5},
"VALE2": {0x4, 0x7, 0x5},
"VALE2IS": {0x4, 0x3, 0x5},
"VALE2OS": {0x4, 0x1, 0x5},
"VALE3": {0x6, 0x7, 0x5},
"VALE3IS": {0x6, 0x3, 0x5},
"VALE3OS": {0x6, 0x1, 0x5},
"VMALLE1": {0x0, 0x7, 0x0},
"VMALLE1IS": {0x0, 0x3, 0x0},
"VMALLE1OS": {0x0, 0x1, 0x0},
"VMALLS12E1": {0x4, 0x7, 0x6},
"VMALLS12E1IS": {0x4, 0x3, 0x6},
"VMALLS12E1OS": {0x4, 0x1, 0x6},
}
// a64DCOps2 maps the DC operation names to their fields; the register
// operand is mandatory.
var a64DCOps2 = map[string]a64SysInst{
"CGDSW": {0x0, 0xa, 0x6},
"CGDVAC": {0x3, 0xa, 0x5},
"CGDVADP": {0x3, 0xd, 0x5},
"CGDVAP": {0x3, 0xc, 0x5},
"CGSW": {0x0, 0xa, 0x4},
"CGVAC": {0x3, 0xa, 0x3},
"CGVADP": {0x3, 0xd, 0x3},
"CGVAP": {0x3, 0xc, 0x3},
"CIGDSW": {0x0, 0xe, 0x6},
"CIGDVAC": {0x3, 0xe, 0x5},
"CIGSW": {0x0, 0xe, 0x4},
"CIGVAC": {0x3, 0xe, 0x3},
"CISW": {0x0, 0xe, 0x2},
"CIVAC": {0x3, 0xe, 0x1},
"CSW": {0x0, 0xa, 0x2},
"CVAC": {0x3, 0xa, 0x1},
"CVADP": {0x3, 0xd, 0x1},
"CVAP": {0x3, 0xc, 0x1},
"CVAU": {0x3, 0xb, 0x1},
"GVA": {0x3, 0x4, 0x3},
"GZVA": {0x3, 0x4, 0x4},
"IGDSW": {0x0, 0x6, 0x6},
"IGDVAC": {0x0, 0x6, 0x5},
"IGSW": {0x0, 0x6, 0x4},
"IGVAC": {0x0, 0x6, 0x3},
"ISW": {0x0, 0x6, 0x2},
"IVAC": {0x0, 0x6, 0x1},
"ZVA": {0x3, 0x4, 0x1},
}
// a64RPRFOps maps the range-prefetch operation names to their 6-bit values.
var a64RPRFOps = map[string]uint32{
"PLDKEEP": 0,
"PLDSTRM": 4,
"PSTKEEP": 1,
"PSTSTRM": 5,
}
+164
View File
@@ -0,0 +1,164 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
package asm
import (
"encoding/binary"
"fmt"
"os"
"path/filepath"
"slices"
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-sdk/parser"
)
// TestARM64SysRegsDifferential proves the whole system-register table against
// the toolchain at once: one TEXT whose body reads every register the table
// carries (and writes every writable one), assembled by gasm and by
// go tool asm, must agree byte for byte. A single wrong op0/op1/CRn/CRm/op2
// packing names its register through the first differing word.
func TestARM64SysRegsDifferential(t *testing.T) {
names := make([]string, 0, len(a64SysRegs))
for name := range a64SysRegs {
names = append(names, name)
}
slices.Sort(names)
var body strings.Builder
for i, name := range names {
// R18 is the arm64 platform register and R29-R31 carry dedicated
// meanings; a plain read/write destination keeps to R0-R17.
reg := fmt.Sprintf("R%d", i%18)
if a64SysRegs[name].read {
body.WriteString(fmt.Sprintf("\tMRS %s, %s\n", name, reg))
}
if a64SysRegs[name].write {
body.WriteString(fmt.Sprintf("\tMSR %s, %s\n", reg, name))
}
}
src := "#include \"textflag.h\"\n\nTEXT ·sysregs(SB), NOSPLIT, $0\n" + body.String() + "\tRET\n"
dir := t.TempDir()
path := filepath.Join(dir, "sysregs_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "sysregs")
}
// TestARM64FamiliesDifferential pins the non-sysreg families the arm64
// campaign added: the LSE compare-and-swap pairs, the VMOVI immediate, the
// SIMD narrow/long shift pairs, the VLD2/VLD3/VLD4 and VST2/VST3/VST4
// structure accesses with their post-index and replicate forms, LDPSW, the
// pointer-authentication hint and the DC maintenance operation. Every
// spelling is the toolchain's own, taken from its arm64 testdata, and the
// bytes must agree word for word.
func TestARM64FamiliesDifferential(t *testing.T) {
src := `#include "textflag.h"
TEXT ·families(SB), NOSPLIT, $0
CASPD (R2, R3), (R2), (R8, R9)
CASPW (R6, R7), (R8), (R4, R5)
VMOVI $82, V0.B16
VMOVI $146, V22.B16
VSSHLL $0, V1.B8, V2.H8
VSSHLL $7, V1.B8, V2.H8
VSSHLL2 $0, V1.B16, V2.H8
VSHRN $7, V1.H8, V0.B8
VSHRN2 $31, V1.D2, V0.S4
VLD2 (R29), [V23.H8, V24.H8]
VLD2.P 16(R0), [V18.B8, V19.B8]
VLD2.P (R1)(R2), [V15.S2, V16.S2]
VLD3 (R27), [V11.S4, V12.S4, V13.S4]
VLD3.P 48(RSP), [V11.S4, V12.S4, V13.S4]
VLD4 (R15), [V10.H4, V11.H4, V12.H4, V13.H4]
VLD4.P 32(R24), [V31.B8, V0.B8, V1.B8, V2.B8]
VLD1R (R1), [V9.B8]
VLD1R.P (R0), [V0.B16]
VLD1R.P 2(R1), [V2.H4]
VLD2R (R15), [V15.H4, V16.H4]
VLD2R.P 16(R0), [V0.D2, V1.D2]
VLD4R (R0), [V0.B8, V1.B8, V2.B8, V3.B8]
VLD4R.P 16(RSP), [V31.S4, V0.S4, V1.S4, V2.S4]
VST2 [V22.H8, V23.H8], (R23)
VST2.P [V14.H4, V15.H4], 16(R17)
VST2.P [V14.H4, V15.H4], (R3)(R17)
VST3 [V1.D2, V2.D2, V3.D2], (R11)
VST3.P [V18.S4, V19.S4, V20.S4], 48(R25)
VST4 [V22.D2, V23.D2, V24.D2, V25.D2], (R3)
VST4.P [V14.D2, V15.D2, V16.D2, V17.D2], 64(R15)
LDPSW (R0), (R1, R2)
LDPSW 4(R0), (R1, R2)
LDPSW -4(R0), (R1, R2)
PACIASP
DC IVAC, R1
RET
`
dir := t.TempDir()
path := filepath.Join(dir, "families_arm64.s")
if err := os.WriteFile(path, []byte(src), 0o644); err != nil {
t.Fatal(err)
}
assertARM64Differential(t, path, src, "families")
}
// assertARM64Differential assembles the same source with gasm and with the
// toolchain for arm64 and requires the named function's code bytes to agree.
// The live oracle is a deliberate-run comparison, so -short skips it (the
// push pipeline's mode); the golden bytes of the individual encoders are
// pinned separately in every mode.
func assertARM64Differential(t *testing.T, path, src, fn string) {
t.Helper()
oracle := oracleFuncCode(t, toolAsmObject(t, path, "arm64"))
// The oracle keys its functions by the qualified object name
// (pkg.name); match on the local part.
want := map[string][]byte{}
for name, code := range oracle {
if _, after, ok := strings.Cut(name, "."); ok {
want[after] = code
} else {
want[name] = code
}
}
if want[fn] == nil {
t.Fatalf("the oracle object carries no function %q (has %v)", fn, keysOf(want))
}
f, perrs := parser.Parse(path, src)
if len(perrs) > 0 {
t.Fatalf("parse: %v", perrs[0])
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := trimTrailingZeroWords(img.Code)
wantB := trimTrailingZeroWords(want[fn])
if len(got) != len(wantB) {
t.Fatalf("gasm %d bytes, oracle %d bytes", len(got), len(wantB))
}
for i := range wantB {
if got[i] != wantB[i] {
t.Fatalf("word %d differs: gasm %08x, oracle %08x", i/4,
binary.LittleEndian.Uint32(got[i:i+4]), binary.LittleEndian.Uint32(wantB[i:i+4]))
}
}
}
// trimTrailingZeroWords drops whole zero words off the end of a code span:
// an object pads a function to its alignment, and the raw image does not.
// A difference in the middle survives the trim untouched.
func trimTrailingZeroWords(b []byte) []byte {
for len(b) >= 4 {
last := b[len(b)-4:]
if last[0]|last[1]|last[2]|last[3] != 0 {
break
}
b = b[:len(b)-4]
}
return b
}