feat(asm): encode the arm64 system registers and structure loads

This commit is contained in:
2026-10-02 20:39:26 +02:00
parent e02918c17b
commit 2747fce7d3
4 changed files with 1375 additions and 79 deletions
+477 -65
View File
@@ -458,10 +458,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Excl(mnem, enc.op, ops)
}
// LSE atomics (LDADD, CAS, SWP).
// LSE atomics (LDADD, CAS, SWP) and the compare-and-swap pair.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
return encodeARM64LSEAtom(mnem, enc.op, ops)
}
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCASP {
return encodeARM64CASP(mnem, enc.op, ops)
}
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
@@ -555,9 +558,12 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
// only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
// an empty register-form spec. VSQSHL/VUQSHL keep their shift-by-
// immediate route when the first operand is an immediate.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
return encodeARM64SimdV(mnem, spec, ops)
if !arm64SimdShiftImmRoute(mnem, ops) {
return encodeARM64SimdV(mnem, spec, ops)
}
}
// Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV).
@@ -565,6 +571,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64SimdV2(mnem, spec, ops)
}
// Narrow/long/wide SIMD families whose size and Q bits read off one
// designated operand (VXTN, VSXTL, VUADDW, VUMULL, VSHRN, VSSHLL, VFCVTN
// and friends).
if spec, ok := a64SimdNLTable[mnem]; ok {
return encodeARM64SimdNL(mnem, spec, ops)
}
// SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR,
// VEXT).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 {
@@ -587,6 +600,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64ShiftImm(mnem, enc.op, ops)
}
// SIMD move immediate: VMOVI $imm8, Vd.B8/B16.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVMoviImm {
return encodeARM64MoviImm(ops)
}
// VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit {
return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits)
@@ -2568,6 +2586,263 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64CASP encodes the compare-and-swap pair: CASP (Rs, Rs+1), (Rn),
// (Rt, Rt+1). Both pairs must start on an even register and be contiguous;
// the second register of each pair rides no encoding field.
func encodeARM64CASP(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects (Rs, Rs+1), (Rn), (Rt, Rt+1), got %d operands", mnem, len(ops))
}
rs, rs1, ok := arm64PairOf(ops[0])
if !ok {
return nil, fmt.Errorf("%s expects a source register pair (Rs, Rs+1)", mnem)
}
rn, err := arm64ExclMem(mnem, ops[1])
if err != nil {
return nil, err
}
rt, rt1, ok := arm64PairOf(ops[2])
if !ok {
return nil, fmt.Errorf("%s expects a destination register pair (Rt, Rt+1)", mnem)
}
if rs&1 != 0 {
return nil, fmt.Errorf("%s: source register pair must start from an even register", mnem)
}
if rt&1 != 0 {
return nil, fmt.Errorf("%s: destination register pair must start from an even register", mnem)
}
if rs != rs1-1 {
return nil, fmt.Errorf("%s: source register pair must be contiguous", mnem)
}
if rt != rt1-1 {
return nil, fmt.Errorf("%s: destination register pair must be contiguous", mnem)
}
if rt == 31 {
return nil, fmt.Errorf("%s: illegal destination register", mnem)
}
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
}
// encodeARM64MoviImm encodes VMOVI $imm8, Vd.B8/B16: the modified-immediate
// form of the SIMD move (asm7.go case 86). Only the byte arrangements exist
// and the immediate is one unsigned byte.
func encodeARM64MoviImm(ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("VMOVI expects $immediate, Vd.<T>")
}
vd, ok := arm64VecOf(ops[1])
if !ok || vd.hasIdx || (vd.arr != "B8" && vd.arr != "B16") {
return nil, fmt.Errorf("VMOVI: destination arrangement must be B8 or B16")
}
imm := arm64Imm64(ops[0])
if imm < 0 || imm > 255 {
return nil, fmt.Errorf("VMOVI: immediate constant %d out of range (0..255)", imm)
}
q := uint32(0)
if vd.arr == "B16" {
q = 1 << 30
}
w := 0x0f00e400 | q | uint32(imm>>5&7)<<16 | uint32(imm&0x1f)<<5 | uint32(vd.reg)
return a64wordLE(w), nil
}
// arm64SimdShiftImmRoute reports whether a mnemonic carries both a shift-by-
// immediate and a register form and the operands spell the immediate one: the
// dedicated shift route keeps them.
func arm64SimdShiftImmRoute(mnem string, ops []*ast.Operand) bool {
if mnem != "VSQSHL" && mnem != "VUQSHL" {
return false
}
return len(ops) > 0 && isImmOperand(ops[0])
}
// arm64SimdNLArr describes one arrangement for the narrow/long/wide families:
// the element width in bytes and whether the spelling names the 128-bit form.
func arm64SimdNLArr(arr string) (esize int, wide bool, ok bool) {
switch arr {
case "B8", "B16":
return 1, arr == "B16", true
case "H4", "H8":
return 2, arr == "H8", true
case "S2", "S4":
return 4, arr == "S4", true
case "D1", "D2":
return 8, arr == "D2", true
}
return 0, false, false
}
// arm64SimdLongPair validates a long pairing (source narrow, destination
// wide): the destination element is twice the source's, the destination is
// always spelled the wide way (H8/S4/D2) and the source carries the 128-bit
// flag exactly for the .2 spellings.
func arm64SimdLongPair(mnem, src, dst string, two bool) error {
se, sw, ok1 := arm64SimdNLArr(src)
de, dw, ok2 := arm64SimdNLArr(dst)
if !ok1 || !ok2 || de != 2*se {
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
}
if !dw || sw != two {
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
}
return nil
}
// arm64SimdNarrowPair validates a narrow pairing (source wide, destination
// narrow): the mirror image of arm64SimdLongPair.
func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
se, sw, ok1 := arm64SimdNLArr(src)
de, dw, ok2 := arm64SimdNLArr(dst)
if !ok1 || !ok2 || se != 2*de {
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
}
if !sw || dw != two {
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
}
return nil
}
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
// instruction contributes: the driving arrangement's size and Q bits, or for
// the FCVT family only the Q bit, whose size field is fixed in the base.
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
if spec.qonly {
if two {
return 1 << 30
}
return 0
}
return a64ArrBits[a64ArrIndex(drive)]
}
// encodeARM64SimdNL encodes the narrow/long/wide SIMD families
// (a64SimdNLTable): XTN and FCVTN narrow a wide source, SXTL and FCVTL
// lengthen, the MULL/MLAL/MLSL group multiplies long, UADDW widens, and the
// SSHLL/USHLL and SHRN shifts carry their immediate in the immh:immb field.
// The size and Q bits read off the designated driving operand, and the .2
// spellings force the 128-bit side through their own arrangement.
func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]byte, error) {
two := strings.HasSuffix(mnem, "2")
// vecAt parses operand i as a vector register with an arrangement.
vecAt := func(i int) (a64Vec, bool) {
if i >= len(ops) {
return a64Vec{}, false
}
v, ok := arm64VecOf(ops[i])
if !ok || v.hasIdx {
return a64Vec{}, false
}
return v, true
}
switch spec.form {
case a64NLTwoNarrow, a64NLTwoLong:
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
vn, ok1 := vecAt(0)
vd, ok2 := vecAt(1)
if !ok1 || !ok2 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var drive string
var pairErr error
if spec.form == a64NLTwoNarrow {
// XTN/FCVTN: wide source into a narrow destination; the
// arrangement bits follow the destination.
drive, pairErr = vd.arr, arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two)
} else {
// SXTL/UXTL/FCVTL: narrow source into a wide destination; the
// arrangement bits follow the source.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
}
if pairErr != nil {
return nil, pairErr
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongMul, a64NLThreeWide:
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
}
vm, ok1 := vecAt(0)
vn, ok2 := vecAt(1)
vd, ok3 := vecAt(2)
if !ok1 || !ok2 || !ok3 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
var drive string
var pairErr error
if spec.form == a64NLThreeWide {
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
// the arrangement bits follow the wide side.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
} else {
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
// wide one; the arrangement bits follow the narrow source.
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
}
if pairErr != nil {
return nil, pairErr
}
if spec.form == a64NLThreeWide && vd.arr != vn.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vn.arr, vd.arr)
}
if spec.form == a64NLThreeLongMul && vm.arr != vn.arr {
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
}
arrBits := arm64SimdNLArrBits(spec, drive, two)
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
case a64NLThreeLongShift, a64NLThreeNarrowShift:
if len(ops) != 3 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects ($shift, Vn.<T>, Vd.<T>)", mnem)
}
sh := arm64Imm64(ops[0])
vn, ok1 := vecAt(1)
vd, ok2 := vecAt(2)
if !ok1 || !ok2 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
if spec.form == a64NLThreeLongShift {
// SSHLL/USHLL: the narrow source drives the immediate's size
// (immh:immb = esize + shift), so the arrangement bits carry
// the Q bit alone: the size field belongs to immh, and ORing
// the source's size bits into it would collide with immb.
if err := arm64SimdLongPair(mnem, vn.arr, vd.arr, two); err != nil {
return nil, err
}
se, _, _ := arm64SimdNLArr(vn.arr)
esize := se * 8
if sh < 0 || sh >= int64(esize) {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
}
var qBit uint32
if two {
qBit = 1 << 30
}
return a64wordLE(spec.base | uint32(esize+int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// SHRN: the narrow destination drives the immediate's size
// (immh:immb = esize - shift over the wide source element), so the
// arrangement bits carry the Q bit alone, exactly as above.
if err := arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two); err != nil {
return nil, err
}
se, _, _ := arm64SimdNLArr(vn.arr)
esize := se * 8
if sh < 1 || sh >= int64(esize) {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize-1)
}
var qBit uint32
if two {
qBit = 1 << 30
}
return a64wordLE(spec.base | uint32(esize-int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
// encodeARM64DP1 encodes a data-processing (1 source) instruction:
// RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd.
func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
@@ -2584,7 +2859,8 @@ func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, err
// encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the
// target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31
// selecting the page form.
// selecting the page form. An n(PC) operand resolves to the instruction's
// own address: the toolchain rewrites it away and encodes displacement 0.
func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
@@ -2593,14 +2869,17 @@ func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offset
if rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
target := resolve(arm64Label(ops[0]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel := int64(targetOff - pc)
if rel < -(1<<20) || rel >= 1<<20 {
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
var rel int64
if _, pcRel := arm64PCRelOffset(ops[0]); !pcRel {
target := resolve(arm64Label(ops[0]))
targetOff, ok := offsets[target]
if !ok {
return nil, fmt.Errorf("undefined label %q", target)
}
rel = int64(targetOff - pc)
if rel < -(1<<20) || rel >= 1<<20 {
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
}
}
return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil
}
@@ -2879,11 +3158,16 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// BRK [$imm16] SVC $imm16
// DMB|DSB|ISB $imm4 DC <op>, Rn
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op>
// PRFM (Rn), $imm|<op> RPRFM (Rn), Rm, <op|$imm6>
// SYS $imm[, Rn] SYSL $imm, Rd
// TLBI <op>[, Rn] SB, PACIASP, PACIBSP
//
// The system registers, TLBI and DC aliases and the range-prefetch operations
// come from the toolchain's own data tables in arm64_sysregs.go.
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "PACIASP": 0xd503233f, "PACIBSP": 0xd503237f,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
@@ -2919,6 +3203,12 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil
case "SB":
// Speculation barrier: DSB with a fixed barrier domain.
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(0xd50330ff), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
@@ -2957,7 +3247,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn")
}
base, ok := a64DCOps[operandRegName(ops[0])]
inst, ok := a64DCOps2[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0]))
}
@@ -2965,51 +3255,115 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 {
return nil, fmt.Errorf("DC: invalid register operand")
}
return a64wordLE(base | uint32(rn)&31), nil
w := 0xd5080000 | inst.op1<<16 | 7<<12 | inst.cm<<8 | inst.op2<<5
return a64wordLE(w | uint32(rn)&31), nil
case "TLBI":
// The register operand is optional: TLBI VMALLE1IS alone means ZR.
if len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("TLBI expects <op>[, Rn]")
}
inst, ok := a64TLBIOps[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("TLBI: unknown operation %q", operandRegName(ops[0]))
}
rt := 31
if len(ops) == 2 {
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
return nil, fmt.Errorf("TLBI: invalid register operand")
}
}
w := 0xd5080000 | inst.op1<<16 | 8<<12 | inst.cm<<8 | inst.op2<<5
return a64wordLE(w | uint32(rt)&31), nil
case "SYS", "SYSL":
// SYS $imm[, Rn] / SYSL $imm, Rd: the immediate packs
// op1<<16 | CRn<<12 | CRm<<8 | op2<<5, the register defaults to ZR.
if len(ops) != 1 && len(ops) != 2 {
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
}
if len(ops) == 1 && mnem == "SYSL" {
return nil, fmt.Errorf("SYSL expects $immediate, Rd")
}
if !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
}
imm := arm64Imm64(ops[0])
if imm < 0 || imm&^0x7FFE0 != 0 {
return nil, fmt.Errorf("%s: illegal SYS argument %d", mnem, imm)
}
rt := 31
if len(ops) == 2 {
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
return nil, fmt.Errorf("%s: invalid register operand", mnem)
}
}
base := uint32(0xd5080000)
if mnem == "SYSL" {
base = 0xd5280000
}
return a64wordLE(base | uint32(imm) | uint32(rt)&31), nil
case "MRS":
if len(ops) != 2 {
return nil, fmt.Errorf("MRS expects <sysreg>, Rd")
}
base, ok := a64MRSOps[operandRegName(ops[0])]
reg, ok := a64SysRegs[operandRegName(ops[0])]
if !ok {
return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0]))
}
if !reg.read {
return nil, fmt.Errorf("MRS: system register is not readable: %q", operandRegName(ops[0]))
}
rd := arm64RegNum(operandRegName(ops[1]))
if rd < 0 {
return nil, fmt.Errorf("MRS: invalid register operand")
}
return a64wordLE(base | uint32(rd)&31), nil
return a64wordLE(0xd5300000 | reg.v | uint32(rd)&31), nil
case "MSR":
if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if isImmOperand(ops[0]) {
v := arm64Imm64(ops[0])
// The PSTATE fields keep their dedicated immediate form.
if base, ok := a64MSROps[operandRegName(ops[1])]; ok {
if v < 0 || v > 15 {
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
}
return a64wordLE(base | uint32(v)<<8 | 31), nil
}
// A $0 against a full system register writes it from ZR, exactly
// the way the toolchain preprocesses the constant away; any other
// immediate is the PSTATE-form error.
if v != 0 {
return nil, fmt.Errorf("MSR: illegal PSTATE field for immediate move: %q", operandRegName(ops[1]))
}
reg, ok := a64SysRegs[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
if !reg.write {
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
}
return a64wordLE(base | uint32(rs)&31), nil
return a64wordLE(0xd5100000 | reg.v | 31), nil
}
base, ok := a64MSROps[operandRegName(ops[1])]
// Register form: MSR Rn, <sysreg>.
reg, ok := a64SysRegs[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
v := arm64Imm64(ops[0])
if v < 0 || v > 15 {
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
if !reg.write {
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
}
return a64wordLE(base | uint32(v)<<8 | 31), nil
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(0xd5100000 | reg.v | uint32(rs)&31), nil
case "PRFM":
if len(ops) != 2 {
return nil, fmt.Errorf("PRFM expects (Rn), $immediate|<op>")
}
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
if rn < 0 || off != 0 {
if rn < 0 || off < 0 || off%8 != 0 || off/8 >= 4096 {
return nil, fmt.Errorf("PRFM: invalid memory operand")
}
var prfop int64
@@ -3025,7 +3379,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
prfop = int64(p)
}
return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil
return a64wordLE(0xf9800000 | uint32(off/8)<<10 | uint32(rn)<<5 | uint32(prfop)), nil
case "RPRFM":
// RPRFM (Rn), Rm, <op|$imm6>: the 6-bit operation scatters across
// bits 15, 13, 12 and 2:0 (asm7.go case 110).
if len(ops) != 3 {
return nil, fmt.Errorf("RPRFM expects (Rn), Rm, <op|$immediate>")
}
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
if rn < 0 || off != 0 {
return nil, fmt.Errorf("RPRFM: invalid memory operand")
}
rm := arm64RegNum(operandRegName(ops[1]))
if rm < 0 {
return nil, fmt.Errorf("RPRFM: invalid register operand")
}
var op uint64
if isImmOperand(ops[2]) {
op = uint64(arm64Imm64(ops[2]))
if op > 63 {
return nil, fmt.Errorf("RPRFM: range prefetch immediate %d out of range (0..63)", op)
}
} else {
v, ok := a64RPRFOps[operandRegName(ops[2])]
if !ok {
return nil, fmt.Errorf("RPRFM: unknown range prefetch operation %q", operandRegName(ops[2]))
}
op = uint64(v)
}
scatter := (op&(1<<5))<<10 | (op&(1<<4))<<9 | (op&(1<<3))<<9 | op&7
return a64wordLE(0xf8a04818 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(scatter)), nil
}
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
}
@@ -3448,7 +3831,7 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("invalid destination register in VTBL")
}
for i, t := range ts {
if t.hasIdx || t.reg != ts[0].reg+i {
if t.hasIdx || (ts[0].reg+i)&31 != t.reg {
return nil, fmt.Errorf("VTBL table registers must be consecutive")
}
}
@@ -3605,15 +3988,17 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
// encodeARM64VLDST encodes the SIMD structure loads and stores:
//
// VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn)
// VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn)
// VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane)
// VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3]
// VLD1|2|3|4 (Rn), [Vt.arr, ...] VST1|2|3|4 [Vt.arr, ...], (Rn)
// VLD1|2|3|4.P off(Rn), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], off(Rn)
// VLD1|2|3|4.P (Rn)(Rm), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], (Rn)(Rm)
// VLD1|2|3|4R (Rn), [Vt.arr, ...] (replicating loads)
// VLD1 off(Rn), Vt.T[i] VST1 Vt.T[i], off(Rn) (one lane)
//
// The post-index forms set the post bit and Rm = 11111. A spelled offset
// rides along (the encoding ignores it; the toolchain only checks that it
// matches the access size), and a multi-register post-index list takes no
// offset at all, the increment following from the list.
// The post-index forms set the post bit and carry Rm: 11111 for an immediate
// increment, the spelled register for (Rn)(Rm). A register list may wrap
// around V31: the toolchain checks only (first+i) mod 32. A spelled offset
// rides along on the one-lane forms (the toolchain only checks that it
// matches the access size).
func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) {
load := strings.HasPrefix(mnem, "VLD")
@@ -3651,24 +4036,31 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
if off != 0 && post == 0 {
return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off)
}
// VLD1R loads one register and replicates; VLD4R loads four.
if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") {
want := 1
base := uint32(0x0d40c000)
if strings.HasPrefix(mnem, "VLD4R") {
want, base = 4, 0x0d60e000
// The post-index increment: 11111 for an immediate offset, else the
// spelled (Rn)(Rm) register.
rm := 31
if post != 0 {
if idx := ops[memIdx].Addr.Index; idx != "" {
if rm = arm64RegNum(idx); rm < 0 {
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
}
}
if len(vs) != want {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want)
}
// The replicating loads: VLD1R through VLD4R load one register and
// replicate it across the whole list.
if base := strings.TrimSuffix(mnem, ".P"); load && strings.HasSuffix(base, "R") && len(base) == 5 {
n := int(base[3] - '0')
if len(vs) != n {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
}
size, q, ok := a64ArrSizeQ(vs[0].arr)
if !ok {
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
}
w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
w := a64VLDNReplicate[n] | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
if post != 0 {
w |= 1<<23 | 0x1f<<16
w |= 1<<23 | uint32(rm)<<16
}
return a64wordLE(w), nil
}
@@ -3677,7 +4069,7 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
return nil, fmt.Errorf("%s expects a list of one to four registers", mnem)
}
for i, v := range vs {
if v.hasIdx || v.reg != vs[0].reg+i {
if v.hasIdx || (vs[0].reg+i)&31 != v.reg {
return nil, fmt.Errorf("%s: register list must be consecutive", mnem)
}
_, _, okArr := a64ArrSizeQ(v.arr)
@@ -3689,21 +4081,36 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
if !ok {
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
}
base := a64VLD1Base[len(vs)]
n := len(vs)
base := a64VLD1Base[n]
if !load {
base = a64VST1Base[len(vs)]
base = a64VST1Base[n]
}
// VLD2/VLD3/VLD4 and VST2/VST3/VST4 name the register count in the
// mnemonic and carry their own opcode fields. The count digit sits at
// index 3 of the mnemonic (VLD2, VST3.P, ...), before any .P suffix.
if stem := strings.TrimSuffix(mnem, ".P"); len(stem) >= 4 && stem[3] >= '2' && stem[3] <= '4' {
n := int(stem[3] - '0')
if n != len(vs) {
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
}
if load {
base = a64VLDNBase[n]
} else {
base = a64VSTNBase[n]
}
}
postBits := uint32(0)
if post != 0 {
postBits = 0x9f0000
postBits = 1<<23 | uint32(rm)<<16
}
return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil
}
// encodeARM64VLDSTLane encodes the one-lane structure forms:
// VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and
// VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the
// toolchain's table and is rejected.
// VLD1 off(Rn), Vt.T[i] and VST1 Vt.T[i], off(Rn); the post-index spellings
// add the post bit and Rm: 11111 for an immediate increment, the spelled
// register for (Rn)(Rm).
func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) {
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem)
@@ -3713,14 +4120,19 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
if rn < 0 {
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
}
if !load && post == 0 {
return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem)
rm := 31
if post != 0 {
if idx := ops[memIdx].Addr.Index; idx != "" {
if rm = arm64RegNum(idx); rm < 0 {
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
}
}
}
w := uint32(0x0d400000)
switch strings.ToUpper(v.arr) {
case "B":
// Index at bits 12:10 (the size field doubles as the low index bits).
w |= uint32(v.idx) << 10
// Index<3> rides bit 30, index<2:0> the size field at bits 12:10.
w |= uint32(v.idx&7)<<10 | uint32(v.idx>>3&1)<<30
case "H":
// Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11.
w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30
@@ -3734,12 +4146,12 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr)
}
// The base carries bit 22 (L) set; a store clears it. The post-index
// forms add bit 23 and Rm = 11111.
// forms add bit 23 and Rm.
if !load {
w &^= 1 << 22
}
if post != 0 {
w |= 1<<23 | 0x1f<<16
w |= 1<<23 | uint32(rm)<<16
}
return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil
}