feat(asm): encode the arm64 system registers and structure loads
This commit is contained in:
+477
-65
@@ -458,10 +458,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64Excl(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// LSE atomics (LDADD, CAS, SWP).
|
||||
// LSE atomics (LDADD, CAS, SWP) and the compare-and-swap pair.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
|
||||
return encodeARM64LSEAtom(mnem, enc.op, ops)
|
||||
}
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCASP {
|
||||
return encodeARM64CASP(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
|
||||
@@ -555,9 +558,12 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
||||
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
|
||||
// only in the zero-immediate form (a64SimdVZero), so they route here with
|
||||
// an empty register-form spec.
|
||||
// an empty register-form spec. VSQSHL/VUQSHL keep their shift-by-
|
||||
// immediate route when the first operand is an immediate.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
|
||||
return encodeARM64SimdV(mnem, spec, ops)
|
||||
if !arm64SimdShiftImmRoute(mnem, ops) {
|
||||
return encodeARM64SimdV(mnem, spec, ops)
|
||||
}
|
||||
}
|
||||
|
||||
// Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV).
|
||||
@@ -565,6 +571,13 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64SimdV2(mnem, spec, ops)
|
||||
}
|
||||
|
||||
// Narrow/long/wide SIMD families whose size and Q bits read off one
|
||||
// designated operand (VXTN, VSXTL, VUADDW, VUMULL, VSHRN, VSSHLL, VFCVTN
|
||||
// and friends).
|
||||
if spec, ok := a64SimdNLTable[mnem]; ok {
|
||||
return encodeARM64SimdNL(mnem, spec, ops)
|
||||
}
|
||||
|
||||
// SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR,
|
||||
// VEXT).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 {
|
||||
@@ -587,6 +600,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64ShiftImm(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// SIMD move immediate: VMOVI $imm8, Vd.B8/B16.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVMoviImm {
|
||||
return encodeARM64MoviImm(ops)
|
||||
}
|
||||
|
||||
// VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit {
|
||||
return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits)
|
||||
@@ -2568,6 +2586,263 @@ func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
||||
}
|
||||
|
||||
// encodeARM64CASP encodes the compare-and-swap pair: CASP (Rs, Rs+1), (Rn),
|
||||
// (Rt, Rt+1). Both pairs must start on an even register and be contiguous;
|
||||
// the second register of each pair rides no encoding field.
|
||||
func encodeARM64CASP(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects (Rs, Rs+1), (Rn), (Rt, Rt+1), got %d operands", mnem, len(ops))
|
||||
}
|
||||
rs, rs1, ok := arm64PairOf(ops[0])
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s expects a source register pair (Rs, Rs+1)", mnem)
|
||||
}
|
||||
rn, err := arm64ExclMem(mnem, ops[1])
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
rt, rt1, ok := arm64PairOf(ops[2])
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s expects a destination register pair (Rt, Rt+1)", mnem)
|
||||
}
|
||||
if rs&1 != 0 {
|
||||
return nil, fmt.Errorf("%s: source register pair must start from an even register", mnem)
|
||||
}
|
||||
if rt&1 != 0 {
|
||||
return nil, fmt.Errorf("%s: destination register pair must start from an even register", mnem)
|
||||
}
|
||||
if rs != rs1-1 {
|
||||
return nil, fmt.Errorf("%s: source register pair must be contiguous", mnem)
|
||||
}
|
||||
if rt != rt1-1 {
|
||||
return nil, fmt.Errorf("%s: destination register pair must be contiguous", mnem)
|
||||
}
|
||||
if rt == 31 {
|
||||
return nil, fmt.Errorf("%s: illegal destination register", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
||||
}
|
||||
|
||||
// encodeARM64MoviImm encodes VMOVI $imm8, Vd.B8/B16: the modified-immediate
|
||||
// form of the SIMD move (asm7.go case 86). Only the byte arrangements exist
|
||||
// and the immediate is one unsigned byte.
|
||||
func encodeARM64MoviImm(ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("VMOVI expects $immediate, Vd.<T>")
|
||||
}
|
||||
vd, ok := arm64VecOf(ops[1])
|
||||
if !ok || vd.hasIdx || (vd.arr != "B8" && vd.arr != "B16") {
|
||||
return nil, fmt.Errorf("VMOVI: destination arrangement must be B8 or B16")
|
||||
}
|
||||
imm := arm64Imm64(ops[0])
|
||||
if imm < 0 || imm > 255 {
|
||||
return nil, fmt.Errorf("VMOVI: immediate constant %d out of range (0..255)", imm)
|
||||
}
|
||||
q := uint32(0)
|
||||
if vd.arr == "B16" {
|
||||
q = 1 << 30
|
||||
}
|
||||
w := 0x0f00e400 | q | uint32(imm>>5&7)<<16 | uint32(imm&0x1f)<<5 | uint32(vd.reg)
|
||||
return a64wordLE(w), nil
|
||||
}
|
||||
|
||||
// arm64SimdShiftImmRoute reports whether a mnemonic carries both a shift-by-
|
||||
// immediate and a register form and the operands spell the immediate one: the
|
||||
// dedicated shift route keeps them.
|
||||
func arm64SimdShiftImmRoute(mnem string, ops []*ast.Operand) bool {
|
||||
if mnem != "VSQSHL" && mnem != "VUQSHL" {
|
||||
return false
|
||||
}
|
||||
return len(ops) > 0 && isImmOperand(ops[0])
|
||||
}
|
||||
|
||||
// arm64SimdNLArr describes one arrangement for the narrow/long/wide families:
|
||||
// the element width in bytes and whether the spelling names the 128-bit form.
|
||||
func arm64SimdNLArr(arr string) (esize int, wide bool, ok bool) {
|
||||
switch arr {
|
||||
case "B8", "B16":
|
||||
return 1, arr == "B16", true
|
||||
case "H4", "H8":
|
||||
return 2, arr == "H8", true
|
||||
case "S2", "S4":
|
||||
return 4, arr == "S4", true
|
||||
case "D1", "D2":
|
||||
return 8, arr == "D2", true
|
||||
}
|
||||
return 0, false, false
|
||||
}
|
||||
|
||||
// arm64SimdLongPair validates a long pairing (source narrow, destination
|
||||
// wide): the destination element is twice the source's, the destination is
|
||||
// always spelled the wide way (H8/S4/D2) and the source carries the 128-bit
|
||||
// flag exactly for the .2 spellings.
|
||||
func arm64SimdLongPair(mnem, src, dst string, two bool) error {
|
||||
se, sw, ok1 := arm64SimdNLArr(src)
|
||||
de, dw, ok2 := arm64SimdNLArr(dst)
|
||||
if !ok1 || !ok2 || de != 2*se {
|
||||
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
|
||||
}
|
||||
if !dw || sw != two {
|
||||
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// arm64SimdNarrowPair validates a narrow pairing (source wide, destination
|
||||
// narrow): the mirror image of arm64SimdLongPair.
|
||||
func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
|
||||
se, sw, ok1 := arm64SimdNLArr(src)
|
||||
de, dw, ok2 := arm64SimdNLArr(dst)
|
||||
if !ok1 || !ok2 || se != 2*de {
|
||||
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
|
||||
}
|
||||
if !sw || dw != two {
|
||||
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
|
||||
// instruction contributes: the driving arrangement's size and Q bits, or for
|
||||
// the FCVT family only the Q bit, whose size field is fixed in the base.
|
||||
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
|
||||
if spec.qonly {
|
||||
if two {
|
||||
return 1 << 30
|
||||
}
|
||||
return 0
|
||||
}
|
||||
return a64ArrBits[a64ArrIndex(drive)]
|
||||
}
|
||||
|
||||
// encodeARM64SimdNL encodes the narrow/long/wide SIMD families
|
||||
// (a64SimdNLTable): XTN and FCVTN narrow a wide source, SXTL and FCVTL
|
||||
// lengthen, the MULL/MLAL/MLSL group multiplies long, UADDW widens, and the
|
||||
// SSHLL/USHLL and SHRN shifts carry their immediate in the immh:immb field.
|
||||
// The size and Q bits read off the designated driving operand, and the .2
|
||||
// spellings force the 128-bit side through their own arrangement.
|
||||
func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]byte, error) {
|
||||
two := strings.HasSuffix(mnem, "2")
|
||||
|
||||
// vecAt parses operand i as a vector register with an arrangement.
|
||||
vecAt := func(i int) (a64Vec, bool) {
|
||||
if i >= len(ops) {
|
||||
return a64Vec{}, false
|
||||
}
|
||||
v, ok := arm64VecOf(ops[i])
|
||||
if !ok || v.hasIdx {
|
||||
return a64Vec{}, false
|
||||
}
|
||||
return v, true
|
||||
}
|
||||
|
||||
switch spec.form {
|
||||
case a64NLTwoNarrow, a64NLTwoLong:
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
vn, ok1 := vecAt(0)
|
||||
vd, ok2 := vecAt(1)
|
||||
if !ok1 || !ok2 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
var drive string
|
||||
var pairErr error
|
||||
if spec.form == a64NLTwoNarrow {
|
||||
// XTN/FCVTN: wide source into a narrow destination; the
|
||||
// arrangement bits follow the destination.
|
||||
drive, pairErr = vd.arr, arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two)
|
||||
} else {
|
||||
// SXTL/UXTL/FCVTL: narrow source into a wide destination; the
|
||||
// arrangement bits follow the source.
|
||||
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
|
||||
}
|
||||
if pairErr != nil {
|
||||
return nil, pairErr
|
||||
}
|
||||
arrBits := arm64SimdNLArrBits(spec, drive, two)
|
||||
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
case a64NLThreeLongMul, a64NLThreeWide:
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
vm, ok1 := vecAt(0)
|
||||
vn, ok2 := vecAt(1)
|
||||
vd, ok3 := vecAt(2)
|
||||
if !ok1 || !ok2 || !ok3 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
var drive string
|
||||
var pairErr error
|
||||
if spec.form == a64NLThreeWide {
|
||||
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
|
||||
// the arrangement bits follow the wide side.
|
||||
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
|
||||
} else {
|
||||
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
|
||||
// wide one; the arrangement bits follow the narrow source.
|
||||
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
|
||||
}
|
||||
if pairErr != nil {
|
||||
return nil, pairErr
|
||||
}
|
||||
if spec.form == a64NLThreeWide && vd.arr != vn.arr {
|
||||
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vn.arr, vd.arr)
|
||||
}
|
||||
if spec.form == a64NLThreeLongMul && vm.arr != vn.arr {
|
||||
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
|
||||
}
|
||||
arrBits := arm64SimdNLArrBits(spec, drive, two)
|
||||
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
case a64NLThreeLongShift, a64NLThreeNarrowShift:
|
||||
if len(ops) != 3 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects ($shift, Vn.<T>, Vd.<T>)", mnem)
|
||||
}
|
||||
sh := arm64Imm64(ops[0])
|
||||
vn, ok1 := vecAt(1)
|
||||
vd, ok2 := vecAt(2)
|
||||
if !ok1 || !ok2 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
if spec.form == a64NLThreeLongShift {
|
||||
// SSHLL/USHLL: the narrow source drives the immediate's size
|
||||
// (immh:immb = esize + shift), so the arrangement bits carry
|
||||
// the Q bit alone: the size field belongs to immh, and ORing
|
||||
// the source's size bits into it would collide with immb.
|
||||
if err := arm64SimdLongPair(mnem, vn.arr, vd.arr, two); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
se, _, _ := arm64SimdNLArr(vn.arr)
|
||||
esize := se * 8
|
||||
if sh < 0 || sh >= int64(esize) {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
||||
}
|
||||
var qBit uint32
|
||||
if two {
|
||||
qBit = 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | uint32(esize+int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
// SHRN: the narrow destination drives the immediate's size
|
||||
// (immh:immb = esize - shift over the wide source element), so the
|
||||
// arrangement bits carry the Q bit alone, exactly as above.
|
||||
if err := arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
se, _, _ := arm64SimdNLArr(vn.arr)
|
||||
esize := se * 8
|
||||
if sh < 1 || sh >= int64(esize) {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize-1)
|
||||
}
|
||||
var qBit uint32
|
||||
if two {
|
||||
qBit = 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | uint32(esize-int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
||||
}
|
||||
|
||||
// encodeARM64DP1 encodes a data-processing (1 source) instruction:
|
||||
// RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd.
|
||||
func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
@@ -2584,7 +2859,8 @@ func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, err
|
||||
|
||||
// encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the
|
||||
// target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31
|
||||
// selecting the page form.
|
||||
// selecting the page form. An n(PC) operand resolves to the instruction's
|
||||
// own address: the toolchain rewrites it away and encodes displacement 0.
|
||||
func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
@@ -2593,14 +2869,17 @@ func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offset
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
target := resolve(arm64Label(ops[0]))
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q", target)
|
||||
}
|
||||
rel := int64(targetOff - pc)
|
||||
if rel < -(1<<20) || rel >= 1<<20 {
|
||||
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
|
||||
var rel int64
|
||||
if _, pcRel := arm64PCRelOffset(ops[0]); !pcRel {
|
||||
target := resolve(arm64Label(ops[0]))
|
||||
targetOff, ok := offsets[target]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("undefined label %q", target)
|
||||
}
|
||||
rel = int64(targetOff - pc)
|
||||
if rel < -(1<<20) || rel >= 1<<20 {
|
||||
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
|
||||
}
|
||||
}
|
||||
return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil
|
||||
}
|
||||
@@ -2879,11 +3158,16 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
// BRK [$imm16] SVC $imm16
|
||||
// DMB|DSB|ISB $imm4 DC <op>, Rn
|
||||
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
||||
// PRFM (Rn), $imm|<op>
|
||||
// PRFM (Rn), $imm|<op> RPRFM (Rn), Rm, <op|$imm6>
|
||||
// SYS $imm[, Rn] SYSL $imm, Rd
|
||||
// TLBI <op>[, Rn] SB, PACIASP, PACIBSP
|
||||
//
|
||||
// The system registers, TLBI and DC aliases and the range-prefetch operations
|
||||
// come from the toolchain's own data tables in arm64_sysregs.go.
|
||||
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
// Operand-less returns and pointer-authentication hints.
|
||||
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
|
||||
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
|
||||
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "PACIASP": 0xd503233f, "PACIBSP": 0xd503237f,
|
||||
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
|
||||
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
|
||||
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
|
||||
@@ -2919,6 +3203,12 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
}
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<8), nil
|
||||
case "SB":
|
||||
// Speculation barrier: DSB with a fixed barrier domain.
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("%s expects no operand", mnem)
|
||||
}
|
||||
return a64wordLE(0xd50330ff), nil
|
||||
case "HINT":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
@@ -2957,7 +3247,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("DC expects <op>, Rn")
|
||||
}
|
||||
base, ok := a64DCOps[operandRegName(ops[0])]
|
||||
inst, ok := a64DCOps2[operandRegName(ops[0])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0]))
|
||||
}
|
||||
@@ -2965,51 +3255,115 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if rn < 0 {
|
||||
return nil, fmt.Errorf("DC: invalid register operand")
|
||||
}
|
||||
return a64wordLE(base | uint32(rn)&31), nil
|
||||
w := 0xd5080000 | inst.op1<<16 | 7<<12 | inst.cm<<8 | inst.op2<<5
|
||||
return a64wordLE(w | uint32(rn)&31), nil
|
||||
case "TLBI":
|
||||
// The register operand is optional: TLBI VMALLE1IS alone means ZR.
|
||||
if len(ops) != 1 && len(ops) != 2 {
|
||||
return nil, fmt.Errorf("TLBI expects <op>[, Rn]")
|
||||
}
|
||||
inst, ok := a64TLBIOps[operandRegName(ops[0])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("TLBI: unknown operation %q", operandRegName(ops[0]))
|
||||
}
|
||||
rt := 31
|
||||
if len(ops) == 2 {
|
||||
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
|
||||
return nil, fmt.Errorf("TLBI: invalid register operand")
|
||||
}
|
||||
}
|
||||
w := 0xd5080000 | inst.op1<<16 | 8<<12 | inst.cm<<8 | inst.op2<<5
|
||||
return a64wordLE(w | uint32(rt)&31), nil
|
||||
case "SYS", "SYSL":
|
||||
// SYS $imm[, Rn] / SYSL $imm, Rd: the immediate packs
|
||||
// op1<<16 | CRn<<12 | CRm<<8 | op2<<5, the register defaults to ZR.
|
||||
if len(ops) != 1 && len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
|
||||
}
|
||||
if len(ops) == 1 && mnem == "SYSL" {
|
||||
return nil, fmt.Errorf("SYSL expects $immediate, Rd")
|
||||
}
|
||||
if !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
|
||||
}
|
||||
imm := arm64Imm64(ops[0])
|
||||
if imm < 0 || imm&^0x7FFE0 != 0 {
|
||||
return nil, fmt.Errorf("%s: illegal SYS argument %d", mnem, imm)
|
||||
}
|
||||
rt := 31
|
||||
if len(ops) == 2 {
|
||||
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid register operand", mnem)
|
||||
}
|
||||
}
|
||||
base := uint32(0xd5080000)
|
||||
if mnem == "SYSL" {
|
||||
base = 0xd5280000
|
||||
}
|
||||
return a64wordLE(base | uint32(imm) | uint32(rt)&31), nil
|
||||
case "MRS":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("MRS expects <sysreg>, Rd")
|
||||
}
|
||||
base, ok := a64MRSOps[operandRegName(ops[0])]
|
||||
reg, ok := a64SysRegs[operandRegName(ops[0])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0]))
|
||||
}
|
||||
if !reg.read {
|
||||
return nil, fmt.Errorf("MRS: system register is not readable: %q", operandRegName(ops[0]))
|
||||
}
|
||||
rd := arm64RegNum(operandRegName(ops[1]))
|
||||
if rd < 0 {
|
||||
return nil, fmt.Errorf("MRS: invalid register operand")
|
||||
}
|
||||
return a64wordLE(base | uint32(rd)&31), nil
|
||||
return a64wordLE(0xd5300000 | reg.v | uint32(rd)&31), nil
|
||||
case "MSR":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
|
||||
}
|
||||
if !isImmOperand(ops[0]) {
|
||||
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
|
||||
base, ok := a64MSRRegOps[operandRegName(ops[1])]
|
||||
if isImmOperand(ops[0]) {
|
||||
v := arm64Imm64(ops[0])
|
||||
// The PSTATE fields keep their dedicated immediate form.
|
||||
if base, ok := a64MSROps[operandRegName(ops[1])]; ok {
|
||||
if v < 0 || v > 15 {
|
||||
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
|
||||
}
|
||||
return a64wordLE(base | uint32(v)<<8 | 31), nil
|
||||
}
|
||||
// A $0 against a full system register writes it from ZR, exactly
|
||||
// the way the toolchain preprocesses the constant away; any other
|
||||
// immediate is the PSTATE-form error.
|
||||
if v != 0 {
|
||||
return nil, fmt.Errorf("MSR: illegal PSTATE field for immediate move: %q", operandRegName(ops[1]))
|
||||
}
|
||||
reg, ok := a64SysRegs[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("MSR: invalid source register")
|
||||
if !reg.write {
|
||||
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
|
||||
}
|
||||
return a64wordLE(base | uint32(rs)&31), nil
|
||||
return a64wordLE(0xd5100000 | reg.v | 31), nil
|
||||
}
|
||||
base, ok := a64MSROps[operandRegName(ops[1])]
|
||||
// Register form: MSR Rn, <sysreg>.
|
||||
reg, ok := a64SysRegs[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 15 {
|
||||
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
|
||||
if !reg.write {
|
||||
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
|
||||
}
|
||||
return a64wordLE(base | uint32(v)<<8 | 31), nil
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("MSR: invalid source register")
|
||||
}
|
||||
return a64wordLE(0xd5100000 | reg.v | uint32(rs)&31), nil
|
||||
case "PRFM":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("PRFM expects (Rn), $immediate|<op>")
|
||||
}
|
||||
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
|
||||
if rn < 0 || off != 0 {
|
||||
if rn < 0 || off < 0 || off%8 != 0 || off/8 >= 4096 {
|
||||
return nil, fmt.Errorf("PRFM: invalid memory operand")
|
||||
}
|
||||
var prfop int64
|
||||
@@ -3025,7 +3379,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
}
|
||||
prfop = int64(p)
|
||||
}
|
||||
return a64wordLE(0xf9800000 | uint32(rn)<<5 | uint32(prfop)), nil
|
||||
return a64wordLE(0xf9800000 | uint32(off/8)<<10 | uint32(rn)<<5 | uint32(prfop)), nil
|
||||
case "RPRFM":
|
||||
// RPRFM (Rn), Rm, <op|$imm6>: the 6-bit operation scatters across
|
||||
// bits 15, 13, 12 and 2:0 (asm7.go case 110).
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("RPRFM expects (Rn), Rm, <op|$immediate>")
|
||||
}
|
||||
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
|
||||
if rn < 0 || off != 0 {
|
||||
return nil, fmt.Errorf("RPRFM: invalid memory operand")
|
||||
}
|
||||
rm := arm64RegNum(operandRegName(ops[1]))
|
||||
if rm < 0 {
|
||||
return nil, fmt.Errorf("RPRFM: invalid register operand")
|
||||
}
|
||||
var op uint64
|
||||
if isImmOperand(ops[2]) {
|
||||
op = uint64(arm64Imm64(ops[2]))
|
||||
if op > 63 {
|
||||
return nil, fmt.Errorf("RPRFM: range prefetch immediate %d out of range (0..63)", op)
|
||||
}
|
||||
} else {
|
||||
v, ok := a64RPRFOps[operandRegName(ops[2])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("RPRFM: unknown range prefetch operation %q", operandRegName(ops[2]))
|
||||
}
|
||||
op = uint64(v)
|
||||
}
|
||||
scatter := (op&(1<<5))<<10 | (op&(1<<4))<<9 | (op&(1<<3))<<9 | op&7
|
||||
return a64wordLE(0xf8a04818 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(scatter)), nil
|
||||
}
|
||||
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
||||
}
|
||||
@@ -3448,7 +3831,7 @@ func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("invalid destination register in VTBL")
|
||||
}
|
||||
for i, t := range ts {
|
||||
if t.hasIdx || t.reg != ts[0].reg+i {
|
||||
if t.hasIdx || (ts[0].reg+i)&31 != t.reg {
|
||||
return nil, fmt.Errorf("VTBL table registers must be consecutive")
|
||||
}
|
||||
}
|
||||
@@ -3605,15 +3988,17 @@ func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
|
||||
// encodeARM64VLDST encodes the SIMD structure loads and stores:
|
||||
//
|
||||
// VLD1 (Rn), [Vt.arr, ...] VST1 [Vt.arr, ...], (Rn)
|
||||
// VLD1.P off(Rn), [Vt.arr, ...] VST1.P [Vt.arr, ...], off(Rn)
|
||||
// VLD1.P off(Rn), Vt.T[i] VST1.P Vt.T[i], off(Rn) (one lane)
|
||||
// VLD1R (Rn), [Vt.arr] VLD4R (Rn), [Vt.arr, Vt+1, Vt+2, Vt+3]
|
||||
// VLD1|2|3|4 (Rn), [Vt.arr, ...] VST1|2|3|4 [Vt.arr, ...], (Rn)
|
||||
// VLD1|2|3|4.P off(Rn), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], off(Rn)
|
||||
// VLD1|2|3|4.P (Rn)(Rm), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], (Rn)(Rm)
|
||||
// VLD1|2|3|4R (Rn), [Vt.arr, ...] (replicating loads)
|
||||
// VLD1 off(Rn), Vt.T[i] VST1 Vt.T[i], off(Rn) (one lane)
|
||||
//
|
||||
// The post-index forms set the post bit and Rm = 11111. A spelled offset
|
||||
// rides along (the encoding ignores it; the toolchain only checks that it
|
||||
// matches the access size), and a multi-register post-index list takes no
|
||||
// offset at all, the increment following from the list.
|
||||
// The post-index forms set the post bit and carry Rm: 11111 for an immediate
|
||||
// increment, the spelled register for (Rn)(Rm). A register list may wrap
|
||||
// around V31: the toolchain checks only (first+i) mod 32. A spelled offset
|
||||
// rides along on the one-lane forms (the toolchain only checks that it
|
||||
// matches the access size).
|
||||
func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
load := strings.HasPrefix(mnem, "VLD")
|
||||
|
||||
@@ -3651,24 +4036,31 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
|
||||
if off != 0 && post == 0 {
|
||||
return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off)
|
||||
}
|
||||
|
||||
// VLD1R loads one register and replicates; VLD4R loads four.
|
||||
if strings.HasPrefix(mnem, "VLD1R") || strings.HasPrefix(mnem, "VLD4R") {
|
||||
want := 1
|
||||
base := uint32(0x0d40c000)
|
||||
if strings.HasPrefix(mnem, "VLD4R") {
|
||||
want, base = 4, 0x0d60e000
|
||||
// The post-index increment: 11111 for an immediate offset, else the
|
||||
// spelled (Rn)(Rm) register.
|
||||
rm := 31
|
||||
if post != 0 {
|
||||
if idx := ops[memIdx].Addr.Index; idx != "" {
|
||||
if rm = arm64RegNum(idx); rm < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
|
||||
}
|
||||
}
|
||||
if len(vs) != want {
|
||||
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, want)
|
||||
}
|
||||
|
||||
// The replicating loads: VLD1R through VLD4R load one register and
|
||||
// replicate it across the whole list.
|
||||
if base := strings.TrimSuffix(mnem, ".P"); load && strings.HasSuffix(base, "R") && len(base) == 5 {
|
||||
n := int(base[3] - '0')
|
||||
if len(vs) != n {
|
||||
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
|
||||
}
|
||||
size, q, ok := a64ArrSizeQ(vs[0].arr)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
|
||||
}
|
||||
w := base | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
|
||||
w := a64VLDNReplicate[n] | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
|
||||
if post != 0 {
|
||||
w |= 1<<23 | 0x1f<<16
|
||||
w |= 1<<23 | uint32(rm)<<16
|
||||
}
|
||||
return a64wordLE(w), nil
|
||||
}
|
||||
@@ -3677,7 +4069,7 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
|
||||
return nil, fmt.Errorf("%s expects a list of one to four registers", mnem)
|
||||
}
|
||||
for i, v := range vs {
|
||||
if v.hasIdx || v.reg != vs[0].reg+i {
|
||||
if v.hasIdx || (vs[0].reg+i)&31 != v.reg {
|
||||
return nil, fmt.Errorf("%s: register list must be consecutive", mnem)
|
||||
}
|
||||
_, _, okArr := a64ArrSizeQ(v.arr)
|
||||
@@ -3689,21 +4081,36 @@ func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, err
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
|
||||
}
|
||||
base := a64VLD1Base[len(vs)]
|
||||
n := len(vs)
|
||||
base := a64VLD1Base[n]
|
||||
if !load {
|
||||
base = a64VST1Base[len(vs)]
|
||||
base = a64VST1Base[n]
|
||||
}
|
||||
// VLD2/VLD3/VLD4 and VST2/VST3/VST4 name the register count in the
|
||||
// mnemonic and carry their own opcode fields. The count digit sits at
|
||||
// index 3 of the mnemonic (VLD2, VST3.P, ...), before any .P suffix.
|
||||
if stem := strings.TrimSuffix(mnem, ".P"); len(stem) >= 4 && stem[3] >= '2' && stem[3] <= '4' {
|
||||
n := int(stem[3] - '0')
|
||||
if n != len(vs) {
|
||||
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
|
||||
}
|
||||
if load {
|
||||
base = a64VLDNBase[n]
|
||||
} else {
|
||||
base = a64VSTNBase[n]
|
||||
}
|
||||
}
|
||||
postBits := uint32(0)
|
||||
if post != 0 {
|
||||
postBits = 0x9f0000
|
||||
postBits = 1<<23 | uint32(rm)<<16
|
||||
}
|
||||
return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64VLDSTLane encodes the one-lane structure forms:
|
||||
// VLD1 off(Rn), Vt.T[i] (post-index adds the post bit and Rm=11111) and
|
||||
// VST1.P Vt.T[i], off(Rn); the plain VST1 lane form does not exist in the
|
||||
// toolchain's table and is rejected.
|
||||
// VLD1 off(Rn), Vt.T[i] and VST1 Vt.T[i], off(Rn); the post-index spellings
|
||||
// add the post bit and Rm: 11111 for an immediate increment, the spelled
|
||||
// register for (Rn)(Rm).
|
||||
func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) {
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem)
|
||||
@@ -3713,14 +4120,19 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
|
||||
if rn < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
|
||||
}
|
||||
if !load && post == 0 {
|
||||
return nil, fmt.Errorf("%s: the toolchain only spells a post-index single-lane store", mnem)
|
||||
rm := 31
|
||||
if post != 0 {
|
||||
if idx := ops[memIdx].Addr.Index; idx != "" {
|
||||
if rm = arm64RegNum(idx); rm < 0 {
|
||||
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
|
||||
}
|
||||
}
|
||||
}
|
||||
w := uint32(0x0d400000)
|
||||
switch strings.ToUpper(v.arr) {
|
||||
case "B":
|
||||
// Index at bits 12:10 (the size field doubles as the low index bits).
|
||||
w |= uint32(v.idx) << 10
|
||||
// Index<3> rides bit 30, index<2:0> the size field at bits 12:10.
|
||||
w |= uint32(v.idx&7)<<10 | uint32(v.idx>>3&1)<<30
|
||||
case "H":
|
||||
// Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11.
|
||||
w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30
|
||||
@@ -3734,12 +4146,12 @@ func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load boo
|
||||
return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr)
|
||||
}
|
||||
// The base carries bit 22 (L) set; a store clears it. The post-index
|
||||
// forms add bit 23 and Rm = 11111.
|
||||
// forms add bit 23 and Rm.
|
||||
if !load {
|
||||
w &^= 1 << 22
|
||||
}
|
||||
if post != 0 {
|
||||
w |= 1<<23 | 0x1f<<16
|
||||
w |= 1<<23 | uint32(rm)<<16
|
||||
}
|
||||
return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user