6276 lines
219 KiB
Go
6276 lines
219 KiB
Go
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
|
// SPDX-License-Identifier: BSD-3-Clause
|
|
|
|
package asm
|
|
|
|
import (
|
|
"fmt"
|
|
"math"
|
|
"math/bits"
|
|
"slices"
|
|
"strconv"
|
|
"strings"
|
|
|
|
"sourcedock.dev/petrbalvin/gasm-sdk/arch"
|
|
"sourcedock.dev/petrbalvin/gasm-sdk/ast"
|
|
)
|
|
|
|
// assembleARM64 assembles an AArch64 (arm64) TEXT function body into machine
|
|
// code. Every instruction is 4 bytes; the MOV pseudo-instruction and the
|
|
// immediate-arithmetic forms expand to 2-4 instructions when the immediate
|
|
// does not fit, so the layout is computed in two passes (sizes, then encoding
|
|
// with resolved branch targets). The returned literals carry the read-only
|
|
// constants any VMOVS/VMOVD/VMOVQ load refers to; the file assembler lays
|
|
// them out in the data section.
|
|
//
|
|
// The emitted bytes match the Go toolchain's arm64 assembler, which is the
|
|
// ground-truth oracle: prologue/epilogue, FP/SP frame mapping, branch
|
|
// encodings and the MOV immediate expansions all follow cmd/internal/obj/
|
|
// arm64's asmout cases. One deliberate difference: the stack-growth guard
|
|
// (the morestack check in the prologue and the call back into the runtime in
|
|
// the epilogue) is not emitted, so the bytes match only for NOSPLIT functions
|
|
// or zero-frame leaves, where the toolchain emits no guard either.
|
|
func assembleARM64(t *ast.Text, tls map[string]bool) ([]byte, map[string]int, []Reloc, []LineEntry, []SpadjStep, []Arm64Literal, error) {
|
|
fi := arm64ComputeFrame(t)
|
|
fi.tls = tls
|
|
prologue := arm64Prologue(fi)
|
|
guardLen := arm64GuardLen(fi)
|
|
chain := arm64JumpChain(t)
|
|
resolve := func(name string) string {
|
|
if r, ok := chain[name]; ok {
|
|
return r
|
|
}
|
|
return name
|
|
}
|
|
|
|
var relocs []Reloc
|
|
var spadj []SpadjStep
|
|
lits := &arm64Literals{}
|
|
pool := &arm64Pool{}
|
|
|
|
// The prologue (3 instructions when a small frame, 4 for large)
|
|
// raises the SP delta by autosize. The guard prefix shifts its PC.
|
|
if fi.autosize != 0 {
|
|
spadj = append(spadj, SpadjStep{PC: guardLen + arm64PrologueSpadjPC(fi), Value: fi.autosize})
|
|
}
|
|
|
|
// Pass 1 plans the whole layout in one walk, the way the toolchain's
|
|
// span7 runs its own single linear pass: the label offsets come out of
|
|
// the same positions the encoder lays down, and the literal pool's
|
|
// references are harvested per statement with a probe encoding, so
|
|
// checkpool's flush condition is evaluated exactly as the toolchain's
|
|
// is and a reference that would leave the load-literal displacement
|
|
// bound drains the pool inside the body, at the point the toolchain
|
|
// would drain it.
|
|
offsets := map[string]int{}
|
|
layout := arm64PlanPool(t, fi, guardLen, len(prologue), pool, offsets, resolve)
|
|
pool.probe = false
|
|
|
|
// Pass 2: encode. The guard prefix precedes the prologue; its branches
|
|
// target the morestack block at the end of the function, whose position
|
|
// the plan has settled.
|
|
var out []byte
|
|
if fi.needSplit {
|
|
out = append(out, arm64GuardBytes(fi, layout.blockStart)...)
|
|
}
|
|
out = append(out, prologue...)
|
|
pc := guardLen + len(prologue)
|
|
preCount := len(relocs)
|
|
var lines []LineEntry
|
|
// The drained segments replay in the plan's own order: the segment open
|
|
// at a statement holds its references, and the flush event after it
|
|
// emits the guard word and the words the segment drained.
|
|
segIdx := 0
|
|
for i, stmt := range t.Body {
|
|
in, ok := stmt.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
flushAt := -1
|
|
if len(pool.segs) > 0 {
|
|
for segIdx < len(pool.segs)-1 && pool.segs[segIdx].after < i {
|
|
segIdx++
|
|
}
|
|
pool.active = segIdx
|
|
if pool.segs[segIdx].after == i {
|
|
flushAt = segIdx
|
|
}
|
|
}
|
|
switch strings.ToUpper(in.Mnemonic.Text) {
|
|
case "PCALIGN":
|
|
pad := arm64PCAlignPad(pc, in)
|
|
for j := 0; j < pad/4; j++ {
|
|
out = append(out, a64wordLE(a64NOP)...)
|
|
pc += 4
|
|
}
|
|
case "BYTE":
|
|
for _, op := range in.Operands {
|
|
out = append(out, byte(arm64Imm64(op)))
|
|
pc++
|
|
}
|
|
default:
|
|
code, err := encodeARM64Instr(in, pc, offsets, fi, &relocs, resolve, lits, pool, pool.wordsBase())
|
|
if err != nil {
|
|
return nil, nil, nil, nil, nil, nil, fmt.Errorf("%s: %w", in.Mnemonic.Text, err)
|
|
}
|
|
for j := preCount; j < len(relocs); j++ {
|
|
// Make the relocation offsets function-relative: each instruction
|
|
// records its reloc offset relative to its own start, and pc is
|
|
// that instruction's offset from the function start (prologue
|
|
// included). After shifts by the same amount.
|
|
relocs[j].Off += pc
|
|
relocs[j].After += pc
|
|
}
|
|
preCount = len(relocs)
|
|
lines = append(lines, LineEntry{Offset: pc, Line: in.Pos().Line})
|
|
// The RET's epilogue closes the frame: the SP delta returns to zero.
|
|
if strings.ToUpper(in.Mnemonic.Text) == "RET" && fi.autosize != 0 {
|
|
epi := arm64ReturnEpilogueLen(fi)
|
|
spadj = append(spadj, SpadjStep{PC: pc + epi, Value: 0})
|
|
}
|
|
out = append(out, code...)
|
|
pc += len(code)
|
|
}
|
|
if flushAt >= 0 {
|
|
seg := pool.segs[flushAt]
|
|
out = appendARM64PoolSeg(out, seg)
|
|
pc += seg.guard + seg.size
|
|
segIdx++
|
|
}
|
|
}
|
|
if fi.needSplit {
|
|
block, blReloc := arm64MoreStackBlock(pc)
|
|
out = append(out, block...)
|
|
relocs = append(relocs, blReloc)
|
|
pc += len(block)
|
|
}
|
|
// The plan's closing segment, the toolchain's end-of-function flush: the
|
|
// function's last Prog branches away (the morestack block's B when the
|
|
// function splits, the RET otherwise), so the words follow bare, and a
|
|
// body that would fall through gets the UNDEF word between.
|
|
if n := len(pool.segs); n > 0 && pool.segs[n-1].after == len(t.Body) {
|
|
seg := pool.segs[n-1]
|
|
out = appendARM64PoolSeg(out, seg)
|
|
pc += seg.guard + seg.size
|
|
}
|
|
if pc != layout.total {
|
|
return nil, nil, nil, nil, nil, nil, fmt.Errorf("internal: layout diverged from the literal-pool plan (%d bytes, planned %d)", pc, layout.total)
|
|
}
|
|
return out, offsets, relocs, lines, spadj, lits.list(), nil
|
|
}
|
|
|
|
// arm64PoolLayout carries what the encode pass needs from the plan: the
|
|
// whole function image's length, to hold the encoder to the plan, and the
|
|
// position the trailing morestack block lands at for a splitting function,
|
|
// which the guard prefix's conditional branches target.
|
|
type arm64PoolLayout struct {
|
|
total int
|
|
blockStart int
|
|
}
|
|
|
|
// a64MaxPCDisp is the toolchain's conservative bound on a PC-relative
|
|
// literal displacement (asm7.go's maxPCDisp): a load literal reaches ±1 MiB,
|
|
// and the flush points sit at half that, so the span-dependent branch
|
|
// enlargements of the later passes cannot push a reference out of reach.
|
|
const a64MaxPCDisp = 512 * 1024
|
|
|
|
// a64IsPCDisp ports the toolchain's ispcdisp.
|
|
func a64IsPCDisp(v int) bool {
|
|
return -a64MaxPCDisp < v && v < a64MaxPCDisp && v&3 == 0
|
|
}
|
|
|
|
// arm64EndsBlock reports whether a mnemonic hands control away
|
|
// unconditionally, the toolchain's flushpool exemption (AB, ARET and AERET):
|
|
// a pool drained after one needs no guard word, execution cannot fall into
|
|
// it.
|
|
func arm64EndsBlock(mnem string) bool {
|
|
switch mnem {
|
|
case "RET", "B", "JMP", "ERET":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// arm64PlanPool walks the body once, planning the layout and the literal
|
|
// pool's flush points together. The label offsets are the positions the
|
|
// encode pass lays down, flush bytes included; the pool references are
|
|
// harvested per statement by a probe encoding, whose band decisions hang on
|
|
// the operands alone and never on the positions, so the flush condition can
|
|
// run exactly as the toolchain's checkpool does:
|
|
//
|
|
// - the segment's accounting size has reached 0xffff0, or
|
|
// - the segment's far side has left the conservative displacement bound
|
|
// from the statement that would reference it, or
|
|
// - the statement is the function's last.
|
|
//
|
|
// A flush drains the open segment after the triggering statement: bare
|
|
// after a statement that branches away unconditionally, behind the UNDEF
|
|
// word at the function's end, behind a branch over the words inside the
|
|
// body. Every drained word carries the triggering statement's source line,
|
|
// the toolchain's own choice, so the pc-line tables see no deltas across
|
|
// the words.
|
|
func arm64PlanPool(t *ast.Text, fi arm64FrameInfo, guardLen, prologueLen int, pool *arm64Pool, offsets map[string]int, resolve func(string) string) arm64PoolLayout {
|
|
pool.probe = true
|
|
pos := guardLen + prologueLen
|
|
// The function's last real statement: END closes the body without
|
|
// becoming one, skipped the way a trailing label is, and this is the
|
|
// statement the toolchain's p.Link == nil check lands on.
|
|
last := -1
|
|
for i, v := range slices.Backward(t.Body) {
|
|
in, ok := v.(*ast.Instr)
|
|
if !ok || strings.ToUpper(in.Mnemonic.Text) == "END" {
|
|
continue
|
|
}
|
|
last = i
|
|
break
|
|
}
|
|
for i, stmt := range t.Body {
|
|
var in *ast.Instr
|
|
switch s := stmt.(type) {
|
|
case *ast.Label:
|
|
offsets[s.Name.Text] = pos
|
|
case *ast.Instr:
|
|
in = s
|
|
}
|
|
if in == nil {
|
|
continue
|
|
}
|
|
mnem := strings.ToUpper(in.Mnemonic.Text)
|
|
pc := pos
|
|
opened := pool.activeSeg() // the segment before the probe
|
|
switch mnem {
|
|
case "PCALIGN":
|
|
pos += arm64PCAlignPad(pc, in)
|
|
case "BYTE":
|
|
pos += len(in.Operands)
|
|
default:
|
|
// The probe: the encoder's own band decisions name the pool
|
|
// references into the segment open here. Its errors carry
|
|
// nothing: a forward branch cannot resolve yet, and the reach
|
|
// check would run against a base that is not final.
|
|
var probeRelocs []Reloc
|
|
_, _ = encodeARM64Instr(in, pc, offsets, fi, &probeRelocs, resolve, &arm64Literals{}, pool, pool.wordsBase())
|
|
pos += arm64InstrSize(in, fi, pc)
|
|
}
|
|
seg := pool.activeSeg()
|
|
if seg == nil {
|
|
continue
|
|
}
|
|
if opened == nil {
|
|
// The statement that opened the segment, the toolchain's
|
|
// pool.start: the first reference decides the distances the
|
|
// flush condition measures.
|
|
seg.start = pc
|
|
}
|
|
v := pc + 4 + seg.acct - seg.start + 8
|
|
end := !fi.needSplit && i == last
|
|
if seg.acct >= 0xffff0 || !a64IsPCDisp(v) || end {
|
|
seg.base = pos
|
|
seg.line = in.Pos().Line
|
|
seg.after = i
|
|
switch {
|
|
case arm64EndsBlock(mnem):
|
|
case end:
|
|
seg.guard = 4 // the UNDEF word, execution must not fall in
|
|
default:
|
|
seg.guard, seg.branch = 4, true // a branch over the words
|
|
}
|
|
pool.open()
|
|
pos += seg.guard + seg.size
|
|
}
|
|
}
|
|
var layout arm64PoolLayout
|
|
// The morestack block trails the body whatever the pool holds, and the
|
|
// guard prefix's conditional branches target it.
|
|
if fi.needSplit {
|
|
layout.blockStart = pos
|
|
pos += arm64MoreStackBlockLen
|
|
}
|
|
// The closing segment, the toolchain's flush at the function's last
|
|
// Prog. For a splitting function that Prog is the morestack block's
|
|
// trailing branch, so the words follow the block bare.
|
|
if seg := pool.activeSeg(); seg != nil {
|
|
seg.after = len(t.Body)
|
|
seg.base = pos
|
|
if last >= 0 {
|
|
seg.line = t.Body[last].(*ast.Instr).Pos().Line
|
|
}
|
|
if !fi.needSplit && !arm64EndsBlock(strings.ToUpper(t.Body[last].(*ast.Instr).Mnemonic.Text)) {
|
|
seg.guard = 4
|
|
}
|
|
pos += seg.guard + seg.size
|
|
}
|
|
layout.total = pos
|
|
return layout
|
|
}
|
|
|
|
// appendARM64PoolSeg emits a drained pool segment: the guard word, the
|
|
// toolchain's word-zero UNDEF or the branch over the words, when one is
|
|
// planned, then the words in first-use order.
|
|
func appendARM64PoolSeg(out []byte, seg *arm64PoolSeg) []byte {
|
|
if seg.guard > 0 {
|
|
if seg.branch {
|
|
out = append(out, a64wordLE(a64Branch(0, int32((seg.guard+seg.size)>>2)))...)
|
|
} else {
|
|
// The UNDEF word: not the BRK the UNDEF statement spells, only a
|
|
// faulting word nothing jumps to.
|
|
out = append(out, a64wordLE(0)...)
|
|
}
|
|
}
|
|
for _, e := range seg.order {
|
|
out = append(out, e.data...)
|
|
}
|
|
return out
|
|
}
|
|
|
|
// arm64JumpChain precomputes jump-to-jump folding: a label whose first
|
|
// instruction is an unconditional local jump redirects its own jumpers to
|
|
// the ultimate target. The Go toolchain chases these chains before it
|
|
// encodes branches, so matching its bytes requires the same redirection.
|
|
func arm64JumpChain(t *ast.Text) map[string]string {
|
|
leadsTo := map[string]string{}
|
|
for i, stmt := range t.Body {
|
|
l, ok := stmt.(*ast.Label)
|
|
if !ok {
|
|
continue
|
|
}
|
|
j := i + 1
|
|
for j < len(t.Body) {
|
|
if _, isLabel := t.Body[j].(*ast.Label); !isLabel {
|
|
break
|
|
}
|
|
j++
|
|
}
|
|
if j >= len(t.Body) {
|
|
continue
|
|
}
|
|
in, ok := t.Body[j].(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
mnem := strings.ToUpper(in.Mnemonic.Text)
|
|
if (mnem != "JMP" && mnem != "B") || len(in.Operands) != 1 {
|
|
continue
|
|
}
|
|
if name, ok := arm64LabelOK(in.Operands[0]); ok {
|
|
leadsTo[l.Name.Text] = name
|
|
}
|
|
}
|
|
chain := map[string]string{}
|
|
for name := range leadsTo {
|
|
visited := map[string]bool{name: true}
|
|
cur := name
|
|
for {
|
|
next, ok := leadsTo[cur]
|
|
if !ok || visited[next] {
|
|
break
|
|
}
|
|
visited[next] = true
|
|
cur = next
|
|
}
|
|
if cur != name {
|
|
chain[name] = cur
|
|
}
|
|
}
|
|
return chain
|
|
}
|
|
|
|
// arm64LabelOK returns the local label name of a jump operand.
|
|
func arm64LabelOK(op *ast.Operand) (string, bool) {
|
|
if op.Kind == ast.OpAddr && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
|
op.Addr.Base == "" && op.Addr.Sym.Name != "" {
|
|
return op.Addr.Sym.Name, true
|
|
}
|
|
return "", false
|
|
}
|
|
|
|
// arm64WritebackSuffix splits a mnemonic carrying the toolchain's post-index
|
|
// (.P) or pre-index (.W) suffix, as in MOVD.P or LDP.W. It reports the base
|
|
// mnemonic, the suffix letter and whether a suffix was present.
|
|
func arm64WritebackSuffix(mnem string) (base, wb string, ok bool) {
|
|
if before, ok0 := strings.CutSuffix(mnem, ".P"); ok0 {
|
|
return before, "P", true
|
|
}
|
|
if before, ok0 := strings.CutSuffix(mnem, ".W"); ok0 {
|
|
return before, "W", true
|
|
}
|
|
return mnem, "", false
|
|
}
|
|
|
|
// arm64PCAlignPad returns the padding PCALIGN inserts before the next
|
|
// instruction so that it starts at the requested boundary relative to the
|
|
// function start. The boundary must be a power of two between 8 and 2048,
|
|
// as the toolchain requires.
|
|
func arm64PCAlignPad(pos int, instr *ast.Instr) int {
|
|
if len(instr.Operands) != 1 || !isImmOperand(instr.Operands[0]) {
|
|
return 0
|
|
}
|
|
align := int(arm64Imm64(instr.Operands[0]))
|
|
if align < 8 || align > 2048 || align&(align-1) != 0 {
|
|
return 0
|
|
}
|
|
return (align - pos%align) % align
|
|
}
|
|
|
|
// isARM64MovMnemonic reports whether m is one of the MOV-family spellings the
|
|
// arm64 encoder treats as the MOV pseudo-instruction. FMOVQ rides the same
|
|
// load/store machinery but carries no register-move or immediate form.
|
|
func isARM64MovMnemonic(m string) bool {
|
|
switch m {
|
|
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
|
"FMOVS", "FMOVD", "FMOVQ":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// arm64InstrSize returns the encoded size of an instruction: 4 bytes for
|
|
// most, more for the multi-instruction expansions.
|
|
func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
|
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
|
ops := instr.Operands
|
|
|
|
if mnem == "RET" {
|
|
return len(arm64Return(fi))
|
|
}
|
|
if mnem == "PCALIGN" {
|
|
return arm64PCAlignPad(pos, instr)
|
|
}
|
|
if mnem == "BYTE" {
|
|
return len(ops)
|
|
}
|
|
if mnem == "DWORD" {
|
|
// Eight little-endian bytes per immediate.
|
|
return 8 * len(ops)
|
|
}
|
|
// The remainder family: the division into REGTMP plus the MSUB tail.
|
|
switch mnem {
|
|
case "REM", "REMW", "UREM", "UREMW":
|
|
return 8
|
|
}
|
|
if mnem == "NOP" {
|
|
return 0 // the zero-size pseudo-instruction, operand or not
|
|
}
|
|
switch mnem {
|
|
case "VMOVS", "VMOVD", "VMOVQ":
|
|
// ADRP + ADD + wide load against a pooled literal.
|
|
return 12
|
|
}
|
|
// Writeback (.P/.W) forms are always a single instruction.
|
|
if base, _, ok := arm64WritebackSuffix(mnem); ok {
|
|
if isARM64MovMnemonic(base) || a64InstrTable[base].format == a64FPair {
|
|
return 4
|
|
}
|
|
}
|
|
// The load/store pair family: one word in range, otherwise the REGTMP
|
|
// expansion the toolchain lowers to (two words within ±4095, three
|
|
// beyond, the pool words riding the function end).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair {
|
|
return arm64PairSize(mnem, ops, fi)
|
|
}
|
|
// The funcdata pseudo-statements contribute no bytes, the expanded
|
|
// FUNCDATA/PCDATA forms included.
|
|
switch mnem {
|
|
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED", "END", "FUNCDATA", "PCDATA":
|
|
return 0
|
|
}
|
|
// The extended-instruction layer is one instruction word in every form
|
|
// the registry takes: pass 1 must size a pinned statement at the 4 bytes
|
|
// the encoder will lay down, ahead of the scalar immediate expansions
|
|
// below, whose sizes would misread a Z destination (asm/arm64_ext.go).
|
|
if _, pinned, _ := arm64ExtStatement(mnem, ops); pinned {
|
|
return 4
|
|
}
|
|
switch mnem {
|
|
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
|
"FMOVS", "FMOVD", "FMOVQ":
|
|
return arm64MovSize(mnem, ops, fi)
|
|
}
|
|
// The logical-immediate family: one word on the bitmask fast path (a
|
|
// real destination, the flags-only TST spellings, or the flag-setting
|
|
// forms to ZR), otherwise the constant materialisation into REGTMP plus
|
|
// the register-form tail. Mirrors encodeARM64DPSR's decision exactly,
|
|
// the materialisation word count included: a MOVZ plus up to three MOVKs
|
|
// makes five words.
|
|
if len(ops) >= 2 && len(ops) <= 3 && isImmOperand(ops[0]) {
|
|
switch mnem {
|
|
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
|
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
|
|
"TST", "TSTW":
|
|
if v, ok := arm64ImmOperandValue(ops[0]); ok {
|
|
written := v
|
|
switch mnem {
|
|
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
|
|
v = ^v
|
|
}
|
|
width := 64
|
|
mwMnem := "MOVD"
|
|
if strings.HasSuffix(mnem, "W") {
|
|
width = 32
|
|
mwMnem = "MOVW"
|
|
}
|
|
sLogical := mnem == "ANDS" || mnem == "ANDSW" || mnem == "BICS" || mnem == "BICSW"
|
|
zrDest := false
|
|
if mnem != "TST" && mnem != "TSTW" {
|
|
zrDest = strings.EqualFold(operandRegName(ops[len(ops)-1]), "ZR")
|
|
}
|
|
if _, _, _, bc := a64LogicalImm(v, width); bc && (sLogical || mnem == "TST" || mnem == "TSTW" || !zrDest) {
|
|
return 4
|
|
}
|
|
if mw, err := encodeARM64LoadImm(27, written, mwMnem); err == nil {
|
|
return len(mw) + 4
|
|
}
|
|
return 4
|
|
}
|
|
}
|
|
}
|
|
switch mnem {
|
|
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
|
|
"ADDS", "ADDSW", "SUBS", "SUBSW":
|
|
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
|
// Size exactly as the encoder will emit: a single imm12 word, the
|
|
// two-word ADDCON2 split, or a materialisation into REGTMP plus
|
|
// the register form. Anything else would desynchronise the label
|
|
// offsets of pass 1 from the bytes pass 2 lays down.
|
|
if v, ok := arm64ImmOperandValue(ops[0]); ok {
|
|
rn, rd := 0, 0
|
|
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
|
|
rd = n
|
|
}
|
|
if len(ops) == 3 {
|
|
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
|
|
rn = n
|
|
}
|
|
}
|
|
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd, false); err == nil {
|
|
return 4 * len(ws)
|
|
}
|
|
}
|
|
return 4
|
|
}
|
|
}
|
|
return 4
|
|
}
|
|
|
|
// encodeARM64Instr encodes a single AArch64 instruction. lits collects the
|
|
// read-only literals a VMOVS/VMOVD/VMOVQ constant load needs; the file
|
|
// assembler lays them out once every function is encoded.
|
|
func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64FrameInfo, relocs *[]Reloc, resolve func(string) string, lits *arm64Literals, pool *arm64Pool, poolBase int) ([]byte, error) {
|
|
mnem := strings.ToUpper(instr.Mnemonic.Text)
|
|
ops := instr.Operands
|
|
|
|
// The extended-instruction layer: a statement whose mnemonic is
|
|
// registered in the extension registry and whose operands carry a
|
|
// scalable vector or predicate register encodes through the registry,
|
|
// before any scalar route can misread those operands. Scalar, NEON and
|
|
// FP operand lists never pin, so everything below runs exactly as it
|
|
// did (asm/arm64_ext.go).
|
|
if extops, pinned, convErr := arm64ExtStatement(mnem, ops); pinned {
|
|
if convErr != nil {
|
|
return nil, convErr
|
|
}
|
|
code, encErr := EncodeExtension(arch.ARM64, mnem, extops...)
|
|
if encErr != nil {
|
|
return nil, encErr
|
|
}
|
|
return code, nil
|
|
}
|
|
|
|
// Pseudo-instructions and special cases first.
|
|
switch mnem {
|
|
case "RET":
|
|
return arm64RetInstr(fi, ops, relocs), nil
|
|
case "NOP":
|
|
// The toolchain's ANOP is a zero-size pseudo-instruction: no bytes
|
|
// whatever operand rides it, the immediate, register and vector
|
|
// classes included (asm7.go's ANOP rows), and anything else is an
|
|
// illegal combination. A bare register operand parses as a symbol
|
|
// reference with no base, the way operandRegName reads it.
|
|
if len(ops) > 1 {
|
|
return nil, fmt.Errorf("NOP: illegal combination")
|
|
}
|
|
if len(ops) == 1 && !isImmOperand(ops[0]) {
|
|
op := ops[0]
|
|
name := operandRegName(op)
|
|
bare := op.Addr.Base == "" && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" &&
|
|
!op.Addr.HasOff && op.Addr.Index == ""
|
|
reg := arm64RegNum(name) >= 0 ||
|
|
len(name) > 1 && (name[0] == 'F' || name[0] == 'V') && strings.Trim(name[1:], "0123456789") == ""
|
|
if !bare || !reg {
|
|
return nil, fmt.Errorf("NOP: illegal combination")
|
|
}
|
|
}
|
|
return nil, nil
|
|
case "NOOP":
|
|
// NOOP is the real hint instruction: one word, and it takes no
|
|
// operand at all.
|
|
if len(ops) != 0 {
|
|
return nil, fmt.Errorf("NOOP: illegal combination")
|
|
}
|
|
return a64wordLE(a64NOP), nil
|
|
case "UNDEF":
|
|
return a64wordLE(a64BRK(0)), nil
|
|
case "WORD":
|
|
if len(ops) != 1 {
|
|
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
|
}
|
|
w := arm64Imm64(ops[0])
|
|
if w < 0 || w > 0xFFFFFFFF {
|
|
return nil, fmt.Errorf("WORD: immediate %d does not fit a 32-bit word", w)
|
|
}
|
|
return a64wordLE(uint32(w)), nil
|
|
case "DWORD":
|
|
// The toolchain's ADWORD (asm7.go case 11): eight little-endian
|
|
// bytes per immediate. A symbol operand would need the data
|
|
// section's relocation plumbing, which the code path does not
|
|
// reach, so the immediate form alone is supported.
|
|
if len(ops) == 0 {
|
|
return nil, fmt.Errorf("DWORD expects an operand")
|
|
}
|
|
var out []byte
|
|
for _, op := range ops {
|
|
if !isImmOperand(op) || op.Imm.Sym != nil {
|
|
return nil, fmt.Errorf("DWORD: only an immediate operand is supported")
|
|
}
|
|
v := arm64Imm64(op)
|
|
out = append(out, byte(v), byte(v>>8), byte(v>>16), byte(v>>24),
|
|
byte(v>>32), byte(v>>40), byte(v>>48), byte(v>>56))
|
|
}
|
|
return out, nil
|
|
case "GETCALLERPC":
|
|
// The toolchain's rewrite (obj7.go AGETCALLERPC): a leaf reads the
|
|
// link register, a function with a frame reads the saved LR at
|
|
// 0(SP), where both prologue shapes leave it.
|
|
if len(ops) != 1 {
|
|
return nil, fmt.Errorf("GETCALLERPC expects 1 operand, got %d", len(ops))
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[0]))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in GETCALLERPC")
|
|
}
|
|
if fi.leaf {
|
|
return a64wordLE(0xaa000000 | 30<<16 | 31<<5 | uint32(rd)), nil // MOVD R30, Rd
|
|
}
|
|
return a64wordLE(a64LSU(3, 0, 1, 0, 31, uint32(rd))), nil // MOVD (RSP), Rd
|
|
case "REM", "REMW", "UREM", "UREMW":
|
|
return encodeARM64Rem(mnem, ops)
|
|
case "B", "JMP":
|
|
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
|
|
case "BL", "CALL":
|
|
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
|
|
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
|
"FMOVS", "FMOVD", "FMOVQ":
|
|
return encodeARM64Mov(instr, mnem, "", fi, relocs, pool, poolBase, pc)
|
|
}
|
|
|
|
// Post-index (.P) and pre-index (.W) writeback forms: the MOV family and
|
|
// the load/store pair family carry the suffix on the mnemonic itself.
|
|
// (The SIMD VLD1.P/VST1.P/VLD1R.P/VLD4R.P spellings also end in .P, but
|
|
// for them the suffix is part of the mnemonic and the table routes them.)
|
|
if base, wb, ok := arm64WritebackSuffix(mnem); ok {
|
|
switch {
|
|
case isARM64MovMnemonic(base):
|
|
return encodeARM64Mov(instr, base, wb, fi, relocs, pool, poolBase, pc)
|
|
case a64InstrTable[base].format == a64FPair:
|
|
return encodeARM64Pair(base, a64InstrTable[base].op, ops, pc, fi, wb, relocs, pool, poolBase)
|
|
}
|
|
}
|
|
|
|
// The funcdata.h pseudo-statements (NO_LOCAL_POINTERS, GO_ARGS,
|
|
// GO_RESULTS_INITIALIZED) carry metadata for the linker, not machine
|
|
// code: the toolchain emits zero instruction bytes for them, and so does
|
|
// the encoder here. Files that include funcdata.h spell them after
|
|
// macro expansion as FUNCDATA $n, sym(SB), so the expanded forms are
|
|
// bookkeeping too (the same treatment the loong64 encoder applies).
|
|
switch mnem {
|
|
case "NO_LOCAL_POINTERS", "GO_ARGS", "GO_RESULTS_INITIALIZED":
|
|
return nil, nil
|
|
case "END":
|
|
if len(ops) != 0 {
|
|
return nil, fmt.Errorf("END expects no operands, got %d", len(ops))
|
|
}
|
|
return nil, nil
|
|
case "FUNCDATA":
|
|
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("FUNCDATA expects $n, sym(SB)")
|
|
}
|
|
return nil, nil
|
|
case "PCDATA":
|
|
if len(ops) != 2 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
|
|
return nil, fmt.Errorf("PCDATA expects $n, $n")
|
|
}
|
|
return nil, nil
|
|
}
|
|
|
|
// Conditional branches (BEQ, BNE, BGE, BLT, BGT, BLE, etc.).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranchCond {
|
|
return encodeARM64BranchCond(mnem, enc.op, ops, pc, offsets, resolve)
|
|
}
|
|
|
|
// Unconditional register branches (BR, BLR).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FUncondBranch {
|
|
return encodeARM64RegBranch(mnem, enc.op, ops)
|
|
}
|
|
|
|
// ADD/SUB immediate.
|
|
if mnem == "ADD" || mnem == "ADDW" || mnem == "SUB" || mnem == "SUBW" ||
|
|
mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
|
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" {
|
|
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
|
return encodeARM64AddSubImm(mnem, ops)
|
|
}
|
|
}
|
|
|
|
// Shifts: immediate forms alias SBFM/UBFM/EXTR, register forms are the
|
|
// two-source LSLV/LSRV/ASRV/RORV.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShift {
|
|
return encodeARM64Shift(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Multiply-accumulate: MADD/MSUB Rm, Ra, Rn, Rd.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPR4 {
|
|
return encodeARM64MAddSub(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Register-register data processing.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDPSR {
|
|
return encodeARM64DPSR(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP 3-operand (Rm, Rn, Rd).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP3 {
|
|
return encodeARM64FP3(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP unary (Rn, Rd).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPUnary {
|
|
return encodeARM64FPUnary(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP 4-operand FMA (Ra, Rm, Rn, Rd).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFP4 {
|
|
return encodeARM64FP4(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP compare (Rm, Rn or #0, Rn).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCmp {
|
|
return encodeARM64FPCmp(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP conditional compare (Rm, Rn, #nzcv, cond).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCCmp {
|
|
return encodeARM64FPCCmp(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP conditional select (Rm, Rn, Rd, cond).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPSel {
|
|
return encodeARM64FPSel(mnem, enc.op, ops)
|
|
}
|
|
|
|
// FP ↔ integer conversion.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FFPCvt {
|
|
return encodeARM64FPCvt(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Conditional select (CSEL, CSINC, CSINV, CSNEG, CSET, CSETM, CINC, CINV, CNEG).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCSEL {
|
|
return encodeARM64CSEL(mnem, enc.op, ops)
|
|
}
|
|
|
|
// CRC32.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCRC32 {
|
|
return encodeARM64CRC32(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Exclusive load/store (LDXR, STXR, LDAXR, STLXR and the register-pair
|
|
// forms LDXP, STXP).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FExcl {
|
|
return encodeARM64Excl(mnem, enc.op, ops)
|
|
}
|
|
|
|
// LSE atomics (LDADD, CAS, SWP) and the compare-and-swap pair.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FLSE {
|
|
return encodeARM64LSEAtom(mnem, enc.op, ops)
|
|
}
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCASP {
|
|
return encodeARM64CASP(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Bitfield/shift (ASR, LSL, LSR, ROR, BFI, BFXIL, SBFM, UBFM).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield {
|
|
return encodeARM64Bitfield(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
|
|
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
|
|
}
|
|
|
|
// EXTR.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
|
return encodeARM64Extr(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Acquire/release loads and stores (LDAR family, STLR family).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FAcqRel {
|
|
return encodeARM64AcqRel(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Load/store pairs (LDP, STP, LDPW, STPW, FLDPD, FSTPD). The .P/.W
|
|
// writeback forms are routed earlier, straight from the mnemonic.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair {
|
|
return encodeARM64Pair(mnem, enc.op, ops, pc, fi, "", relocs, pool, poolBase)
|
|
}
|
|
|
|
// Compare-and-branch and test-and-branch to a label.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBranch19 {
|
|
return encodeARM64Branch19(mnem, enc.op, ops, pc, offsets, resolve)
|
|
}
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FTestBranch {
|
|
return encodeARM64TestBranch(mnem, enc.op, ops, pc, offsets, resolve)
|
|
}
|
|
|
|
// Data-processing (1 source): RBIT, REV, CLZ, CLS.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FDP1 {
|
|
return encodeARM64DP1(mnem, enc.op, ops)
|
|
}
|
|
|
|
// ADR/ADRP: (label, Rd) with the byte distance split into immlo and
|
|
// immhi.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FADR {
|
|
return encodeARM64ADR(mnem, enc.op, ops, pc, offsets)
|
|
}
|
|
|
|
// Bitfield extract with wrapping immr: UBFX, SBFX.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfield2 {
|
|
return encodeARM64Bitfield2(mnem, enc.op, ops)
|
|
}
|
|
|
|
// Conditional compare: CCMP, CCMN.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCondCmp {
|
|
return encodeARM64CondCmp(mnem, enc.op, ops)
|
|
}
|
|
|
|
// System operations: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSys {
|
|
return encodeARM64Sys(mnem, ops)
|
|
}
|
|
|
|
// Crypto: AESD, AESE, SHA1C, SHA256H and friends.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCrypto2 {
|
|
return encodeARM64Crypto(mnem, enc.op, ops, 2)
|
|
}
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FCrypto3 {
|
|
return encodeARM64Crypto(mnem, enc.op, ops, 3)
|
|
}
|
|
|
|
// Move wide with an explicit immediate: MOVK.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMovWide {
|
|
return encodeARM64MoveWide(mnem, enc.op, ops)
|
|
}
|
|
|
|
// SIMD element moves (VDUP, VMOV with lane indices) take precedence
|
|
// over the plain arrangement paths, which carry no index.
|
|
if mnem == "VDUP" || mnem == "VMOV" {
|
|
if arm64SimdHasElement(ops) {
|
|
return encodeARM64Dup(mnem, ops)
|
|
}
|
|
// VMOV/VDUP Rn, Vd.<T>: a general register into an arranged whole
|
|
// vector (asm7.go case 82, shared by both mnemonics). The element
|
|
// paths above only run when a lane index is spelled, so this is the
|
|
// whole-vector shape's only route.
|
|
if b, ok, err := encodeARM64GPToVec(mnem, ops); ok {
|
|
return b, err
|
|
}
|
|
}
|
|
|
|
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
|
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
|
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
|
|
// only in the zero-immediate form (a64SimdVZero), so they route here with
|
|
// an empty register-form spec. VSQSHL/VUQSHL keep their shift-by-
|
|
// immediate route when the first operand is an immediate.
|
|
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
|
|
if !arm64SimdShiftImmRoute(mnem, ops) {
|
|
return encodeARM64SimdV(mnem, spec, ops)
|
|
}
|
|
}
|
|
|
|
// Arrangement-aware SIMD two-register (VREV32, VREV64, VUADDLV, VMOV).
|
|
if spec, ok := a64SimdV2Table[mnem]; ok {
|
|
return encodeARM64SimdV2(mnem, spec, ops)
|
|
}
|
|
|
|
// Narrow/long/wide SIMD families whose size and Q bits read off one
|
|
// designated operand (VXTN, VSXTL, VUADDW, VUMULL, VSHRN, VSSHLL, VFCVTN
|
|
// and friends).
|
|
if spec, ok := a64SimdNLTable[mnem]; ok {
|
|
return encodeARM64SimdNL(mnem, spec, ops)
|
|
}
|
|
|
|
// SIMD four-register and immediate three-register (VEOR3, VBCAX, VXAR,
|
|
// VEXT).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FSIMDV4 {
|
|
return encodeARM64SimdV4(mnem, enc.op, ops)
|
|
}
|
|
|
|
// SIMD table lookup.
|
|
if mnem == "VTBL" || mnem == "VTBX" {
|
|
return encodeARM64VTBL(mnem, ops)
|
|
}
|
|
|
|
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
|
|
// VLD4R).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVLDST {
|
|
return encodeARM64VLDST(mnem, enc.op, ops)
|
|
}
|
|
|
|
// SIMD shift by immediate (VSHL, VUSHR, VSRI).
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FShiftImm {
|
|
return encodeARM64ShiftImm(mnem, enc.op, ops)
|
|
}
|
|
|
|
// SIMD move immediate: VMOVI $imm8, Vd.B8/B16.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FVMoviImm {
|
|
return encodeARM64MoviImm(ops)
|
|
}
|
|
|
|
// VMOVS/VMOVD/VMOVQ with a large constant: a literal pool load.
|
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FMoviLit {
|
|
return encodeARM64MoviLit(mnem, enc.op, ops, relocs, lits)
|
|
}
|
|
|
|
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
|
}
|
|
|
|
// ---- branch encoding ----
|
|
|
|
// encodeARM64Branch encodes an unconditional branch (B/BL) to a label.
|
|
func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[string]int, link bool, relocs *[]Reloc, resolve func(string) string) ([]byte, error) {
|
|
if len(ops) != 1 {
|
|
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
|
|
}
|
|
op := ops[0]
|
|
|
|
// Branch to the program counter: JMP (PC) spins forever, and a spelled
|
|
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
|
|
// units. The toolchain encodes both as a plain branch of that offset.
|
|
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
|
rel := op.Addr.Offset
|
|
if rel < -(1<<25) || rel >= (1<<25) {
|
|
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
|
|
}
|
|
bop := uint32(0) // B
|
|
if link {
|
|
bop = 1 // BL
|
|
}
|
|
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
|
}
|
|
|
|
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
|
|
// toolchain's spelling carries no offset and no index; anything else
|
|
// is reported rather than silently dropped.
|
|
if op.Addr.Sym == nil && op.Addr.Base != "" {
|
|
if op.Addr.Offset != 0 || op.Addr.Index != "" {
|
|
return nil, fmt.Errorf("%s: invalid indirect branch operand %q", mnem, op.Raw)
|
|
}
|
|
rn := arm64RegNum(op.Addr.Base)
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("%s: unknown branch register %q", mnem, op.Addr.Base)
|
|
}
|
|
opc := uint32(0) // BR
|
|
if link {
|
|
opc = 1 // BLR
|
|
}
|
|
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
|
|
}
|
|
|
|
// The bare spelling BL R9 is the same indirect branch: the parser reads
|
|
// a bare identifier as a symbol, and one named for a register is an
|
|
// indirect branch through it, which the toolchain accepts alongside the
|
|
// parenthesised form (BL (R3) and BL R3 both encode BLR R3).
|
|
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" && op.Addr.Index == "" {
|
|
if rn := arm64RegNum(op.Addr.Sym.Name); rn >= 0 {
|
|
opc := uint32(0) // BR
|
|
if link {
|
|
opc = 1 // BLR
|
|
}
|
|
return a64wordLE(a64UncondBranch(opc, uint32(rn), 0)), nil
|
|
}
|
|
}
|
|
|
|
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
|
|
// relocation (R_CALLARM64 either way).
|
|
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
|
if relocs != nil {
|
|
*relocs = append(*relocs, Reloc{
|
|
Off: 0,
|
|
After: 4,
|
|
Name: op.Addr.Sym.Name,
|
|
Addend: op.Addr.Sym.Offset,
|
|
Kind: RelArm64Branch,
|
|
})
|
|
}
|
|
// Emit B/BL with zero offset; the linker fills in the target.
|
|
bop := uint32(0) // B
|
|
if link {
|
|
bop = 1 // BL
|
|
}
|
|
return a64wordLE(a64Branch(bop, 0)), nil
|
|
}
|
|
|
|
target := resolve(arm64Label(op))
|
|
targetOff, ok := offsets[target]
|
|
if !ok {
|
|
return nil, fmt.Errorf("undefined label %q", target)
|
|
}
|
|
rel := (targetOff - pc) >> 2
|
|
if rel < -(1<<25) || rel >= (1<<25) {
|
|
return nil, fmt.Errorf("branch to %q too far (26-bit range)", target)
|
|
}
|
|
bop := uint32(0) // B
|
|
if link {
|
|
bop = 1 // BL
|
|
}
|
|
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
|
}
|
|
|
|
// encodeARM64RegBranch encodes BR/BLR through a register operand:
|
|
// BR Xn = 0xd61f0000 | Rn<<5, BLR Xn = 0xd63f0000 | Rn<<5.
|
|
func encodeARM64RegBranch(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 1 {
|
|
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[0]))
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("%s expects a register operand", mnem)
|
|
}
|
|
return a64wordLE(uint32(baseOp) | 31<<16 | uint32(rn)<<5), nil
|
|
}
|
|
|
|
// encodeARM64BranchCond encodes a conditional branch (B.cond) to a label.
|
|
func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
|
if len(ops) != 1 {
|
|
return nil, fmt.Errorf("%s expects 1 operand, got %d", mnem, len(ops))
|
|
}
|
|
rel, pcRel := arm64PCRelOffset(ops[0])
|
|
if !pcRel {
|
|
target := resolve(arm64Label(ops[0]))
|
|
targetOff, ok := offsets[target]
|
|
if !ok {
|
|
return nil, fmt.Errorf("undefined label %q", target)
|
|
}
|
|
rel = (targetOff - pc) >> 2
|
|
}
|
|
if rel < -(1<<18) || rel >= (1<<18) {
|
|
return nil, fmt.Errorf("%s: branch offset %d out of 19-bit range", mnem, rel)
|
|
}
|
|
// The condition code is in the low 4 bits of baseOp.
|
|
cond := baseOp & 0xF
|
|
return a64wordLE(a64BranchCond(int32(rel), cond)), nil
|
|
}
|
|
|
|
// ---- data-processing (shifted register) ----
|
|
|
|
// encodeARM64DPSR encodes a data-processing (shifted register) instruction.
|
|
// For most instructions: OP Rm, Rn, Rd (3 operands) or OP Rm, Rd (2 operands, Rn=Rd).
|
|
// For CMP/CMN/TST: CMP Rm, Rn (Rd=ZR).
|
|
// For NEG: NEG Rm, Rd (Rn=ZR).
|
|
// The first operand may carry the toolchain's modifier shapes: a shifted
|
|
// register (R0<<2, R1>>3) or an extend modifier (R0.UXTW, R3.SXTW<<2).
|
|
// Logical instructions (AND/ANDS/BIC/…) also accept a bitmask immediate, and
|
|
// the ADD/SUB-with-flags family an add/sub immediate.
|
|
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
|
|
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
|
|
mnem == "MVN" || mnem == "MVNW" ||
|
|
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
|
|
|
|
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
|
|
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
|
|
// complement of the written value.
|
|
if len(ops) >= 2 && len(ops) <= 3 && isImmOperand(ops[0]) {
|
|
var logical bool
|
|
switch mnem {
|
|
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
|
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
|
|
"TST", "TSTW":
|
|
logical = true
|
|
}
|
|
if logical {
|
|
v, ok := arm64ImmOperandValue(ops[0])
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
|
|
}
|
|
// The value as written: the class rules (the SP destination's
|
|
// addcon band among them) read the written immediate, not the
|
|
// complement the inverted mnemonics encode.
|
|
writtenImm := v
|
|
inverted := false
|
|
switch mnem {
|
|
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
|
|
inverted = true
|
|
}
|
|
if inverted {
|
|
v = ^v
|
|
}
|
|
width := 64
|
|
if strings.HasSuffix(mnem, "W") {
|
|
width = 32
|
|
}
|
|
var rn, rd int
|
|
zrDest := false
|
|
switch len(ops) {
|
|
case 3:
|
|
rn = arm64RegNum(operandRegName(ops[1]))
|
|
rd = arm64RegNum(operandRegName(ops[2]))
|
|
zrDest = strings.EqualFold(operandRegName(ops[2]), "ZR")
|
|
default:
|
|
rd = arm64RegNum(operandRegName(ops[1]))
|
|
rn = rd
|
|
zrDest = strings.EqualFold(operandRegName(ops[1]), "ZR")
|
|
}
|
|
if isCmp {
|
|
// CMP/CMN/TST write the flags alone: the destination is ZR
|
|
// whatever the spelling says.
|
|
rd = 31
|
|
}
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// SP rules the toolchain enforces on the logical immediates: a
|
|
// two-operand spelling with SP as the source is rejected outright
|
|
// (illegal source register), a flag-setting logical rejects SP as
|
|
// the destination, and a plain logical to SP takes the fast
|
|
// bitmask path only inside the addcon band (asm7.go cases 62/13).
|
|
if !zrDest && len(ops) == 2 && rn == 31 {
|
|
if name := operandRegName(ops[1]); strings.EqualFold(name, "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal source register RSP", mnem)
|
|
}
|
|
}
|
|
rspDest := false
|
|
if len(ops) == 3 {
|
|
rspDest = strings.EqualFold(operandRegName(ops[2]), "RSP")
|
|
}
|
|
sLogical := mnem == "ANDS" || mnem == "ANDSW" || mnem == "BICS" || mnem == "BICSW"
|
|
if rspDest && sLogical {
|
|
return nil, fmt.Errorf("%s: illegal combination: the destination cannot be RSP", mnem)
|
|
}
|
|
n, immr, imms, ok := a64LogicalImm(v, width)
|
|
// The toolchain's logical-immediate rows take a real destination
|
|
// only for the non-flag-setting forms (omovconst guards the
|
|
// bitmask path with rt != REGZERO): a plain AND/ORR/EOR to ZR
|
|
// materialises the constant into REGTMP (R27) and takes the
|
|
// register form. The flag-setting forms (ANDS, BICS, the TST
|
|
// spellings) keep the fast path with a ZR destination: case 53
|
|
// encodes ANDS ZR, Rn, #imm directly. RSP is a real register
|
|
// here, but keeps the fast bitmask path inside the addcon band
|
|
// alone.
|
|
if rspDest {
|
|
inBand := writtenImm > 0 && (writtenImm <= 0xFFF || (writtenImm&0xFFF == 0 && writtenImm>>12 <= 0xFFF))
|
|
if !ok || !inBand {
|
|
return nil, fmt.Errorf("%s: illegal combination: the destination cannot be RSP", mnem)
|
|
}
|
|
}
|
|
if ok && (isCmp || !zrDest || sLogical) {
|
|
opc := (baseOp >> 29) & 7
|
|
sf := (baseOp >> 31) & 1
|
|
return a64wordLE(sf<<31 | opc<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 |
|
|
uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
// Beyond the bitmask immediates, and for the ZR destinations, the
|
|
// toolchain materialises the constant into REGTMP and uses the
|
|
// register form (asm7.go cases 62 and 13). A REGTMP source
|
|
// register is refused, the materialisation clobbering it before
|
|
// the register form reads it. BIC/ORN/EON read the written
|
|
// value, so the materialisation uses v before any inversion.
|
|
if rn == 27 {
|
|
return nil, fmt.Errorf("%s: cannot use REGTMP as source", mnem)
|
|
}
|
|
written := v
|
|
if inverted {
|
|
written = ^v
|
|
}
|
|
mwMnem := "MOVD"
|
|
if strings.HasSuffix(mnem, "W") {
|
|
mwMnem = "MOVW"
|
|
}
|
|
mw, merr := encodeARM64LoadImm(27, written, mwMnem)
|
|
if merr != nil {
|
|
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
|
}
|
|
// The register tail against SP takes the extended form, like the
|
|
// plain register path below.
|
|
tail := baseOp | 27<<16 | uint32(rn)<<5 | uint32(rd)
|
|
if opt, spok := arm64SpExtendOpt(mnem, ops[1:]); spok {
|
|
tail = baseOp | 1<<21 | opt<<13 | 27<<16 | uint32(rn)<<5 | uint32(rd)
|
|
}
|
|
return append(mw, a64wordLE(tail)...), nil
|
|
}
|
|
}
|
|
|
|
// Shifted-register and extend-modifier first operand: OP Rm<<k, Rn, Rd,
|
|
// OP Rm.UXTW, Rn, Rd, and the two-operand spellings of the same.
|
|
if len(ops) >= 2 && arm64RegMod(ops[0]) {
|
|
rm, shiftBits, extendOpt, isExtend, amount, ok := arm64RegModifier(ops[0])
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid register modifier %q", mnem, ops[0].Raw)
|
|
}
|
|
isAddSub := strings.HasPrefix(mnem, "ADD") || strings.HasPrefix(mnem, "SUB") ||
|
|
isCmp || isNeg || mnem == "ADC" || mnem == "ADCS" || mnem == "SBC" || mnem == "SBCS" ||
|
|
mnem == "ADCW" || mnem == "ADCSW" || mnem == "SBCW" || mnem == "SBCSW"
|
|
if isExtend && !isAddSub {
|
|
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
|
|
}
|
|
if !isExtend {
|
|
// ROR rides the shifted-register field only for the logical
|
|
// group; the toolchain reports "unsupported shift operator" for
|
|
// the arithmetic forms, whose shift=11 encoding is unallocated.
|
|
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
|
|
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
|
|
}
|
|
// The imm6 field is 5 bits and truncates at the 32-bit width.
|
|
limit := 63
|
|
if strings.HasSuffix(mnem, "W") {
|
|
limit = 31
|
|
}
|
|
if amount < 0 || amount > limit {
|
|
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
|
|
}
|
|
// SP-based ADD/SUB have no shifted-register encoding: the
|
|
// toolchain canonicalises LSL #n to the extend form (UXTX, or
|
|
// UXTW in the 32-bit forms) and rejects a right shift.
|
|
spInvolved := false
|
|
for _, op := range ops[1:] {
|
|
if n := operandRegName(op); n == "SP" || n == "RSP" {
|
|
spInvolved = true
|
|
}
|
|
}
|
|
if spInvolved && (strings.HasPrefix(mnem, "ADD") || strings.HasPrefix(mnem, "SUB") || isCmp) {
|
|
if shiftBits != 0 {
|
|
return nil, fmt.Errorf("%s: right shift not encodable against SP", mnem)
|
|
}
|
|
// The extend field carries 0..4 only (asm7.go: shift amount
|
|
// out of range 0 to 4).
|
|
if amount > 4 {
|
|
return nil, fmt.Errorf("%s: shift amount out of range 0 to 4", mnem)
|
|
}
|
|
opt := uint32(3) // UXTX
|
|
if strings.HasSuffix(mnem, "W") {
|
|
opt = 2 // UXTW
|
|
}
|
|
baseOp |= 1<<21 | opt<<13 | uint32(amount)<<10
|
|
} else {
|
|
baseOp |= shiftBits<<22 | uint32(amount)<<10
|
|
}
|
|
} else {
|
|
baseOp |= 1<<21 | extendOpt<<13 | uint32(amount)<<10
|
|
}
|
|
// The flag-setting add/sub family rejects SP as its destination.
|
|
switch mnem {
|
|
case "ADDS", "ADDSW", "SUBS", "SUBSW":
|
|
if strings.EqualFold(operandRegName(ops[len(ops)-1]), "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal destination register RSP", mnem)
|
|
}
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
|
rn := rd
|
|
if len(ops) == 3 {
|
|
rn = arm64RegNum(operandRegName(ops[1]))
|
|
}
|
|
if isCmp {
|
|
rd = 31
|
|
}
|
|
if isNeg && len(ops) == 2 {
|
|
rn = 31
|
|
}
|
|
if rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// ADD/SUB against SP take the extended-register form with the
|
|
// identity extend, the toolchain's spelling of a plain register
|
|
// operand against the stack pointer (asm7.go opxrrr against C_RSP).
|
|
if opt, ok := arm64SpExtendOpt(mnem, ops[1:]); ok {
|
|
return a64wordLE(baseOp | 1<<21 | opt<<13 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
switch len(ops) {
|
|
case 3:
|
|
// The carry family carries an immediate spelling in three operands
|
|
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
|
|
// operand, the same shape the two-operand form takes.
|
|
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
|
|
if v := arm64Imm64(ops[0]); v != 0 {
|
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[2]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
// OP Rm, Rn, Rd
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[2]))
|
|
if rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// The flag-setting add/sub family rejects SP as its destination.
|
|
switch mnem {
|
|
case "ADDS", "ADDSW", "SUBS", "SUBSW":
|
|
if strings.EqualFold(operandRegName(ops[2]), "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal destination register RSP", mnem)
|
|
}
|
|
}
|
|
// ADD/SUB against SP take the extended-register form with the
|
|
// identity extend, the toolchain's spelling of a plain register
|
|
// operand against the stack pointer (asm7.go opxrrr against C_RSP).
|
|
if opt, ok := arm64SpExtendOpt(mnem, ops[1:]); ok {
|
|
return a64wordLE(baseOp | 1<<21 | opt<<13 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
case 2:
|
|
// ADC family carries an immediate spelling: ADC $0, Rd reads the
|
|
// carry into Rd and takes ZR as the register operand. The
|
|
// encoding has no immediate field, so $0 is the only value.
|
|
if isImmOperand(ops[0]) {
|
|
v := arm64Imm64(ops[0])
|
|
if v != 0 {
|
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | 31<<16 | uint32(rd)<<5 | uint32(rd)), nil
|
|
}
|
|
if isCmp {
|
|
// CMP Rm, Rn → SUBS XZR, Rn, Rm
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
if rm < 0 || rn < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
if opt, ok := arm64SpExtendOpt(mnem, ops[1:]); ok {
|
|
return a64wordLE(baseOp | 1<<21 | opt<<13 | uint32(rm)<<16 | uint32(rn)<<5 | 31), nil
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | 31), nil
|
|
}
|
|
if isNeg {
|
|
// NEG Rm, Rd → SUB Rd, ZR, Rm
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rm < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | 31<<5 | uint32(rd)), nil
|
|
}
|
|
// OP Rm, Rd → OP Rm, Rd, Rd
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rm < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
|
|
}
|
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
|
|
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
|
|
// shifted-register group, the only forms whose register operand accepts the
|
|
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
|
|
func arm64LogicalShifted(mnem string) bool {
|
|
switch mnem {
|
|
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
|
|
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
|
|
"TST", "TSTW", "MVN", "MVNW":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
|
|
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
|
|
// data-processing instructions the toolchain accepts an immediate $0
|
|
// operand spelling for.
|
|
func arm64CarryOp(mnem string) bool {
|
|
switch mnem {
|
|
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
// arm64RegMod reports whether a register operand carries the shifted-register
|
|
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
|
|
// option (R0.UXTW, R3.SXTW<<2).
|
|
func arm64RegMod(op *ast.Operand) bool {
|
|
if op.Addr.Shift != "" {
|
|
return true
|
|
}
|
|
name := operandRegName(op)
|
|
if _, after, ok := strings.Cut(name, "."); ok {
|
|
return strings.IndexByte(after, '[') < 0 // element selectors are not extend modifiers
|
|
}
|
|
return false
|
|
}
|
|
|
|
// arm64RegModifier resolves a modified register operand: the register number,
|
|
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
|
|
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
|
|
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
|
|
// before the operator match.
|
|
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
|
|
name := operandRegName(op)
|
|
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
|
|
if before, after, ok0 := strings.Cut(name, "."); ok0 {
|
|
switch strings.ToUpper(strings.TrimSpace(after)) {
|
|
case "UXTB":
|
|
extendOpt = 0
|
|
case "UXTH":
|
|
extendOpt = 1
|
|
case "UXTW", "UXTW32":
|
|
extendOpt = 2
|
|
case "UXTX":
|
|
extendOpt = 3
|
|
case "SXTB":
|
|
extendOpt = 4
|
|
case "SXTH":
|
|
extendOpt = 5
|
|
case "SXTW":
|
|
extendOpt = 6
|
|
case "SXTX":
|
|
extendOpt = 7
|
|
default:
|
|
return 0, 0, 0, false, 0, false
|
|
}
|
|
extend = true
|
|
rm = arm64RegNum(strings.TrimSpace(before))
|
|
if rm < 0 {
|
|
return 0, 0, 0, false, 0, false
|
|
}
|
|
amount, ok = arm64ShiftAmount(shift)
|
|
if !ok || amount < 0 || amount > 4 {
|
|
return 0, 0, 0, false, 0, false
|
|
}
|
|
return rm, 0, extendOpt, true, amount, true
|
|
}
|
|
shiftKind = 0 // LSL
|
|
switch {
|
|
case strings.HasPrefix(shift, "<<"):
|
|
shiftKind = 0
|
|
case strings.HasPrefix(shift, ">>"):
|
|
shiftKind = 1 // LSR
|
|
case strings.HasPrefix(shift, "->"):
|
|
shiftKind = 2 // ASR
|
|
case strings.HasPrefix(shift, "@>"):
|
|
shiftKind = 3 // ROR
|
|
default:
|
|
return 0, 0, 0, false, 0, false
|
|
}
|
|
amount, ok = arm64ShiftAmount(shift)
|
|
if !ok {
|
|
return 0, 0, 0, false, 0, false
|
|
}
|
|
rm = arm64RegNum(name)
|
|
if rm < 0 {
|
|
return 0, 0, 0, false, 0, false
|
|
}
|
|
return rm, shiftKind, 0, false, amount, true
|
|
}
|
|
|
|
// arm64ShiftAmount extracts the integer after the shift operator in a
|
|
// shift suffix (<<, >>, ->, @>).
|
|
func arm64ShiftAmount(shift string) (int, bool) {
|
|
s := strings.TrimSpace(shift)
|
|
s = strings.TrimPrefix(s, "<<")
|
|
s = strings.TrimPrefix(s, ">>")
|
|
s = strings.TrimPrefix(s, "->")
|
|
s = strings.TrimPrefix(s, "@>")
|
|
s = strings.TrimSpace(s)
|
|
if s == "" {
|
|
if shift == "" {
|
|
return 0, true
|
|
}
|
|
return 0, false
|
|
}
|
|
v, err := strconv.Atoi(s)
|
|
if err != nil {
|
|
return 0, false
|
|
}
|
|
return v, true
|
|
}
|
|
|
|
// encodeARM64Shift encodes LSL/LSR/ASR/ROR in both widths. The operand order
|
|
// is source first, destination last: OP $sh|Rm, Rn, Rd or OP $sh|Rm, Rd.
|
|
// With an immediate the shift is the SBFM/UBFM (ROR: EXTR) alias, with a
|
|
// register it is the data-processing (2 source) LSLV/LSRV/ASRV/RORV; the
|
|
// two-source opcode rides the same 0xd6<<21 field as SDIV/UDIV, with
|
|
// LSLV=0b001000, LSRV=0b001001, ASRV=0b001010, RORV=0b001011 at bits 15:10.
|
|
func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 && len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
|
|
if isImmOperand(ops[0]) {
|
|
width := uint32(64)
|
|
if strings.HasSuffix(mnem, "W") {
|
|
width = 32
|
|
}
|
|
sh := arm64Imm64(ops[0])
|
|
if sh < 0 || uint32(sh) >= width {
|
|
return nil, fmt.Errorf("%s: shift amount %d out of range for %d-bit form", mnem, sh, width)
|
|
}
|
|
switch mnem {
|
|
case "LSL", "LSLW":
|
|
// UBFM Rd, Rn, #(-sh) mod W, #(W-1)-sh
|
|
immr := (width - uint32(sh)) % width
|
|
return a64wordLE(baseOp | immr<<16 | (width-1-uint32(sh))<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
case "LSR", "LSRW":
|
|
// UBFM Rd, Rn, #sh, #(W-1)
|
|
return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
case "ASR", "ASRW":
|
|
// SBFM Rd, Rn, #sh, #(W-1)
|
|
return a64wordLE(baseOp | uint32(sh)<<16 | (width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
default:
|
|
// ROR, RORW: EXTR Rd, Rn, Rn, #sh (Rm = Rn, imms = sh).
|
|
return a64wordLE(baseOp | uint32(rn)<<16 | uint32(sh)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
}
|
|
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
if rm < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
op2 := uint32(8) // LSLV
|
|
switch mnem {
|
|
case "LSR", "LSRW":
|
|
op2 = 9 // LSRV
|
|
case "ASR", "ASRW":
|
|
op2 = 10 // ASRV
|
|
case "ROR", "RORW":
|
|
op2 = 11 // RORV
|
|
}
|
|
sf := uint32(1)
|
|
if strings.HasSuffix(mnem, "W") {
|
|
sf = 0
|
|
}
|
|
return a64wordLE(sf<<31 | 0xd6<<21 | op2<<10 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64MAddSub encodes MADD/MSUB/MADDW/MSUBW. The toolchain's operand
|
|
// order is Rm, Ra, Rn, Rd (its optab case 15 comment says exactly that), so
|
|
// the accumulate register is the SECOND operand: base | Rm<<16 | Ra<<10 |
|
|
// Rn<<5 | Rd. The optab has no shorter row for these mnemonics, so all four
|
|
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
|
|
// MUL mnemonic, not to these.
|
|
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
// The widening three-operand forms (SMULL, UMNEGL, …) read the
|
|
// accumulate register as ZR, already preset in the table's base word.
|
|
if len(ops) == 3 {
|
|
switch mnem {
|
|
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
|
|
default:
|
|
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
|
|
}
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[2]))
|
|
if rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// ADD/SUB against SP take the extended-register form with the
|
|
// identity extend, the toolchain's spelling of a plain register
|
|
// operand against the stack pointer (asm7.go opxrrr against C_RSP).
|
|
if opt, ok := arm64SpExtendOpt(mnem, ops[1:]); ok {
|
|
return a64wordLE(baseOp | 1<<21 | opt<<13 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
|
|
}
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
ra := arm64RegNum(operandRegName(ops[1]))
|
|
rn := arm64RegNum(operandRegName(ops[2]))
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rm < 0 || rn < 0 || ra < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(ra)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// ---- ADD/SUB immediate ----
|
|
|
|
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
|
|
// following the toolchain's immediate classification (asm7.go conclass and
|
|
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
|
|
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
|
|
// otherwise a constant materialisation into REGTMP (R27) followed by the
|
|
// register form.
|
|
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 && len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
v, ok := arm64ImmOperandValue(ops[0])
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
|
rn := rd
|
|
if len(ops) == 3 {
|
|
rn = arm64RegNum(operandRegName(ops[1]))
|
|
}
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// The flag-setting add/sub family rejects SP as its destination
|
|
// (asm7.go: illegal destination register).
|
|
switch mnem {
|
|
case "ADDS", "ADDSW", "SUBS", "SUBSW":
|
|
if strings.EqualFold(operandRegName(ops[len(ops)-1]), "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal destination register RSP", mnem)
|
|
}
|
|
}
|
|
// CMP/CMN discard the destination.
|
|
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
|
|
rd = 31 // ZR
|
|
}
|
|
// Against SP the register tail is the extended form with the identity
|
|
// extend; the comparison zeroing above hides the destination, so the
|
|
// decision reads the written destination first.
|
|
rdWritten := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
|
ext := rn == 31 || rdWritten == 31
|
|
switch mnem {
|
|
case "ADD", "ADDW", "ADDS", "ADDSW", "SUB", "SUBW", "SUBS", "SUBSW",
|
|
"CMP", "CMPW", "CMN", "CMNW":
|
|
default:
|
|
ext = false
|
|
}
|
|
ws, err := arm64AddSubImmWords(mnem, v, rn, rd, ext)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("%s: %w", mnem, err)
|
|
}
|
|
return a64WordsLE(ws...), nil
|
|
}
|
|
|
|
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
|
|
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
|
|
// their W forms. rn and rd are resolved register numbers (a comparison
|
|
// discards rd, so the caller passes 31).
|
|
func arm64AddSubImmWords(mnem string, v int64, rn, rd int, ext bool) ([]uint32, error) {
|
|
w := strings.HasSuffix(mnem, "W")
|
|
sf := uint32(1) // 64-bit
|
|
d := v
|
|
if w {
|
|
sf = 0 // 32-bit
|
|
// The W forms classify the 32-bit value (asm7.go con32class).
|
|
d = int64(uint32(v))
|
|
}
|
|
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
|
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
|
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
|
op := uint32(0) // ADD
|
|
S := uint32(0)
|
|
if isSub {
|
|
op = 1
|
|
}
|
|
if isS {
|
|
S = 1
|
|
}
|
|
single := func(sh, imm12 uint32) []uint32 {
|
|
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
|
|
}
|
|
|
|
// imm12: plain, then the one-shifted-by-12 form.
|
|
if d >= 0 && d <= 0xFFF {
|
|
return single(0, uint32(d)), nil
|
|
}
|
|
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
|
|
return single(1, uint32(d>>12)), nil
|
|
}
|
|
|
|
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
|
|
// split into two imm12 instructions, low half first (asm7.go case 48).
|
|
// The encoding is complete in itself: no REGTMP, no register form. The S
|
|
// forms must not break addition/subtraction, so the toolchain
|
|
// reclassifies them and falls through to the materialisation below.
|
|
dm := ^d
|
|
if w {
|
|
dm = ^d & 0xFFFFFFFF
|
|
}
|
|
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
|
|
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
|
|
return []uint32{
|
|
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
|
|
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
|
|
}, nil
|
|
}
|
|
|
|
// Constant into REGTMP (R27), then the register form. The first word
|
|
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
|
|
// the complement form, the bitmask ORR otherwise, and the full
|
|
// omovlconst sequence when no single word carries the value. A REGTMP
|
|
// source register is refused: the materialisation would clobber it
|
|
// before the register form reads it (asm7.go cases 13 and 62).
|
|
if rn == 27 {
|
|
return nil, fmt.Errorf("%s: cannot use REGTMP as source", mnem)
|
|
}
|
|
var seq []uint32
|
|
switch s := arm64Movcon(d); {
|
|
case s >= 0:
|
|
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
|
|
case arm64Movcon(dm) >= 0:
|
|
s := arm64Movcon(dm)
|
|
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
|
|
case isBitcon:
|
|
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
|
|
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
|
|
default:
|
|
seq = arm64MovLConst(d, sf)
|
|
}
|
|
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
|
|
// Against SP it takes the extended form with the identity extend,
|
|
// UXTX, or UXTW in the 32-bit forms.
|
|
tail := a64InstrTable[mnem].op | 27<<16 | uint32(rn)<<5 | uint32(rd)
|
|
if ext {
|
|
opt := uint32(3)
|
|
if w {
|
|
opt = 2
|
|
}
|
|
tail = a64InstrTable[mnem].op | 1<<21 | opt<<13 | 27<<16 | uint32(rn)<<5 | uint32(rd)
|
|
}
|
|
seq = append(seq, tail)
|
|
for i := range seq[:len(seq)-1] {
|
|
seq[i] |= 27 // REGTMP
|
|
}
|
|
return seq, nil
|
|
}
|
|
|
|
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
|
|
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
|
|
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
|
|
// returned with the destination field clear so the caller can OR its own
|
|
// register in. movcon and movcon-of-complement must fail for d before this
|
|
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
|
|
func arm64MovLConst(d int64, sf uint32) []uint32 {
|
|
if sf == 0 {
|
|
// omovlconst AMOVW: both 16-bit halves, low first.
|
|
return []uint32{
|
|
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
|
|
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
|
|
}
|
|
}
|
|
dn := ^d
|
|
var immh [4]uint64
|
|
zero, neg := 0, 0
|
|
for i := range immh {
|
|
immh[i] = uint64(d>>(i*16)) & 0xFFFF
|
|
switch immh[i] {
|
|
case 0:
|
|
zero++
|
|
case 0xFFFF:
|
|
neg++
|
|
}
|
|
}
|
|
mw := func(opc uint32, val int64, chunk int) uint32 {
|
|
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
|
|
}
|
|
var os []uint32
|
|
switch {
|
|
case zero == 2:
|
|
// one MOVZ and one MOVK
|
|
i := 0
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0 {
|
|
os = append(os, mw(2, d, i))
|
|
i++
|
|
break
|
|
}
|
|
}
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0 {
|
|
os = append(os, mw(3, d, i))
|
|
}
|
|
}
|
|
case neg == 2:
|
|
// one MOVN and one MOVK
|
|
i := 0
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0xFFFF {
|
|
os = append(os, mw(0, dn, i))
|
|
i++
|
|
break
|
|
}
|
|
}
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0xFFFF {
|
|
os = append(os, mw(3, d, i))
|
|
}
|
|
}
|
|
default:
|
|
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
|
|
// a single MOVK (constants from strength-reduced division).
|
|
if zero == 0 && neg == 0 {
|
|
for i := range 4 {
|
|
mask := uint64(0xFFFF) << (i * 16)
|
|
for period := 2; period <= 32; period *= 2 {
|
|
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
|
|
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
|
|
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
|
|
os = append(os, mw(3, d, i))
|
|
return os
|
|
}
|
|
}
|
|
}
|
|
}
|
|
switch {
|
|
case zero >= 1:
|
|
// one MOVZ and up to three MOVKs
|
|
i := 0
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0 {
|
|
os = append(os, mw(2, d, i))
|
|
i++
|
|
break
|
|
}
|
|
}
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0 {
|
|
os = append(os, mw(3, d, i))
|
|
}
|
|
}
|
|
case neg >= 1:
|
|
// one MOVN and up to three MOVKs
|
|
i := 0
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0xFFFF {
|
|
os = append(os, mw(0, dn, i))
|
|
i++
|
|
break
|
|
}
|
|
}
|
|
for ; i < 4; i++ {
|
|
if immh[i] != 0xFFFF {
|
|
os = append(os, mw(3, d, i))
|
|
}
|
|
}
|
|
default:
|
|
// one MOVZ and three MOVKs
|
|
os = append(os, mw(2, d, 0))
|
|
for i := 1; i < 4; i++ {
|
|
os = append(os, mw(3, d, i))
|
|
}
|
|
}
|
|
}
|
|
return os
|
|
}
|
|
|
|
// ---- MOV pseudo-instruction ----
|
|
|
|
// encodeARM64Mov encodes the MOV family, the load/store/immediate workhorse
|
|
// of Go's arm64 assembly. MOV is an alias of MOVD (the width mnemonics
|
|
// select the access width). The forms, mirroring the toolchain:
|
|
//
|
|
// MOVx $imm, rd load immediate (MOVZ/MOVN/MOVK)
|
|
// MOVx mem, rd load from memory
|
|
// MOVx rd, mem store to memory
|
|
// MOVx rs, rd register move (ORR Rd, ZR, Rs)
|
|
// MOVx $sym(SB), rd address of a static symbol (ADRP+ADD)
|
|
// MOVx sym(SB), rd load from a static symbol (ADRP+LDR)
|
|
// MOVx rd, sym(SB) store to a static symbol (ADRP+STR)
|
|
//
|
|
// With the writeback suffix (MOVD.P, MOVD.W, …) the memory form becomes a
|
|
// post-index or pre-index access whose offset is the base writeback amount.
|
|
// Storing a $0 immediate stores ZR; any other immediate is rejected, matching
|
|
// the toolchain.
|
|
func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, relocs *[]Reloc, pool *arm64Pool, poolBase int, pc int) ([]byte, error) {
|
|
ops := instr.Operands
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
|
|
if mnem == "FMOVQ" {
|
|
// The 128-bit FP move is a pure memory access in the toolchain: its
|
|
// table carries no register-to-register or immediate row, and both
|
|
// spellings are rejected ("illegal combination"), so only the
|
|
// load/store and static-symbol forms exist here either.
|
|
loadOK := arm64IsMemOperand(src) && !arm64IsMemOperand(dst) && !isImmOperand(src)
|
|
storeOK := arm64IsMemOperand(dst) && !arm64IsMemOperand(src) && !isImmOperand(src)
|
|
sbLoad := src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB"
|
|
sbStore := dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB"
|
|
if !loadOK && !storeOK && !sbLoad && !sbStore {
|
|
return nil, fmt.Errorf("%s: only the memory load and store forms exist", mnem)
|
|
}
|
|
}
|
|
|
|
if wb != "" {
|
|
switch {
|
|
case isMemOperand(src) && !isMemOperand(dst):
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s: invalid destination register", mnem)
|
|
}
|
|
return encodeARM64MemOp(mnem, src, rd, true, fi, wb, pool, poolBase, pc)
|
|
case isMemOperand(dst) && !isMemOperand(src):
|
|
rs := arm64RegNum(operandRegName(src))
|
|
if rs < 0 {
|
|
if !isImmOperand(src) || arm64Imm64(src) != 0 {
|
|
return nil, fmt.Errorf("%s: invalid source register", mnem)
|
|
}
|
|
// Storing a constant zero stores the zero register.
|
|
rs = 31
|
|
}
|
|
return encodeARM64MemOp(mnem, dst, rs, false, fi, wb, pool, poolBase, pc)
|
|
default:
|
|
return nil, fmt.Errorf("%s: writeback form needs a register and a memory operand", mnem)
|
|
}
|
|
}
|
|
|
|
// Immediate → register (including $sym(SB)).
|
|
if isImmOperand(src) && !isMemOperand(src) {
|
|
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s $sym(SB): invalid destination register", mnem)
|
|
}
|
|
return encodeARM64SBAddr(src.Imm.Sym, rd, relocs), nil
|
|
}
|
|
// Frame-relative immediate address: MOVD $sym+off(FP|SP), Rd
|
|
// materialises the address with ADD/SUB from the hardware SP
|
|
// (asm7.go case 4), the shape every runtime address-of-argument
|
|
// load is written in.
|
|
if src.Imm.Sym != nil && (src.Imm.Sym.Pseudo == "FP" || src.Imm.Sym.Pseudo == "SP") {
|
|
return encodeARM64FrameAddr(src.Imm.Sym, dst, mnem, fi)
|
|
}
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
// The FP immediates: FMOVS/FMOVD $f, Fd ride the FMOV (immediate)
|
|
// instruction when the 8-bit field carries the value and the
|
|
// FMOV-from-ZR move for zero; the toolchain pools anything else
|
|
// (obj7.go's preprocess), which needs pool symbols gasm does not
|
|
// carry. An integer immediate to an FP register is rejected, the
|
|
// zero spelling alone excepted, and FMOVQ has no immediate row.
|
|
if mnem == "FMOVS" || mnem == "FMOVD" {
|
|
if arm64RegClassOf(operandRegName(dst)) == arm64ClsFP {
|
|
f, ok := arm64FloatOperandValue(src)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: illegal combination: an integer immediate needs a real register", mnem)
|
|
}
|
|
if enc := arm64ChipFloat(f); enc > 0 {
|
|
base := uint32(0x1e201000) // FMOV Sd, #imm
|
|
if mnem == "FMOVD" {
|
|
base = 0x1e601000 // FMOV Dd, #imm
|
|
}
|
|
return a64wordLE(base | uint32(enc)<<13 | uint32(rd)), nil
|
|
}
|
|
zero := math.Float64bits(f) == 0
|
|
if mnem == "FMOVS" {
|
|
zero = math.Float32bits(float32(f)) == 0
|
|
}
|
|
if zero {
|
|
base := uint32(0x1e270000) // FMOV Sd, (W)ZR
|
|
if mnem == "FMOVD" {
|
|
base = 0x9e670000 // FMOV Dd, XZR
|
|
}
|
|
return a64wordLE(base | 31<<5 | uint32(rd)), nil
|
|
}
|
|
// The toolchain pools this through its $f64 symbols with a
|
|
// PC-relative relocation; gasm keeps the materialised
|
|
// sequence the pre-pool encoder used (a documented byte
|
|
// deviation outside the corpus families).
|
|
return encodeARM64LoadImmClass(rd, arm64Imm64(src), mnem, !strings.EqualFold(operandRegName(dst), "ZR"))
|
|
}
|
|
}
|
|
// The con(register) form: MOVD $con(Rn), Rd adds the displacement to
|
|
// the base register (asm7.go case 4). The toolchain rejects every
|
|
// other width and the ZR destination outright (RSP is a real register
|
|
// here, the C_RSP row).
|
|
if rn, ok := arm64ImmWithBase(src); ok {
|
|
if mnem != "MOVD" && mnem != "MOV" {
|
|
return nil, fmt.Errorf("%s: illegal combination: the con(register) form exists for MOVD only", mnem)
|
|
}
|
|
if strings.EqualFold(operandRegName(dst), "ZR") {
|
|
return nil, fmt.Errorf("%s: illegal combination: the con(register) form needs a real destination register", mnem)
|
|
}
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s $con(Rn): invalid destination register", mnem)
|
|
}
|
|
con, _ := arm64ImmOperandValue(src)
|
|
return encodeARM64ConRn(rn, con, rd, pool, poolBase, pc)
|
|
}
|
|
// Immediate → memory: only storing zero is encodable (the ZR
|
|
// register); the toolchain rejects any other immediate-to-memory
|
|
// combination ("illegal combination").
|
|
if isMemOperand(dst) {
|
|
if arm64Imm64(src) != 0 {
|
|
return nil, fmt.Errorf("%s: illegal combination: an immediate store must be zero", mnem)
|
|
}
|
|
return encodeARM64MemOp(mnem, dst, 31, false, fi, "", pool, poolBase, pc)
|
|
}
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s $imm: invalid destination register", mnem)
|
|
}
|
|
if strings.EqualFold(operandRegName(dst), "ZR") {
|
|
// The destination is ZR: omovconst's bitmask path needs a real
|
|
// register, so the value rides MOVZ/MOVN.
|
|
return encodeARM64LoadImmClass(rd, arm64Imm64(src), mnem, false)
|
|
}
|
|
return encodeARM64LoadImm(rd, arm64Imm64(src), mnem)
|
|
}
|
|
|
|
// Static symbol load/store via ADRP.
|
|
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB" && isMemOperand(src) {
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s sym(SB): invalid destination register", mnem)
|
|
}
|
|
// A TLSBSS symbol's load rides the local-exec model: one MOVZ word
|
|
// with the R_ARM64_TLS_LE relocation (asm7.go case 69), the width
|
|
// rows existing for MOVD alone.
|
|
if fi.tls[src.Addr.Sym.Name] {
|
|
if mnem != "MOVD" && mnem != "MOV" {
|
|
return nil, fmt.Errorf("%s: illegal combination: the TLS load exists for MOVD only", mnem)
|
|
}
|
|
if relocs != nil {
|
|
*relocs = append(*relocs, Reloc{
|
|
Off: 0,
|
|
After: 4,
|
|
Name: src.Addr.Sym.Name,
|
|
Addend: src.Addr.Sym.Offset,
|
|
Kind: RelArm64TLSLE,
|
|
})
|
|
}
|
|
return a64wordLE(a64MoveWide(1, 2, 0, 0, uint32(rd))), nil
|
|
}
|
|
return encodeARM64SBLoad(src.Addr.Sym, rd, mnem, relocs)
|
|
}
|
|
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB" && isMemOperand(dst) {
|
|
rs := arm64RegNum(operandRegName(src))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("%s rd, sym(SB): invalid source register", mnem)
|
|
}
|
|
return encodeARM64SBStore(dst.Addr.Sym, rs, mnem, relocs)
|
|
}
|
|
|
|
// System-register moves: MOVD NZCV, R0 reads (MRS) and MOVD R0, NZCV
|
|
// writes (MSR) the flag and FP status registers.
|
|
if src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "" && src.Addr.Base == "" {
|
|
if base, ok := a64MRSOps[src.Addr.Sym.Name]; ok {
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s %s: invalid destination register", mnem, src.Addr.Sym.Name)
|
|
}
|
|
return a64wordLE(base | uint32(rd)&31), nil
|
|
}
|
|
}
|
|
if dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "" && dst.Addr.Base == "" {
|
|
if base, ok := a64MSRRegOps[dst.Addr.Sym.Name]; ok {
|
|
rs := arm64RegNum(operandRegName(src))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("%s %s: invalid source register", mnem, dst.Addr.Sym.Name)
|
|
}
|
|
return a64wordLE(base | uint32(rs)&31), nil
|
|
}
|
|
}
|
|
|
|
// Memory load/store with offset.
|
|
if arm64IsMemOperand(src) && !arm64IsMemOperand(dst) {
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s: invalid destination register", mnem)
|
|
}
|
|
return encodeARM64MemOp(mnem, src, rd, true, fi, "", pool, poolBase, pc)
|
|
}
|
|
if !arm64IsMemOperand(src) && arm64IsMemOperand(dst) {
|
|
rs := arm64RegNum(operandRegName(src))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("%s: invalid source register", mnem)
|
|
}
|
|
return encodeARM64MemOp(mnem, dst, rs, false, fi, "", pool, poolBase, pc)
|
|
}
|
|
|
|
// Register → register.
|
|
return encodeARM64RegMove(mnem, src, dst)
|
|
}
|
|
|
|
// encodeARM64Rem encodes the remainder family the toolchain's optab case
|
|
// 16 does: the quotient rides REGTMP (R27) through SDIV/UDIV and the
|
|
// remainder is the minuend less REGTMP times the divisor, an MSUB whose W
|
|
// forms keep the 32-bit size. The two-operand spelling divides into the
|
|
// destination (asm7.go: r defaults to rt), and an RSP destination is the
|
|
// toolchain's illegal combination.
|
|
func encodeARM64Rem(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 && len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
|
if strings.EqualFold(operandRegName(ops[len(ops)-1]), "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal combination: the destination cannot be RSP", mnem)
|
|
}
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rn := rd
|
|
if len(ops) == 3 {
|
|
rn = arm64RegNum(operandRegName(ops[1]))
|
|
}
|
|
if rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
sf := uint32(1) // 64-bit
|
|
if strings.HasSuffix(mnem, "W") {
|
|
sf = 0
|
|
}
|
|
div := sf<<31 | 0xd6<<21 | 3<<10 // SDIV / SDIVW
|
|
if strings.HasPrefix(mnem, "U") {
|
|
div = sf<<31 | 0xd6<<21 | 2<<10 // UDIV / UDIVW
|
|
}
|
|
msub := sf<<31 | 0x1b<<24 | 1<<15 // MSUB / MSUBW
|
|
return a64WordsLE(
|
|
div|uint32(rm)<<16|uint32(rn)<<5|27,
|
|
msub|uint32(rm)<<16|uint32(rn)<<10|27<<5|uint32(rd),
|
|
), nil
|
|
}
|
|
|
|
// arm64PairSize returns the encoded size of a load/store pair instruction,
|
|
// mirroring the encoder's branch order exactly: the label offsets of pass 1
|
|
// must match the bytes pass 2 lays down.
|
|
func arm64PairSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
|
if len(ops) != 2 {
|
|
return 4
|
|
}
|
|
load := strings.Contains(mnem, "LDP")
|
|
memOp := ops[0]
|
|
if !load {
|
|
memOp = ops[1]
|
|
}
|
|
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" {
|
|
return 12 // ADRP + ADD + pair
|
|
}
|
|
_, off := arm64MemWithFrame(memOp, fi)
|
|
scale := arm64PairScale(mnem)
|
|
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
|
|
return 4
|
|
}
|
|
if off >= -4095 && off <= 4095 {
|
|
return 8 // ADD/SUB the whole offset into REGTMP + pair
|
|
}
|
|
return 12 // the two-ADD split within 16 MiB, or the pool sequence
|
|
}
|
|
|
|
// arm64MovSize returns the encoded size of a MOV instruction.
|
|
func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
|
if len(ops) != 2 {
|
|
return 4
|
|
}
|
|
src, dst := ops[0], ops[1]
|
|
switch {
|
|
case isImmOperand(src):
|
|
if src.Imm.Sym != nil && src.Imm.Sym.Pseudo == "SB" {
|
|
return 8 // ADRP + ADD
|
|
}
|
|
// The frame-relative immediate address: one ADD/SUB imm12 word in
|
|
// the addcon band, the hi<<12 plus lo pair in the 24-bit band, and
|
|
// an error (so an irrelevant size) for the pool form beyond either.
|
|
if src.Imm.Sym != nil && (src.Imm.Sym.Pseudo == "FP" || src.Imm.Sym.Pseudo == "SP") {
|
|
v := arm64FrameAddrValue(src.Imm.Sym, fi)
|
|
switch {
|
|
case mnem != "MOVD" && mnem != "MOV",
|
|
!arm64IsAddcon(v) && !arm64IsAddcon(-v) && (v < 0 || v > 0xFFFFFF):
|
|
return 4
|
|
case arm64IsAddcon(v) || arm64IsAddcon(-v):
|
|
return 4
|
|
default:
|
|
return 8
|
|
}
|
|
}
|
|
// The con(register) form lowers to the toolchain's ADD/SUB chain:
|
|
// one word in the addcon band, two in the 24-bit band, and the two
|
|
// pool words (LDR X plus the UXTX add) beyond it.
|
|
if _, ok := arm64ImmWithBase(src); ok {
|
|
con, _ := arm64ImmOperandValue(src)
|
|
return arm64ConRnSize(con)
|
|
}
|
|
if isMemOperand(dst) {
|
|
// Only the $0 (ZR store) immediate reaches memory, in one word.
|
|
return 4
|
|
}
|
|
// Size the immediate exactly as the encoder will emit it: multi-chunk
|
|
// values expand to up to four words and the W forms truncate first.
|
|
// Anything else would desynchronise the label offsets of pass 1 from
|
|
// the bytes pass 2 lays down, corrupting every later branch.
|
|
b, err := encodeARM64LoadImmClass(31, arm64Imm64(src), mnem, !strings.EqualFold(operandRegName(dst), "ZR"))
|
|
if err != nil {
|
|
return 4
|
|
}
|
|
return len(b)
|
|
case src.Addr.Sym != nil && src.Addr.Sym.Pseudo == "SB":
|
|
if mnem == "FMOVQ" {
|
|
return 12 // ADRP + ADD + LDR: the unaligned-access fallback
|
|
}
|
|
if fi.tls != nil && fi.tls[src.Addr.Sym.Name] {
|
|
return 4 // the TLS-LE load is one MOVZ word
|
|
}
|
|
return 8 // ADRP + LDR
|
|
case dst.Addr.Sym != nil && dst.Addr.Sym.Pseudo == "SB":
|
|
if mnem == "FMOVQ" {
|
|
return 12 // ADRP + ADD + STR: the unaligned-access fallback
|
|
}
|
|
return 8 // ADRP + STR
|
|
case isMemOperand(src) || isMemOperand(dst):
|
|
mem := src
|
|
if !isMemOperand(src) {
|
|
mem = dst
|
|
}
|
|
_, off := arm64MemWithFrame(mem, fi)
|
|
// Scaled unsigned offset fits if aligned and in range.
|
|
lt, ok := a64LoadTable[mnem]
|
|
if !ok {
|
|
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
|
|
}
|
|
scale := a64LSScale(lt)
|
|
if off >= 0 && off%scale == 0 && off/scale < 4096 {
|
|
return 4
|
|
}
|
|
if off >= -256 && off <= 255 {
|
|
return 4 // unscaled
|
|
}
|
|
if off >= -4095 && off <= 4095 {
|
|
return 8 // ADD/SUB the whole offset into REGTMP
|
|
}
|
|
if arm64OffsetSplitReach(off, lt) {
|
|
if _, _, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale))); ok {
|
|
return 8 // ADD base, REGTMP + access
|
|
}
|
|
}
|
|
return 8 // the pool sequence: LDR literal + register-offset access
|
|
default:
|
|
return 4 // register move
|
|
}
|
|
}
|
|
|
|
// encodeARM64LoadImm loads an immediate into a register, matching the
|
|
// toolchain's MOVZ/MOVN/MOVK sequence. W forms truncate to 32 bits first and
|
|
// every classification (movcon, complement, chunk count) runs on the truncated
|
|
// value, so a 32-bit immediate never reaches the 64-bit halves: MOVW $-1
|
|
// truncates to 0xFFFFFFFF, whose complement is a single zero chunk, and encodes
|
|
// as MOVN W, #0.
|
|
func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
|
return encodeARM64LoadImmClass(rd, v, mnem, true)
|
|
}
|
|
|
|
// encodeARM64LoadImmClass is encodeARM64LoadImm with the classification order
|
|
// in hand: bitmaskOK false skips the logical-immediate paths, which the
|
|
// toolchain's omovconst only takes for a real register (rt != REGZERO); an
|
|
// immediate to ZR rides the MOVZ/MOVN sequence carrying the value.
|
|
func encodeARM64LoadImmClass(rd int, v int64, mnem string, bitmaskOK bool) ([]byte, error) {
|
|
d := v
|
|
sf := uint32(1) // 64-bit
|
|
if mnem == "MOVW" || mnem == "MOVWU" {
|
|
d = int64(uint32(v))
|
|
sf = 0
|
|
}
|
|
|
|
if d == 0 {
|
|
// ORR Rd, ZR, ZR (MOV $0, Rd)
|
|
op := uint32(1<<31 | 1<<29 | 0x0a<<24) // ORR 64-bit
|
|
if sf == 0 {
|
|
op = 0<<31 | 1<<29 | 0x0a<<24 // ORR 32-bit
|
|
}
|
|
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// The Go toolchain classifies immediates (asm7.go conclass):
|
|
// - inside the imm12/shifted-imm12 "addcon" band (C_ABCON0/C_ABCON,
|
|
// 0 < v ≤ 4095 or a 4096 multiple up to 0xFFF000): bitmask first, so
|
|
// `MOVD $4096, R27` is ORR $4096, not MOVZ $(1<<12)
|
|
// - outside that band: MOVZ/MOVN first (C_MOVCON before C_BITCON), and
|
|
// negative values reach MOVN before the bitmask test
|
|
tryBitmaskFirst := bitmaskOK && d > 0 && (d <= 0xFFF || (d&0xFFF == 0 && d <= 0xFFF000))
|
|
|
|
if tryBitmaskFirst {
|
|
// Addcon-band immediate: try bitmask first (Go uses ORR for values
|
|
// like $1, $256 and $65536).
|
|
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
|
if ok {
|
|
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
|
}
|
|
}
|
|
|
|
// Try MOVZ (single non-zero 16-bit chunk) and MOVN (single non-0xFFFF
|
|
// chunk of the complement). The W forms must look inside the 32-bit
|
|
// window only, so the complement is masked to the operand width; d is
|
|
// already truncated and needs no mask.
|
|
width := uint64(0xFFFFFFFF)
|
|
if sf == 1 {
|
|
width = 0xFFFFFFFFFFFFFFFF
|
|
}
|
|
s := arm64Movcon(d)
|
|
if s >= 0 {
|
|
return a64wordLE(a64MoveWide(sf, 2, uint32(s>>4), uint32((d>>uint(s))&0xFFFF), uint32(rd))), nil
|
|
}
|
|
sn := arm64Movcon(^d & int64(width))
|
|
if sn >= 0 {
|
|
return a64wordLE(a64MoveWide(sf, 0, uint32(sn>>4), uint32(((^d)>>uint(sn))&0xFFFF), uint32(rd))), nil
|
|
}
|
|
|
|
// For values outside the bitmask-first range that are not movcon: try bitmask.
|
|
if !tryBitmaskFirst && bitmaskOK {
|
|
N, immr, imms, ok := arm64Bitmask(uint64(d), int(sf))
|
|
if ok {
|
|
return a64wordLE(sf<<31 | 1<<29 | 0x24<<23 | N<<22 | immr<<16 | imms<<10 | 31<<5 | uint32(rd)), nil
|
|
}
|
|
}
|
|
|
|
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
|
|
// for the first special 16-bit chunk, then MOVK per remaining one, with
|
|
// the bitmask-plus-fixup shortcut for strength-reduced constants).
|
|
ws := arm64MovLConst(d, sf)
|
|
for i := range ws {
|
|
ws[i] |= uint32(rd)
|
|
}
|
|
return a64WordsLE(ws...), nil
|
|
}
|
|
|
|
// arm64Bitmask checks whether a value can be encoded as an AArch64 logical
|
|
// immediate (bitmask). Returns the N, immr, imms fields and true if
|
|
// representable. sf is 0 for 32-bit or 1 for 64-bit.
|
|
func arm64Bitmask(v uint64, sf int) (N, immr, imms uint32, ok bool) {
|
|
if v == 0 {
|
|
return
|
|
}
|
|
maxElem := uint(6) // 2^6 = 64
|
|
if sf == 0 {
|
|
maxElem = 5 // 2^5 = 32
|
|
v &= 0xFFFFFFFF
|
|
}
|
|
|
|
for e := uint(0); e < maxElem; e++ {
|
|
esize := uint(1) << (e + 1) // 2, 4, 8, 16, 32, 64
|
|
emask := uint64(1<<esize) - 1
|
|
pattern := v & emask
|
|
if pattern == 0 {
|
|
continue
|
|
}
|
|
|
|
// Check each rotation: is the rotated pattern a contiguous block of 1s at the LSB?
|
|
for r := range esize {
|
|
rotated := (pattern >> r) | ((pattern << (esize - r)) & emask)
|
|
if rotated == 0 {
|
|
continue
|
|
}
|
|
// Count trailing 1s (contiguous block of 1s from bit 0).
|
|
tz := uint(0)
|
|
tmp := ^rotated
|
|
for tmp&1 == 0 && tz < esize {
|
|
tz++
|
|
tmp >>= 1
|
|
}
|
|
if tz == 0 || tz >= esize {
|
|
continue
|
|
}
|
|
mask := uint64(1<<tz) - 1
|
|
if rotated != mask {
|
|
continue
|
|
}
|
|
ones := tz
|
|
|
|
// Verify the pattern repeats to fill the register.
|
|
full := uint64(0)
|
|
for i := uint(0); i < 64/esize; i++ {
|
|
full |= pattern << (i * esize)
|
|
}
|
|
if sf == 0 {
|
|
full &= 0xFFFFFFFF
|
|
}
|
|
if full != v {
|
|
continue
|
|
}
|
|
|
|
// Encode N, immr, imms.
|
|
if esize == 64 && sf == 1 {
|
|
N = 1
|
|
} else {
|
|
N = 0
|
|
}
|
|
// The period marker rides the high imms bits one position above
|
|
// the element size: zero for 32 and 64 (N and the absence carry
|
|
// them), then 0x20, 0x30, 0x38 and 0x3C for 16, 8, 4 and 2. The
|
|
// toolchain's bitconEncode computes it as 63 & ^(period*2 - 1).
|
|
imms = ^uint32(uint32(esize*2-1))&0x3F | uint32(ones-1)
|
|
immr = uint32((esize - r) % esize)
|
|
return N, immr, imms, true
|
|
}
|
|
}
|
|
return
|
|
}
|
|
|
|
// encodeARM64RegMove encodes a register-to-register move.
|
|
// Integer → integer: ORR Rd, ZR, Rs.
|
|
// FP → FP: FMOV Fd, Fn (FP data processing).
|
|
// FP ↔ GP: FMOV general (FPCVTI encoding).
|
|
// Go Plan 9 syntax is source first, destination last: MOV src, dst.
|
|
func encodeARM64RegMove(mnem string, src, dst *ast.Operand) ([]byte, error) {
|
|
rs := arm64RegNum(operandRegName(src))
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rs < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
sc := arm64RegClassOf(operandRegName(src))
|
|
dc := arm64RegClassOf(operandRegName(dst))
|
|
|
|
// FP → FP: FMOV Fd, Fn (FP data processing unary form).
|
|
if sc == arm64ClsFP && dc == arm64ClsFP {
|
|
typ := uint32(1) // 64-bit double
|
|
if mnem == "FMOVS" {
|
|
typ = 0 // 32-bit float
|
|
}
|
|
// FPOP1S encoding: 0x1E204000 | type<<22 | Rn<<5 | Rd
|
|
return a64wordLE(0x1E<<24 | typ<<22 | 1<<21 | 0x10<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// GP ↔ FP: FMOV general (FPCVTI encoding).
|
|
// Go syntax: FMOV GPsrc, FPdst or FMOV FPsrc, GPdst, source first.
|
|
if sc == arm64ClsFP && dc == arm64ClsGR {
|
|
// FP → GP: FMOV Wd/Xd, Sn/Dn. opcode bits[20:16]=6.
|
|
sf, typ := uint32(0), uint32(0)
|
|
if mnem == "FMOVD" {
|
|
sf, typ = 1, 1
|
|
}
|
|
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 6<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
|
}
|
|
if sc == arm64ClsGR && dc == arm64ClsFP {
|
|
// GP → FP: FMOV Vd, Wn/Xn. opcode bits[20:16]=7.
|
|
sf, typ := uint32(0), uint32(0)
|
|
if mnem == "FMOVD" {
|
|
sf, typ = 1, 1
|
|
}
|
|
return a64wordLE(sf<<31 | 0x1E<<24 | typ<<22 | 1<<21 | 7<<16 | uint32(rs)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// Integer → integer. Every truncating register move lowers to an
|
|
// extend in the toolchain (asm7.go case 45): the signed forms to SBFM
|
|
// (SXTB, SXTH, SXTW), the unsigned byte and halfword forms to UBFM
|
|
// (UXTB, UXTH), and only MOVWU to an ORR against WZR. MOVD stays
|
|
// ORR Xd, XZR, Xm.
|
|
//
|
|
// The SP register moves ride the ADD (immediate) form instead: ORR
|
|
// cannot address SP and register 31 encodes ZR there (asm7.go case
|
|
// 24), whose C_RSP row exists for MOVD alone, so every other width is
|
|
// an illegal combination.
|
|
if sn, dn := operandRegName(src), operandRegName(dst); sn == "RSP" || sn == "SP" || dn == "RSP" || dn == "SP" {
|
|
if mnem != "MOVD" && mnem != "MOV" {
|
|
return nil, fmt.Errorf("%s: illegal combination: the SP register move exists for MOVD only", mnem)
|
|
}
|
|
return a64wordLE(a64AddSub(1, 0, 0, 0, 0, uint32(rs), uint32(rd))), nil
|
|
}
|
|
if rs != 31 {
|
|
switch mnem {
|
|
case "MOVB":
|
|
return a64wordLE(0x93400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
|
case "MOVH":
|
|
return a64wordLE(0x93400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
|
case "MOVW":
|
|
return a64wordLE(0x93400000 | 31<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
|
case "MOVBU":
|
|
return a64wordLE(0xd3400000 | 7<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
|
case "MOVHU":
|
|
return a64wordLE(0xd3400000 | 15<<10 | uint32(rs)<<5 | uint32(rd)), nil
|
|
}
|
|
}
|
|
sf := uint32(1) // 64-bit
|
|
if mnem == "MOVWU" {
|
|
sf = 0
|
|
}
|
|
// A narrow move out of the zero register loses its width: the
|
|
// toolchain rewrites it as MOVWU (asm7.go case 45), an ORR against
|
|
// WZR. MOVD and MOV keep the 64-bit form.
|
|
if rs == 31 && mnem != "MOVD" && mnem != "MOV" {
|
|
sf = 0
|
|
}
|
|
op := uint32(1<<29 | 0x0a<<24) // ORR
|
|
return a64wordLE(sf<<31 | op | uint32(rs)<<16 | 31<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64MemOp encodes a memory load or store with offset.
|
|
// encodeARM64MemOp encodes a MOV-family load/store. With wb set ("P" post
|
|
// or "W" pre) the offset is the signed writeback amount applied to the base
|
|
// register after (post) or before (pre) the access; the offset must fit the
|
|
// unscaled 9-bit field and the base must be a real register, since a pseudo
|
|
// frame base cannot be written back.
|
|
func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm64FrameInfo, wb string, pool *arm64Pool, poolBase int, pc int) ([]byte, error) {
|
|
rn, off := arm64MemWithFrame(mem, fi)
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("invalid memory operand")
|
|
}
|
|
// The MOV family addresses memory through a general register only.
|
|
if mem.Addr.Base != "" && arm64RegClassOf(mem.Addr.Base) == arm64ClsFP {
|
|
return nil, fmt.Errorf("%s: illegal combination: the base register cannot be FP", mnem)
|
|
}
|
|
// Writeback with the base doubling as the data register is constrained
|
|
// unpredictable.
|
|
if wb != "" && rn == reg && rn != 31 {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the base rides the data register", mnem)
|
|
}
|
|
lt, ok := a64LoadTable[mnem]
|
|
if !ok {
|
|
// MOV defaults to MOVD (64-bit load/store).
|
|
lt = a64LoadTable["MOVD"]
|
|
}
|
|
|
|
// Register-offset addressing: (Rn)(Rm), (Rn)(Rm<<k), (Rn)(Rm.UXTW) and
|
|
// friends. The toolchain accepts the LSL (UXTX), UXTW, SXTW and SXTX
|
|
// options with a shift amount of either zero or the access size's log2,
|
|
// rejects every other spelling, and rejects the whole form on FMOVQ.
|
|
if mem.Addr.Index != "" {
|
|
return encodeARM64RegOffset(mnem, mem, rn, reg, load, lt, wb)
|
|
}
|
|
|
|
scale := a64LSScale(lt)
|
|
storeOpc := a64StoreOpc(lt)
|
|
var opc int
|
|
if load {
|
|
opc = lt.opc
|
|
} else {
|
|
opc = storeOpc
|
|
}
|
|
|
|
if wb != "" {
|
|
if mem.Addr.Sym != nil && mem.Addr.Sym.Pseudo != "" {
|
|
return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem)
|
|
}
|
|
if off < -256 || off > 255 {
|
|
return nil, fmt.Errorf("%s: writeback offset %d out of range (-256..255)", mnem, off)
|
|
}
|
|
w := a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)
|
|
if wb == "P" {
|
|
w |= 1 << 10 // post-index
|
|
} else {
|
|
w |= 3 << 10 // pre-index
|
|
}
|
|
return a64wordLE(w), nil
|
|
}
|
|
|
|
// Scaled unsigned offset first, then the unscaled ±255 form.
|
|
if off >= 0 && off%scale == 0 && off/scale < 4096 {
|
|
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
|
|
}
|
|
if off >= -256 && off <= 255 {
|
|
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil
|
|
}
|
|
// Offsets within ±4095 that no single-instruction form carries ride the
|
|
// toolchain's ADD/SUB fallback: the whole offset moves into REGTMP and
|
|
// the access reads it back at zero (asm7.go cases 30/31).
|
|
if off >= -4095 && off <= 4095 {
|
|
op, v := uint32(0), off // ADD
|
|
if v < 0 {
|
|
op, v = 1, -v // SUB
|
|
}
|
|
return a64WordsLE(
|
|
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
|
|
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), 0, 27, uint32(reg)),
|
|
), nil
|
|
}
|
|
// Large offset: materialise the base in REGTMP (R27) the way the
|
|
// toolchain does and access what remains (asm7.go cases 30/31). The
|
|
// ADD offsets from the operand's own base register, [SP] and [Rn]
|
|
// alike; beyond the split band the offset reaches the literal pool.
|
|
if !arm64OffsetSplitReach(off, lt) {
|
|
return arm64PoolAccess(mnem, lt, opc, off, rn, reg, pc, pool, poolBase, load)
|
|
}
|
|
hi, lo, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale)))
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
|
}
|
|
return a64WordsLE(
|
|
arm64AddImmWord(hi, uint32(rn), 27), // ADD $hi[<<12], Rn, R27
|
|
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(lo), 27, uint32(reg)),
|
|
), nil
|
|
}
|
|
|
|
// arm64PoolAccess lowers an offset beyond the split band the way the
|
|
// toolchain's pool branch does: a PC-relative literal load of the offset
|
|
// into REGTMP, then the access against the register pair (asm7.go cases
|
|
// 30/31 and omovlit). The pooled words sit at poolBase plus the entry's
|
|
// offset, both function-relative, so the imm19 distance resolves here.
|
|
func arm64PoolAccess(mnem string, lt a64LSType, opc int, off int64, rn, reg, pc int, pool *arm64Pool, poolBase int, load bool) ([]byte, error) {
|
|
// The toolchain refuses a REGTMP data register or base on the pool path
|
|
// (asm7.go cases 30/31): the literal load itself rides REGTMP.
|
|
if reg == 27 || rn == 27 {
|
|
kind := "load"
|
|
if !load {
|
|
kind = "store"
|
|
}
|
|
return nil, fmt.Errorf("%s: REGTMP used in large offset %s", mnem, kind)
|
|
}
|
|
if pool == nil {
|
|
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
|
}
|
|
entryOff, w := pool.add(off)
|
|
dist := (poolBase + entryOff - pc) >> 2
|
|
// The plan pass probes before the segments' bases are final, so its
|
|
// distances carry no meaning.
|
|
if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) {
|
|
return nil, fmt.Errorf("%s: literal pool %d out of 19-bit reach", mnem, dist<<2)
|
|
}
|
|
return a64WordsLE(
|
|
w<<30|3<<27|uint32(dist)&0x7FFFF<<5|27, // LDR R27, pool
|
|
a64LSReg(uint32(lt.size), uint32(lt.V), uint32(opc), 27, uint32(rn), uint32(reg)),
|
|
), nil
|
|
}
|
|
|
|
// a64LSReg encodes a load/store register (register offset) with the LSL-0
|
|
// option the toolchain's register-offset accesses use:
|
|
// size<<30 | 0x38<<24 | V<<26 | opc<<22 | 1<<21 | Rm<<16 | 011<<13 | 10<<10 | Rn<<5 | Rt.
|
|
func a64LSReg(size, V, opc, rm, rn, rt uint32) uint32 {
|
|
return a64LSRegExt(size, V, opc, rm, rn, rt, 3, 0)
|
|
}
|
|
|
|
// a64LSRegExt encodes the extended-register-offset load/store:
|
|
// size<<30 | 0x38<<24 | V<<26 | opc<<22 | 1<<21 | Rm<<16 | option<<13 |
|
|
// S<<12 | 10<<10 | Rn<<5 | Rt, with S meaning "scale by the access size".
|
|
func a64LSRegExt(size, V, opc, rm, rn, rt, option, s uint32) uint32 {
|
|
return size<<30 | 0x38<<24 | V<<26 | opc<<22 | 1<<21 | rm<<16 | option<<13 | s<<12 | 2<<10 | rn<<5 | rt
|
|
}
|
|
|
|
// encodeARM64RegOffset encodes the register-offset addressing forms of the
|
|
// MOV family: (Rn)(Rm), (Rn)(Rm*1), (Rn)(Rm<<k) and the extend spellings
|
|
// (Rn)(Rm.UXTW<<k), (Rn)(Rm.SXTW), (Rn)(Rm.SXTX<<k). The toolchain's
|
|
// contract, probed against `go tool asm`: the four options UXTX/LSL, UXTW,
|
|
// SXTW and SXTX exist (the spellings UXTX, UXTB, SXTH and friends are
|
|
// rejected), the shift amount is either zero or the access size's log2, FMOVQ
|
|
// carries no register-offset form at all, and writeback does not combine with
|
|
// an index.
|
|
func encodeARM64RegOffset(mnem string, mem *ast.Operand, rn, reg int, load bool, lt a64LSType, wb string) ([]byte, error) {
|
|
if wb != "" {
|
|
return nil, fmt.Errorf("%s: writeback does not combine with a register offset", mnem)
|
|
}
|
|
if mnem == "FMOVQ" {
|
|
return nil, fmt.Errorf("%s: illegal combination: the register offset form does not exist", mnem)
|
|
}
|
|
if mem.Addr.Scale > 1 {
|
|
return nil, fmt.Errorf("%s: the register offset addressing mode takes no scale factor", mnem)
|
|
}
|
|
name, ext, _ := strings.Cut(mem.Addr.Index, ".")
|
|
var option uint32
|
|
switch strings.ToUpper(ext) {
|
|
case "":
|
|
option = 3 // UXTX / LSL
|
|
case "UXTW":
|
|
option = 2
|
|
case "SXTW":
|
|
option = 6
|
|
case "SXTX":
|
|
option = 7
|
|
default:
|
|
return nil, fmt.Errorf("%s: invalid shift for the register offset addressing mode: %s", mnem, ext)
|
|
}
|
|
rm := arm64RegNum(name)
|
|
if rm < 0 {
|
|
return nil, fmt.Errorf("%s: invalid index register %q", mnem, name)
|
|
}
|
|
amount := 0
|
|
shift := strings.Join(strings.Fields(mem.Addr.Shift), "")
|
|
shift = strings.TrimSuffix(shift, ")")
|
|
if shift != "" {
|
|
if !strings.HasPrefix(shift, "<<") {
|
|
return nil, fmt.Errorf("%s: invalid shift for the register offset addressing mode: %s", mnem, shift)
|
|
}
|
|
v, err := strconv.Atoi(strings.TrimPrefix(shift, "<<"))
|
|
if err != nil {
|
|
return nil, fmt.Errorf("%s: invalid shift for the register offset addressing mode: %s", mnem, shift)
|
|
}
|
|
amount = v
|
|
}
|
|
log2 := bits.TrailingZeros64(uint64(a64LSScale(lt)))
|
|
if amount != 0 && amount != log2 {
|
|
return nil, fmt.Errorf("%s: invalid index shift amount %d", mnem, amount)
|
|
}
|
|
s := uint32(0)
|
|
if amount == log2 && log2 > 0 {
|
|
s = 1
|
|
}
|
|
opc := lt.opc
|
|
if !load {
|
|
opc = a64StoreOpc(lt)
|
|
}
|
|
return a64wordLE(a64LSRegExt(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(rm), uint32(rn), uint32(reg), option, s)), nil
|
|
}
|
|
|
|
// arm64OffsetSplitReach reports whether an offset stays within the band the
|
|
// toolchain lowers to ADD plus access instead of pooling (loadStoreClass):
|
|
// ±4095 for every width, then the width's own 24-bit band, aligned to the
|
|
// access size or under its small unaligned ceiling. Byte accesses take the
|
|
// full 24 bits with no alignment clause; the Q width, whose size lives in
|
|
// opc, carries the widest band.
|
|
func arm64OffsetSplitReach(off int64, lt a64LSType) bool {
|
|
if off >= -4095 && off <= 4095 {
|
|
return true
|
|
}
|
|
if off < 0 {
|
|
return false
|
|
}
|
|
if lt.size == 0 && lt.V == 1 { // the Q width
|
|
return off <= 0xfff000+0xfff<<4 && off&15 == 0 || off <= 0xfff+0xfff<<4
|
|
}
|
|
switch lt.size {
|
|
case 0: // byte: the full 24-bit band, alignment trivial
|
|
return off <= 0xffffff
|
|
case 1:
|
|
return off <= 0xfff000+0xfff<<1 && off&1 == 0 || off <= 0xfff+0xfff<<1
|
|
case 2:
|
|
return off <= 0xfff000+0xfff<<2 && off&3 == 0 || off <= 0xfff+0xfff<<2
|
|
default:
|
|
return off <= 0xfff000+0xfff<<3 && off&7 == 0 || off <= 0xfff+0xfff<<3
|
|
}
|
|
}
|
|
|
|
// arm64AddImmWord encodes ADD $v, Rn, Rd the way the toolchain's oaddi
|
|
// does: a non-zero multiple of 0x1000 encodes shifted left by twelve.
|
|
func arm64AddImmWord(v int64, rn, rd uint32) uint32 {
|
|
return arm64AddSubImmWord(0, v, rn, rd)
|
|
}
|
|
|
|
// arm64AddSubImmWord encodes ADD (op 0) or SUB (op 1) $v, Rn, Rd the way the
|
|
// toolchain's oaddi does: a non-zero multiple of 0x1000 encodes shifted left
|
|
// by twelve.
|
|
func arm64AddSubImmWord(op uint32, v int64, rn, rd uint32) uint32 {
|
|
sh := arm64AddShift(v)
|
|
if sh == 1 {
|
|
v >>= 12
|
|
}
|
|
return a64AddSub(1, op, 0, sh, uint32(v), rn, rd)
|
|
}
|
|
|
|
// arm64FloatOperandValue recovers a floating-point immediate: the bare
|
|
// spelling fills Imm.Float and the parenthesised one stays in the raw text
|
|
// ($ ( 4.0 )), so the token shape comes off the raw string. An integer zero
|
|
// rides Imm.HasVal and counts (the toolchain's AFMOVD zero row), any other
|
|
// integer immediate does not. ok is false for every other shape.
|
|
func arm64FloatOperandValue(op *ast.Operand) (float64, bool) {
|
|
if op.Imm.Float != "" {
|
|
f, err := strconv.ParseFloat(op.Imm.Float, 64)
|
|
if err != nil {
|
|
return 0, false
|
|
}
|
|
if op.Imm.Neg {
|
|
f = -f
|
|
}
|
|
return f, true
|
|
}
|
|
s := strings.Join(strings.Fields(op.Raw), "")
|
|
s = strings.TrimPrefix(s, "$")
|
|
if strings.HasPrefix(s, "(") && strings.HasSuffix(s, ")") {
|
|
inner := s[1 : len(s)-1]
|
|
if strings.ContainsAny(inner, ".eE") {
|
|
f, err := strconv.ParseFloat(inner, 64)
|
|
if err != nil {
|
|
return 0, false
|
|
}
|
|
return f, true
|
|
}
|
|
}
|
|
if op.Imm.HasVal && op.Imm.Val == 0 && !op.Imm.Neg {
|
|
return 0, true
|
|
}
|
|
return 0, false
|
|
}
|
|
|
|
// arm64ChipFloat ports the toolchain's chipfloat7: the 8-bit FMOV immediate
|
|
// field for a double whose low 48 mantissa bits are zero and whose exponent
|
|
// sits in the -3..4 band, with bit 7 the sign, bit 6 the negative-exponent
|
|
// flag and bits 4:0 the exponent tail plus mantissa top. The toolchain
|
|
// takes the encoded value only when this reports above zero; zero itself
|
|
// pools (or moves from ZR), so 0 is not a valid encoding here either.
|
|
func arm64ChipFloat(e float64) int {
|
|
ei := math.Float64bits(e)
|
|
l := uint32(ei)
|
|
h := uint32(ei >> 32)
|
|
if l != 0 || h&0xffff != 0 {
|
|
return -1
|
|
}
|
|
h1 := h & 0x7fc00000
|
|
if h1 != 0x40000000 && h1 != 0x3fc00000 {
|
|
return -1
|
|
}
|
|
n := 0
|
|
if h&0x80000000 != 0 {
|
|
n |= 1 << 7
|
|
}
|
|
if h1 == 0x3fc00000 {
|
|
n |= 1 << 6
|
|
}
|
|
n |= int((h >> 16) & 0x3f)
|
|
return n
|
|
}
|
|
|
|
// arm64ImmWithBase reports whether an immediate operand spells the
|
|
// con(register) form, $con(REG): the parser leaves it unstructured (an
|
|
// immediate whose raw spelling carries the parenthesised register), so the
|
|
// base register comes off the raw text while the constant rides the parsed
|
|
// immediate. ok is false for every other shape.
|
|
func arm64ImmWithBase(op *ast.Operand) (rn int, ok bool) {
|
|
if op.Kind != 0 || op.Imm.Sym != nil {
|
|
return 0, false
|
|
}
|
|
s := strings.Join(strings.Fields(op.Raw), " ")
|
|
if !strings.HasSuffix(s, ")") {
|
|
return 0, false
|
|
}
|
|
i := strings.LastIndex(s, "(")
|
|
if i < 0 {
|
|
return 0, false
|
|
}
|
|
reg := strings.TrimSpace(s[i+1 : len(s)-1])
|
|
rn = arm64RegNum(reg)
|
|
if rn < 0 {
|
|
return 0, false
|
|
}
|
|
return rn, true
|
|
}
|
|
|
|
// arm64ConRnSize sizes the con(register) lowering of encodeARM64ConRn
|
|
// without touching the pool: the bands mirror the encoder exactly.
|
|
func arm64ConRnSize(con int64) int {
|
|
a := con
|
|
if a < 0 {
|
|
a = -a
|
|
}
|
|
if a <= 0xfff || (a&0xfff == 0 && a <= 0xfff000) {
|
|
return 4
|
|
}
|
|
if con >= 0 && a <= 0xffffff {
|
|
return 8
|
|
}
|
|
return 8 // LDR X, pool + the UXTX add
|
|
}
|
|
|
|
// encodeARM64ConRn lowers MOVD $con(Rn), Rd the way the toolchain's case 4
|
|
// does: a single ADD/SUB immediate inside the addcon band (±4095 or a
|
|
// multiple of 4096 up to 0xfff<<12), the hi<<12 plus lo pair for a positive
|
|
// displacement inside the 24-bit band, and the literal pool plus a UXTX add
|
|
// beyond either (a negative displacement outside the addcon band pools
|
|
// straight away: isaddcon2 only takes non-negative values).
|
|
func encodeARM64ConRn(rn int, con int64, rd int, pool *arm64Pool, poolBase, pc int) ([]byte, error) {
|
|
op := uint32(0) // ADD
|
|
a := con
|
|
if a < 0 {
|
|
op = 1 // SUB
|
|
a = -a
|
|
}
|
|
if a <= 0xfff || (a&0xfff == 0 && a <= 0xfff000) {
|
|
return a64wordLE(arm64AddSubImmWord(op, a, uint32(rn), uint32(rd))), nil
|
|
}
|
|
if con >= 0 && a <= 0xffffff {
|
|
hi := a & 0xfff000
|
|
lo := a & 0xfff
|
|
return a64WordsLE(
|
|
arm64AddSubImmWord(op, hi, uint32(rn), uint32(rd)),
|
|
arm64AddSubImmWord(op, lo, uint32(rd), uint32(rd)),
|
|
), nil
|
|
}
|
|
// Beyond the bands the toolchain pools the displacement (a full 64-bit
|
|
// slot) and adds it back with a UXTX-extended register add (case 34).
|
|
if pool == nil {
|
|
return nil, fmt.Errorf("MOVD $%d(R%d): displacement out of range (literal pool not supported)", con, rn)
|
|
}
|
|
entryOff, w := pool.add64(con)
|
|
dist := (poolBase + entryOff - pc) >> 2
|
|
if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) {
|
|
return nil, fmt.Errorf("MOVD $%d(R%d): literal pool %d out of 19-bit reach", con, rn, dist<<2)
|
|
}
|
|
return a64WordsLE(
|
|
w<<30|3<<27|uint32(dist)&0x7FFFF<<5|27, // LDR R27, pool
|
|
a64AddSubReg(1, 27, uint32(rn), uint32(rd)),
|
|
), nil
|
|
}
|
|
|
|
// ---- static symbol references (ADRP + offset) ----
|
|
|
|
// encodeARM64SBAddr emits ADRP Rd, 0; ADD Rd, Rd, 0 with the
|
|
// R_ADDRARM64 relocation pair, loading a symbol's address.
|
|
func encodeARM64SBAddr(sym *ast.Symbol, rd int, relocs *[]Reloc) []byte {
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
|
|
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, uint32(rd)), // ADRP Rd, 0
|
|
a64AddSub(1, 0, 0, 0, 0, uint32(rd), uint32(rd)), // ADD $0, Rd, Rd
|
|
)
|
|
}
|
|
|
|
// encodeARM64FrameAddr lowers MOVD $sym+off(FP) and its SP spelling: the
|
|
// address rides ADD/SUB from the hardware SP, one imm12 word inside the
|
|
// addcon band and the hi<<12 plus lo pair inside the 24-bit band (asm7.go
|
|
// optab case 4). Beyond either the toolchain pools the constant (case 34),
|
|
// and no pool the plain form reaches exists here, so that shape is refused.
|
|
// Every other width is an illegal combination: the toolchain's AACON rows
|
|
// exist for MOVD alone.
|
|
func encodeARM64FrameAddr(sym *ast.Symbol, dst *ast.Operand, mnem string, fi arm64FrameInfo) ([]byte, error) {
|
|
if mnem != "MOVD" && mnem != "MOV" {
|
|
return nil, fmt.Errorf("%s: illegal combination: the $frame-address form exists for MOVD only", mnem)
|
|
}
|
|
rd := arm64RegNum(operandRegName(dst))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("%s %s: invalid destination register", mnem, sym.Raw)
|
|
}
|
|
v := arm64FrameAddrValue(sym, fi)
|
|
if !arm64IsAddcon(v) && !arm64IsAddcon(-v) && (v < 0 || v > 0xFFFFFF) {
|
|
return nil, fmt.Errorf("%s: %s needs the literal pool, which no arm64 pool reaches here", mnem, sym.Raw)
|
|
}
|
|
return a64WordsLE(arm64FrameAddrWords(v, rd)...), nil
|
|
}
|
|
|
|
// encodeARM64SBLoad emits ADRP R27, 0; LDR Rd, [R27, 0] with relocations,
|
|
// matching the toolchain: the scratch register is REGTMP (R27) and the pair
|
|
// carries R_ARM64_PCREL_LDST64. FMOVQ has no LDST relocation width, so it
|
|
// takes the toolchain's unaligned-access fallback instead (asm7.go case 65):
|
|
// ADRP R27, 0; ADD R27, R27, 0; LDR with the R_ADDRARM64 pair.
|
|
func encodeARM64SBLoad(sym *ast.Symbol, rd int, mnem string, relocs *[]Reloc) ([]byte, error) {
|
|
lt, ok := a64LoadTable[mnem]
|
|
if !ok {
|
|
lt = a64LoadTable["MOVD"]
|
|
}
|
|
if mnem == "FMOVQ" {
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
|
|
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, 27), // ADRP R27, 0
|
|
a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27
|
|
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, (R27)
|
|
), nil
|
|
}
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, 27), // ADRP R27, 0
|
|
a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), 0, 27, uint32(rd)), // LDR Rd, [R27, #0]
|
|
), nil
|
|
}
|
|
|
|
// encodeARM64SBStore emits ADRP R27, 0; STR Rs, [R27, 0] with relocations,
|
|
// matching the toolchain's R27 scratch and R_ARM64_PCREL_LDST64 pair. FMOVQ
|
|
// takes the twelve-byte ADRP + ADD + STR fallback with the R_ADDRARM64 pair,
|
|
// the store-side twin of the load above (asm7.go case 64).
|
|
func encodeARM64SBStore(sym *ast.Symbol, rs int, mnem string, relocs *[]Reloc) ([]byte, error) {
|
|
lt, ok := a64LoadTable[mnem]
|
|
if !ok {
|
|
lt = a64LoadTable["MOVD"]
|
|
}
|
|
storeOpc := a64StoreOpc(lt)
|
|
if mnem == "FMOVQ" {
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 0, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
|
|
Reloc{Off: 4, After: 4, Name: sym.Name, Kind: RelArm64Addr, Addend: sym.Offset},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, 27), // ADRP R27, 0
|
|
a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27
|
|
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, (R27)
|
|
), nil
|
|
}
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 8, Name: sym.Name, Kind: RelArm64LDST64, Addend: sym.Offset},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, 27), // ADRP R27, 0
|
|
a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), 0, 27, uint32(rs)), // STR Rs, [R27, #0]
|
|
), nil
|
|
}
|
|
|
|
// ---- operand helpers ----
|
|
|
|
// arm64Imm64 returns the full 64-bit immediate value of an operand, the
|
|
// raw-spelling-aware value arm64ImmOperandValue recovers.
|
|
func arm64Imm64(op *ast.Operand) int64 {
|
|
v, _ := arm64ImmOperandValue(op)
|
|
return v
|
|
}
|
|
|
|
// arm64ImmOperandValue returns the immediate an operand stands for, falling
|
|
// back to a raw evaluation for the spellings the parser leaves unevaluated:
|
|
// the one's-complement form $~n and parenthesised constant expressions.
|
|
// The second result reports whether a value could be recovered.
|
|
func arm64ImmOperandValue(op *ast.Operand) (int64, bool) {
|
|
if op.Imm.HasVal {
|
|
v := op.Imm.Val
|
|
if op.Imm.Neg {
|
|
v = -v
|
|
}
|
|
// The parser folds a leading literal and drops the arithmetic that
|
|
// trails it ($14*16 parses as 14), so an operator-bearing spelling
|
|
// re-evaluates in full: the raw text is the statement's truth.
|
|
s := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$"))
|
|
if s != "" && strings.ContainsAny(s, "+-*/^|") {
|
|
if e, ok := arm64EvalExpr(s); ok {
|
|
return e, true
|
|
}
|
|
}
|
|
return v, true
|
|
}
|
|
s := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$"))
|
|
inverted := false
|
|
if i := strings.IndexByte(s, '~'); i >= 0 {
|
|
inverted = true
|
|
s = s[i+1:]
|
|
}
|
|
v, ok := arm64EvalExpr(s)
|
|
if !ok {
|
|
return 0, false
|
|
}
|
|
if inverted {
|
|
v = ^v
|
|
}
|
|
return v, true
|
|
}
|
|
|
|
// arm64MemWithFrame resolves a memory operand, translating FP/SP pseudo-
|
|
// registers via the frame mapping. The offset stays 64-bit: the AST carries
|
|
// int64 displacements and truncating here would wrap offsets beyond 2^31
|
|
// silently.
|
|
func arm64MemWithFrame(op *ast.Operand, fi arm64FrameInfo) (rn int, off int64) {
|
|
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo != "" {
|
|
base, pseudo := arm64ResolvePseudo(op.Addr.Sym, fi)
|
|
return base, int64(pseudo)
|
|
}
|
|
if rn, off, ok := arm64ExprMem(op); ok {
|
|
return rn, off
|
|
}
|
|
return arm64RegNum(op.Addr.Base), op.Addr.Offset
|
|
}
|
|
|
|
// arm64IsMemOperand is the arm64-side memory test: the shared syntactic test
|
|
// plus the parenthesised-expression form (8*1)(RSP), which the parser leaves
|
|
// unstructured (empty base) because the offset is not a plain integer.
|
|
func arm64IsMemOperand(op *ast.Operand) bool {
|
|
if isMemOperand(op) {
|
|
return true
|
|
}
|
|
_, _, ok := arm64ExprMem(op)
|
|
return ok
|
|
}
|
|
|
|
// arm64ExprMem recovers a base register and an evaluated offset from a
|
|
// parenthesised-expression memory operand such as (8*22)(RSP) or (0*8)(R0).
|
|
// It reports ok=false for anything else.
|
|
func arm64ExprMem(op *ast.Operand) (rn int, off int64, ok bool) {
|
|
if op.Addr.Base != "" || op.Addr.Sym != nil {
|
|
return 0, 0, false
|
|
}
|
|
s := strings.Join(strings.Fields(op.Raw), " ")
|
|
if !strings.HasSuffix(s, ")") {
|
|
return 0, 0, false
|
|
}
|
|
// Split the trailing "( REG )" from the leading "( EXPR )".
|
|
inner := strings.LastIndex(s, "(")
|
|
if inner <= 0 {
|
|
return 0, 0, false
|
|
}
|
|
regPart := strings.TrimSpace(s[inner+1 : len(s)-1])
|
|
head := strings.TrimSpace(s[:inner])
|
|
if !strings.HasPrefix(head, "(") || !strings.HasSuffix(head, ")") {
|
|
return 0, 0, false
|
|
}
|
|
expr := strings.TrimSpace(head[1 : len(head)-1])
|
|
v, ok := arm64EvalExpr(expr)
|
|
if !ok {
|
|
return 0, 0, false
|
|
}
|
|
rn = arm64RegNum(regPart)
|
|
if rn < 0 {
|
|
return 0, 0, false
|
|
}
|
|
return rn, v, true
|
|
}
|
|
|
|
// arm64EvalExpr evaluates the Plan 9 constant arithmetic the assembler
|
|
// accepts inside memory operands: integers with unary minus and the + - * <<
|
|
// >> & | ^ operators. Operator precedence follows the Plan 9 convention
|
|
// (shifts bind tighter than +, * tighter than shifts); expressions it cannot
|
|
// fully reduce report ok=false.
|
|
func arm64EvalExpr(s string) (int64, bool) {
|
|
type parser struct {
|
|
toks []string
|
|
pos int
|
|
}
|
|
var scan func(string) []string
|
|
scan = func(s string) []string {
|
|
var out []string
|
|
for s = strings.TrimSpace(s); s != ""; s = strings.TrimSpace(s) {
|
|
switch {
|
|
case s[0] == '(' || s[0] == ')':
|
|
out = append(out, s[:1])
|
|
s = s[1:]
|
|
case s[0] >= '0' && s[0] <= '9':
|
|
i := 0
|
|
for i < len(s) && ((s[i] >= '0' && s[i] <= '9') || s[i] == 'x' || s[i] == 'X' ||
|
|
s[i] >= 'a' && s[i] <= 'f' || s[i] >= 'A' && s[i] <= 'F') {
|
|
i++
|
|
}
|
|
out = append(out, s[:i])
|
|
s = s[i:]
|
|
case strings.HasPrefix(s, "<<"), strings.HasPrefix(s, ">>"):
|
|
out = append(out, s[:2])
|
|
s = s[2:]
|
|
case strings.IndexByte("+-*&|^~", s[0]) >= 0:
|
|
out = append(out, s[:1])
|
|
s = s[1:]
|
|
default:
|
|
return nil
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
toks := scan(s)
|
|
if toks == nil {
|
|
return 0, false
|
|
}
|
|
p := &parser{toks: toks}
|
|
|
|
var primary func() (int64, bool)
|
|
var expr func() (int64, bool)
|
|
primary = func() (int64, bool) {
|
|
if p.pos >= len(p.toks) {
|
|
return 0, false
|
|
}
|
|
t := p.toks[p.pos]
|
|
switch {
|
|
case t == "-":
|
|
p.pos++
|
|
v, ok := primary()
|
|
return -v, ok
|
|
case t == "+":
|
|
p.pos++
|
|
return primary()
|
|
case t == "~":
|
|
p.pos++
|
|
v, ok := primary()
|
|
return ^v, ok
|
|
case t == "(":
|
|
p.pos++
|
|
v, ok := expr()
|
|
if !ok || p.pos >= len(p.toks) || p.toks[p.pos] != ")" {
|
|
return 0, false
|
|
}
|
|
p.pos++
|
|
return v, true
|
|
}
|
|
if t[0] < '0' || t[0] > '9' {
|
|
return 0, false
|
|
}
|
|
v, err := strconv.ParseInt(t, 0, 64)
|
|
if err != nil {
|
|
return 0, false
|
|
}
|
|
p.pos++
|
|
return v, true
|
|
}
|
|
var binop func(minLevel int) (int64, bool)
|
|
level := func(op string) int {
|
|
// Go's own precedence (the toolchain's evaluator folds with Go
|
|
// semantics): multiply, shift and AND bind tighter than add, OR and
|
|
// XOR.
|
|
switch op {
|
|
case "|", "^", "+", "-":
|
|
return 4
|
|
case "&", "<<", ">>", "*":
|
|
return 5
|
|
}
|
|
return 0
|
|
}
|
|
var apply func(v int64, op string, w int64) (int64, bool)
|
|
apply = func(v int64, op string, w int64) (int64, bool) {
|
|
switch op {
|
|
case "+":
|
|
return v + w, true
|
|
case "-":
|
|
return v - w, true
|
|
case "*":
|
|
return v * w, true
|
|
case "<<":
|
|
return v << uint(w), true
|
|
case ">>":
|
|
return v >> uint(w), true
|
|
case "&":
|
|
return v & w, true
|
|
case "|":
|
|
return v | w, true
|
|
case "^":
|
|
return v ^ w, true
|
|
}
|
|
return 0, false
|
|
}
|
|
binop = func(minLevel int) (int64, bool) {
|
|
v, ok := primary()
|
|
if !ok {
|
|
return 0, false
|
|
}
|
|
for p.pos < len(p.toks) {
|
|
op := p.toks[p.pos]
|
|
lv := level(op)
|
|
if lv == 0 || lv < minLevel {
|
|
break
|
|
}
|
|
p.pos++
|
|
w, ok := binop(lv + 1)
|
|
if !ok {
|
|
return 0, false
|
|
}
|
|
v, ok = apply(v, op, w)
|
|
if !ok {
|
|
return 0, false
|
|
}
|
|
}
|
|
return v, true
|
|
}
|
|
expr = func() (int64, bool) { return binop(1) }
|
|
v, ok := binop(1)
|
|
if !ok || p.pos != len(p.toks) {
|
|
return 0, false
|
|
}
|
|
return v, true
|
|
}
|
|
|
|
// arm64Label returns the label name of an operand.
|
|
func arm64Label(op *ast.Operand) string {
|
|
if op.Addr.Sym != nil {
|
|
return op.Addr.Sym.Name
|
|
}
|
|
return op.Raw
|
|
}
|
|
|
|
// ---- FP instruction encoding ----
|
|
|
|
// encodeARM64FP3 encodes a FP 3-operand instruction (Rm, Rn, Rd).
|
|
// FADD, FSUB, FMUL, FDIV, FMAX, FMIN, FNMUL. The two-operand spelling
|
|
// (FMULD F3, F5: multiply into the second operand) folds Rn into Rd.
|
|
func encodeARM64FP3(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 3 && len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
|
rn := rd
|
|
if len(ops) == 3 {
|
|
rn = arm64RegNum(operandRegName(ops[1]))
|
|
}
|
|
if rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64FPUnary encodes a FP unary instruction (Rn, Rd).
|
|
// FMOV reg-reg, FABS, FNEG, FSQRT, FCVT cross-precision, FRINT*.
|
|
func encodeARM64FPUnary(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64FP4 encodes a FP 4-operand FMA instruction (Ra, Rm, Rn, Rd).
|
|
// FMADD, FMSUB, FNMADD, FNMSUB.
|
|
func encodeARM64FP4(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
var ra, rm, rn, rd int
|
|
switch len(ops) {
|
|
case 4:
|
|
ra = arm64RegNum(operandRegName(ops[0]))
|
|
rm = arm64RegNum(operandRegName(ops[1]))
|
|
rn = arm64RegNum(operandRegName(ops[2]))
|
|
rd = arm64RegNum(operandRegName(ops[3]))
|
|
case 3:
|
|
// 3-operand form: Fa, Fm, Fd → Fd = Fa ± Fd*Fm (Rn = Rd)
|
|
ra = arm64RegNum(operandRegName(ops[0]))
|
|
rm = arm64RegNum(operandRegName(ops[1]))
|
|
rd = arm64RegNum(operandRegName(ops[2]))
|
|
rn = rd
|
|
default:
|
|
return nil, fmt.Errorf("%s expects 3 or 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
if ra < 0 || rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(ra)<<16 | uint32(rm)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64FPCmp encodes a FP compare instruction.
|
|
// Go assembler syntax: FCMP Fn, Fm (register) or FCMP $0.0, Fn (compare with zero).
|
|
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5].
|
|
// Go puts first operand → Rm, second → Rn.
|
|
func encodeARM64FPCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
// Check if first operand is #0 (compare with zero): FCMP $0.0, Fn.
|
|
if isImmOperand(ops[0]) && arm64Imm64(ops[0]) == 0 {
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// For compare with zero: Rm=0, op2 bit 3 set (|= 8).
|
|
return a64wordLE((baseOp | 8) | 0<<16 | uint32(rn)<<5), nil
|
|
}
|
|
// Register compare: FCMP Fn, Fm.
|
|
// Go puts first operand in Rm field, second in Rn field.
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
if rm < 0 || rn < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5), nil
|
|
}
|
|
|
|
// encodeARM64FPCCmp encodes a FP conditional compare.
|
|
// Go assembler syntax: FCCMP cond, Fn, Fm, $nzcv
|
|
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5].
|
|
// Go puts ops[1] in Rm field, ops[2] in Rn field.
|
|
func encodeARM64FPCCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
condName := operandRegName(ops[0])
|
|
cond, ok := arm64CondMap[condName]
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
|
|
}
|
|
// Go puts ops[1] in Rm (bits 20:16), ops[2] in Rn (bits 9:5).
|
|
rm := arm64RegNum(operandRegName(ops[1]))
|
|
rn := arm64RegNum(operandRegName(ops[2]))
|
|
if rm < 0 || rn < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
nzcv := arm64Imm64(ops[3])
|
|
if nzcv < 0 || nzcv > 0xF {
|
|
return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil
|
|
}
|
|
|
|
// encodeARM64FPSel encodes a FP conditional select.
|
|
// Go assembler syntax: FCSEL cond, Fn, Fm, Fd
|
|
func encodeARM64FPSel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
// Operand order: cond, Fn, Fm, Fd
|
|
condName := operandRegName(ops[0])
|
|
cond, ok := arm64CondMap[condName]
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rm := arm64RegNum(operandRegName(ops[2]))
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rn < 0 || rm < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64FPCvt encodes a FP ↔ integer conversion instruction.
|
|
// The operand order depends on direction: FCVTZS Fd, Rn (FP→int) or SCVTF Rd, Fn (int→FP).
|
|
func encodeARM64FPCvt(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
src := arm64RegNum(operandRegName(ops[0]))
|
|
dst := arm64RegNum(operandRegName(ops[1]))
|
|
if src < 0 || dst < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(src)<<5 | uint32(dst)), nil
|
|
}
|
|
|
|
// encodeARM64CSEL encodes a conditional select instruction.
|
|
// CSEL Rm, Rn, Rd, cond (4 operands) or CSET Rd, cond (2 operands).
|
|
func encodeARM64CSEL(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
isAlias := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW" ||
|
|
mnem == "CINC" || mnem == "CINCW" || mnem == "CINV" || mnem == "CINVW" ||
|
|
mnem == "CNEG" || mnem == "CNEGW"
|
|
|
|
if isAlias {
|
|
// The aliases invert the condition, which is undefined for AL and NV
|
|
// (asm7.go: invalid condition).
|
|
if condName := operandRegName(ops[0]); strings.EqualFold(condName, "AL") || strings.EqualFold(condName, "NV") {
|
|
return nil, fmt.Errorf("%s: invalid condition %s", mnem, condName)
|
|
}
|
|
is2op := mnem == "CSET" || mnem == "CSETW" || mnem == "CSETM" || mnem == "CSETMW"
|
|
if is2op {
|
|
// CSET cond, Rd → CSEL XZR, XZR, Rd, inverted_cond
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
condName := operandRegName(ops[0])
|
|
cond, ok := arm64CondMap[condName]
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
invCond := cond ^ 1
|
|
return a64wordLE(baseOp | 31<<16 | invCond<<12 | 31<<5 | uint32(rd)), nil
|
|
}
|
|
// CINC cond, Rn, Rd → CSINC Rn, Rn, Rd, inverted_cond
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
condName := operandRegName(ops[0])
|
|
cond, ok := arm64CondMap[condName]
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[2]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
invCond := cond ^ 1
|
|
return a64wordLE(baseOp | uint32(rn)<<16 | invCond<<12 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// CSEL cond, Rn, Rm, Rd (4 operands), condition first.
|
|
// Go assembler syntax: CSEL cond, Rn, Rm, Rd
|
|
// ARM64 encoding: Rm in bits[20:16], Rn in bits[9:5], Rd in bits[4:0].
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
condName := operandRegName(ops[0])
|
|
cond, ok := arm64CondMap[condName]
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rm := arm64RegNum(operandRegName(ops[2]))
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rn < 0 || rm < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | cond<<12 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64CRC32 encodes a CRC32 instruction.
|
|
// Go assembler syntax: CRC32B Rm, Rd (2 operands, Rn=Rd).
|
|
func encodeARM64CRC32(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) == 3 {
|
|
// 3-operand form: CRC32B Rm, Rn, Rd → use Rm and Rd, Rn=Rd.
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[2]))
|
|
if rm < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
|
|
}
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
rm := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rm < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rd)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// ---- Atomics encoding ----
|
|
|
|
// arm64ExclMem resolves the memory operand of an exclusive or atomic
|
|
// instruction. These encodings have no immediate field: the toolchain
|
|
// rejects `LDXR 8(R1), R2` as an illegal combination, so a non-zero offset is
|
|
// reported rather than silently dropped (which would read the wrong address).
|
|
func arm64ExclMem(mnem string, op *ast.Operand) (int, error) {
|
|
rn, off := arm64MemWithFrame(op, arm64FrameInfo{})
|
|
if rn < 0 {
|
|
return 0, fmt.Errorf("invalid memory operand in %s", mnem)
|
|
}
|
|
if off != 0 {
|
|
return 0, fmt.Errorf("%s: offset %d not supported, exclusive and atomic accesses take a plain (Rn) operand", mnem, off)
|
|
}
|
|
return rn, nil
|
|
}
|
|
|
|
// arm64PairOf parses a register-pair operand `(R1, R2)`, reporting false
|
|
// when the operand is not a pair. The toolchain takes the second register of
|
|
// the pair from the operand's Offset (its C_PAIR class,
|
|
// cmd/internal/obj/arm64/asm7.go cases 58/59).
|
|
func arm64PairOf(op *ast.Operand) (int, int, bool) {
|
|
raw := strings.TrimSpace(op.Raw)
|
|
if !strings.HasPrefix(raw, "(") || !strings.HasSuffix(raw, ")") {
|
|
return -1, -1, false
|
|
}
|
|
parts := strings.Split(raw[1:len(raw)-1], ",")
|
|
if len(parts) != 2 {
|
|
return -1, -1, false
|
|
}
|
|
r1 := arm64RegNum(strings.TrimSpace(parts[0]))
|
|
r2 := arm64RegNum(strings.TrimSpace(parts[1]))
|
|
if r1 < 0 || r2 < 0 {
|
|
return -1, -1, false
|
|
}
|
|
return r1, r2, true
|
|
}
|
|
|
|
// encodeARM64Excl encodes the exclusive load/store family with the operand
|
|
// order the toolchain parses (cmd/internal/obj/arm64/asm7.go cases 58 and 59,
|
|
// and its own spellings in arm64enc.s):
|
|
//
|
|
// STXR Rt, (Rn), Rs store, single register
|
|
// STXP (Rt1, Rt2), (Rn), Rs store, register pair
|
|
// LDXR (Rn), Rt load, single register
|
|
// LDXP (Rn), (Rt1, Rt2) load, register pair
|
|
//
|
|
// Decoded toolchain evidence: `STXR R1, (R2), R3` assembles to 0xc8037c41,
|
|
// whose fields are Rs=3, Rn=2, Rt=1: the FIRST register operand is the data
|
|
// register and the LAST the status register.
|
|
func encodeARM64Excl(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
isLoad := strings.HasPrefix(mnem, "LD")
|
|
if isLoad {
|
|
// LDXR (Rn), Rt / LDXP (Rn), (Rt1, Rt2): 2 operands.
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
rn, err := arm64ExclMem(mnem, ops[0])
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
if rt1, rt2, ok := arm64PairOf(ops[1]); ok {
|
|
// The single-register opcodes pre-set the unused Rs (bits 20:16)
|
|
// and Rt2 (bits 14:10) fields to 31; the pair forms carry a real
|
|
// Rt2 and keep Rs at 31.
|
|
// Constrained unpredictable: the base rides no pair member and
|
|
// the pair registers differ.
|
|
if rt1 == rt2 {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the pair registers match", mnem)
|
|
}
|
|
if bname := operandRegName(ops[0]); strings.EqualFold(bname, fmt.Sprintf("R%d", rt1)) || strings.EqualFold(bname, fmt.Sprintf("R%d", rt2)) {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the base rides a pair register", mnem)
|
|
}
|
|
return a64wordLE(baseOp | 0x1F<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
|
}
|
|
rt := arm64RegNum(operandRegName(ops[1]))
|
|
if rt < 0 {
|
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
|
|
}
|
|
// STXR Rt, (Rn), Rs / STXP (Rt1, Rt2), (Rn), Rs: 3 operands.
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
rn, err := arm64ExclMem(mnem, ops[1])
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
// The status register cannot be SP (asm7.go: illegal combination).
|
|
if strings.EqualFold(operandRegName(ops[2]), "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal combination: the status register cannot be RSP", mnem)
|
|
}
|
|
rs := arm64RegNum(operandRegName(ops[2]))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
|
}
|
|
if rt1, rt2, ok := arm64PairOf(ops[0]); ok {
|
|
// Constrained unpredictable (asm7.go case 59): the pair registers
|
|
// differ, and the status register differs from both pair members and
|
|
// from a non-SP base.
|
|
if rt1 == rt2 {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the pair registers match", mnem)
|
|
}
|
|
if rs == rt1 || rs == rt2 || (rs == rn && rn != 31) {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the status register rides a pair register or the base", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
|
}
|
|
// Constrained unpredictable (asm7.go case 59): the status register
|
|
// differs from Rt and from a non-SP base.
|
|
rt := arm64RegNum(operandRegName(ops[0]))
|
|
if rt < 0 {
|
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
|
}
|
|
if rs == rt || (rs == rn && rn != 31) {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the status register matches Rt or the base", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
|
}
|
|
|
|
// encodeARM64LSEAtom encodes an LSE atomic instruction (LDADD, CAS, SWP).
|
|
// LDADD Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
|
|
// CAS Rs, (Rn), Rt → 3 operands: Rs, mem, Rt
|
|
func encodeARM64LSEAtom(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
rs := arm64RegNum(operandRegName(ops[0]))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
|
}
|
|
rn, err := arm64ExclMem(mnem, ops[1])
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
// The result register cannot be SP (asm7.go: illegal combination).
|
|
if strings.EqualFold(operandRegName(ops[2]), "RSP") {
|
|
return nil, fmt.Errorf("%s: illegal combination: the result register cannot be RSP", mnem)
|
|
}
|
|
rt := arm64RegNum(operandRegName(ops[2]))
|
|
if rt < 0 {
|
|
return nil, fmt.Errorf("invalid operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
|
}
|
|
|
|
// encodeARM64CASP encodes the compare-and-swap pair: CASP (Rs, Rs+1), (Rn),
|
|
// (Rt, Rt+1). Both pairs must start on an even register and be contiguous;
|
|
// the second register of each pair rides no encoding field.
|
|
func encodeARM64CASP(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects (Rs, Rs+1), (Rn), (Rt, Rt+1), got %d operands", mnem, len(ops))
|
|
}
|
|
rs, rs1, ok := arm64PairOf(ops[0])
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s expects a source register pair (Rs, Rs+1)", mnem)
|
|
}
|
|
rn, err := arm64ExclMem(mnem, ops[1])
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
rt, rt1, ok := arm64PairOf(ops[2])
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s expects a destination register pair (Rt, Rt+1)", mnem)
|
|
}
|
|
if rs&1 != 0 {
|
|
return nil, fmt.Errorf("%s: source register pair must start from an even register", mnem)
|
|
}
|
|
if rt&1 != 0 {
|
|
return nil, fmt.Errorf("%s: destination register pair must start from an even register", mnem)
|
|
}
|
|
if rs != rs1-1 {
|
|
return nil, fmt.Errorf("%s: source register pair must be contiguous", mnem)
|
|
}
|
|
if rt != rt1-1 {
|
|
return nil, fmt.Errorf("%s: destination register pair must be contiguous", mnem)
|
|
}
|
|
if rt == 31 {
|
|
return nil, fmt.Errorf("%s: illegal destination register", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rs)<<16 | uint32(rn)<<5 | uint32(rt)), nil
|
|
}
|
|
|
|
// encodeARM64MoviImm encodes VMOVI $imm8, Vd.B8/B16: the modified-immediate
|
|
// form of the SIMD move (asm7.go case 86). Only the byte arrangements exist
|
|
// and the immediate is one unsigned byte.
|
|
func encodeARM64MoviImm(ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("VMOVI expects $immediate, Vd.<T>")
|
|
}
|
|
vd, ok := arm64VecOf(ops[1])
|
|
if !ok || vd.hasIdx || (vd.arr != "B8" && vd.arr != "B16") {
|
|
return nil, fmt.Errorf("VMOVI: destination arrangement must be B8 or B16")
|
|
}
|
|
imm := arm64Imm64(ops[0])
|
|
if imm < 0 || imm > 255 {
|
|
return nil, fmt.Errorf("VMOVI: immediate constant %d out of range (0..255)", imm)
|
|
}
|
|
q := uint32(0)
|
|
if vd.arr == "B16" {
|
|
q = 1 << 30
|
|
}
|
|
w := 0x0f00e400 | q | uint32(imm>>5&7)<<16 | uint32(imm&0x1f)<<5 | uint32(vd.reg)
|
|
return a64wordLE(w), nil
|
|
}
|
|
|
|
// arm64SimdShiftImmRoute reports whether a mnemonic carries both a shift-by-
|
|
// immediate and a register form and the operands spell the immediate one: the
|
|
// dedicated shift route keeps them.
|
|
func arm64SimdShiftImmRoute(mnem string, ops []*ast.Operand) bool {
|
|
if mnem != "VSQSHL" && mnem != "VUQSHL" {
|
|
return false
|
|
}
|
|
return len(ops) > 0 && isImmOperand(ops[0])
|
|
}
|
|
|
|
// arm64SimdNLArr describes one arrangement for the narrow/long/wide families:
|
|
// the element width in bytes and whether the spelling names the 128-bit form.
|
|
func arm64SimdNLArr(arr string) (esize int, wide bool, ok bool) {
|
|
switch arr {
|
|
case "B8", "B16":
|
|
return 1, arr == "B16", true
|
|
case "H4", "H8":
|
|
return 2, arr == "H8", true
|
|
case "S2", "S4":
|
|
return 4, arr == "S4", true
|
|
case "D1", "D2":
|
|
return 8, arr == "D2", true
|
|
}
|
|
return 0, false, false
|
|
}
|
|
|
|
// arm64SimdLongPair validates a long pairing (source narrow, destination
|
|
// wide): the destination element is twice the source's, the destination is
|
|
// always spelled the wide way (H8/S4/D2) and the source carries the 128-bit
|
|
// flag exactly for the .2 spellings.
|
|
func arm64SimdLongPair(mnem, src, dst string, two bool) error {
|
|
se, sw, ok1 := arm64SimdNLArr(src)
|
|
de, dw, ok2 := arm64SimdNLArr(dst)
|
|
if !ok1 || !ok2 || de != 2*se {
|
|
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
|
|
}
|
|
if !dw || sw != two {
|
|
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// arm64SimdNarrowPair validates a narrow pairing (source wide, destination
|
|
// narrow): the mirror image of arm64SimdLongPair.
|
|
func arm64SimdNarrowPair(mnem, src, dst string, two bool) error {
|
|
se, sw, ok1 := arm64SimdNLArr(src)
|
|
de, dw, ok2 := arm64SimdNLArr(dst)
|
|
if !ok1 || !ok2 || se != 2*de {
|
|
return fmt.Errorf("%s: incompatible arrangements %s and %s", mnem, src, dst)
|
|
}
|
|
if !sw || dw != two {
|
|
return fmt.Errorf("%s: operand mismatch for the %s spelling", mnem, mnem)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// arm64SimdFCVTLongPair validates the FCVTL width pair: S to D alone, with
|
|
// the plain spelling reading S2 and the .2 spelling S4, the destination
|
|
// always D2.
|
|
func arm64SimdFCVTLongPair(mnem, src, dst string, two bool) error {
|
|
wantSrc := "S2"
|
|
if two {
|
|
wantSrc = "S4"
|
|
}
|
|
if src != wantSrc || dst != "D2" {
|
|
return fmt.Errorf("%s: operand mismatch for the %s spelling: want %s, %s", mnem, mnem, wantSrc, "D2")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// arm64SimdFCVTNarrowPair validates the FCVTN width pair: D to S alone, the
|
|
// source always D2, the destination S2 for the plain spelling and S4 for
|
|
// the .2 spelling.
|
|
func arm64SimdFCVTNarrowPair(mnem, src, dst string, two bool) error {
|
|
wantDst := "S2"
|
|
if two {
|
|
wantDst = "S4"
|
|
}
|
|
if src != "D2" || dst != wantDst {
|
|
return fmt.Errorf("%s: operand mismatch for the %s spelling: want %s, %s", mnem, mnem, "D2", wantDst)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// arm64SimdNLArrBits returns the arrangement bits a narrow/long/wide
|
|
// instruction contributes: the driving arrangement's size and Q bits, for
|
|
// the FCVT family only the Q bit (whose size field is fixed in the base),
|
|
// and for the long extend family the immh shift field the long forms imply
|
|
// (immh = esize/8) plus the Q bit for the .2 spellings.
|
|
func arm64SimdNLArrBits(spec a64SimdNLSpec, drive string, two bool) uint32 {
|
|
if spec.qonly {
|
|
if two {
|
|
return 1 << 30
|
|
}
|
|
return 0
|
|
}
|
|
if spec.form == a64NLTwoLong {
|
|
se, _, _ := arm64SimdNLArr(drive)
|
|
bits := uint32(se) << 19
|
|
if two {
|
|
bits |= 1 << 30
|
|
}
|
|
return bits
|
|
}
|
|
return a64ArrBits[a64ArrIndex(drive)]
|
|
}
|
|
|
|
// encodeARM64SimdNL encodes the narrow/long/wide SIMD families
|
|
// (a64SimdNLTable): XTN and FCVTN narrow a wide source, SXTL and FCVTL
|
|
// lengthen, the MULL/MLAL/MLSL group multiplies long, UADDW widens, and the
|
|
// SSHLL/USHLL and SHRN shifts carry their immediate in the immh:immb field.
|
|
// The size and Q bits read off the designated driving operand, and the .2
|
|
// spellings force the 128-bit side through their own arrangement.
|
|
func encodeARM64SimdNL(mnem string, spec a64SimdNLSpec, ops []*ast.Operand) ([]byte, error) {
|
|
two := strings.HasSuffix(mnem, "2")
|
|
|
|
// vecAt parses operand i as a vector register with an arrangement.
|
|
vecAt := func(i int) (a64Vec, bool) {
|
|
if i >= len(ops) {
|
|
return a64Vec{}, false
|
|
}
|
|
v, ok := arm64VecOf(ops[i])
|
|
if !ok || v.hasIdx {
|
|
return a64Vec{}, false
|
|
}
|
|
return v, true
|
|
}
|
|
|
|
switch spec.form {
|
|
case a64NLTwoNarrow, a64NLTwoLong:
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
vn, ok1 := vecAt(0)
|
|
vd, ok2 := vecAt(1)
|
|
if !ok1 || !ok2 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
var drive string
|
|
var pairErr error
|
|
if spec.qonly {
|
|
// The FCVT conversions are pinned to one width pair: S to D for
|
|
// the lengthening (FCVTL S2→D2, FCVTL2 S4→D2), D to S for the
|
|
// narrowing (FCVTN D2→S2, FCVTN2 D2→S4); the size field is
|
|
// fixed in the opcode and no other arrangement exists.
|
|
if spec.form == a64NLTwoNarrow {
|
|
drive = vd.arr
|
|
pairErr = arm64SimdFCVTNarrowPair(mnem, vn.arr, vd.arr, two)
|
|
} else {
|
|
drive = vn.arr
|
|
pairErr = arm64SimdFCVTLongPair(mnem, vn.arr, vd.arr, two)
|
|
}
|
|
} else if spec.form == a64NLTwoNarrow {
|
|
// XTN/FCVTN: wide source into a narrow destination; the
|
|
// arrangement bits follow the destination.
|
|
drive, pairErr = vd.arr, arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two)
|
|
} else {
|
|
// SXTL/UXTL/FCVTL: narrow source into a wide destination; the
|
|
// arrangement bits follow the source.
|
|
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
|
|
}
|
|
if pairErr != nil {
|
|
return nil, pairErr
|
|
}
|
|
arrBits := arm64SimdNLArrBits(spec, drive, two)
|
|
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
case a64NLThreeLongMul, a64NLThreeWide:
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
vm, ok1 := vecAt(0)
|
|
vn, ok2 := vecAt(1)
|
|
vd, ok3 := vecAt(2)
|
|
if !ok1 || !ok2 || !ok3 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
var drive string
|
|
var pairErr error
|
|
if spec.form == a64NLThreeWide {
|
|
// UADDW: Vn and Vd spell the wide arrangement, Vm the narrow one;
|
|
// the size bits follow the narrow side (Vm) and the 128-bit flag
|
|
// follows the spelling: the plain form keeps Q clear, the .2
|
|
// form sets it.
|
|
drive, pairErr = vm.arr, arm64SimdLongPair(mnem, vm.arr, vn.arr, two)
|
|
} else {
|
|
// MULL/MLAL/MLSL: Vm and Vn spell the narrow arrangement, Vd the
|
|
// wide one; the arrangement bits follow the narrow source.
|
|
drive, pairErr = vn.arr, arm64SimdLongPair(mnem, vn.arr, vd.arr, two)
|
|
}
|
|
if pairErr != nil {
|
|
return nil, pairErr
|
|
}
|
|
if spec.form == a64NLThreeWide && vd.arr != vn.arr {
|
|
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vn.arr, vd.arr)
|
|
}
|
|
if spec.form == a64NLThreeLongMul && vm.arr != vn.arr {
|
|
return nil, fmt.Errorf("%s: operand mismatch: %s and %s", mnem, vm.arr, vn.arr)
|
|
}
|
|
arrBits := arm64SimdNLArrBits(spec, drive, two)
|
|
if spec.form == a64NLThreeWide {
|
|
// The size bits ride the narrow side's letter with Q forced by
|
|
// the spelling alone.
|
|
arrBits = a64ArrBits[a64ArrIndex(drive)] &^ (1 << 30)
|
|
if two {
|
|
arrBits |= 1 << 30
|
|
}
|
|
}
|
|
return a64wordLE(spec.base | arrBits | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
case a64NLThreeLongShift, a64NLThreeNarrowShift:
|
|
if len(ops) != 3 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects ($shift, Vn.<T>, Vd.<T>)", mnem)
|
|
}
|
|
sh := arm64Imm64(ops[0])
|
|
vn, ok1 := vecAt(1)
|
|
vd, ok2 := vecAt(2)
|
|
if !ok1 || !ok2 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
if spec.form == a64NLThreeLongShift {
|
|
// SSHLL/USHLL: the narrow source drives the immediate's size
|
|
// (immh:immb = esize + shift), so the arrangement bits carry
|
|
// the Q bit alone: the size field belongs to immh, and ORing
|
|
// the source's size bits into it would collide with immb.
|
|
if err := arm64SimdLongPair(mnem, vn.arr, vd.arr, two); err != nil {
|
|
return nil, err
|
|
}
|
|
se, _, _ := arm64SimdNLArr(vn.arr)
|
|
esize := se * 8
|
|
if sh < 0 || sh >= int64(esize) {
|
|
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
|
}
|
|
var qBit uint32
|
|
if two {
|
|
qBit = 1 << 30
|
|
}
|
|
return a64wordLE(spec.base | uint32(esize+int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
}
|
|
// SHRN: the narrow destination drives the immediate's size
|
|
// (immh:immb = esize - shift over the wide source element), so the
|
|
// arrangement bits carry the Q bit alone, exactly as above.
|
|
if err := arm64SimdNarrowPair(mnem, vn.arr, vd.arr, two); err != nil {
|
|
return nil, err
|
|
}
|
|
se, _, _ := arm64SimdNLArr(vn.arr)
|
|
esize := se * 8
|
|
if sh < 1 || sh >= int64(esize) {
|
|
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize-1)
|
|
}
|
|
var qBit uint32
|
|
if two {
|
|
qBit = 1 << 30
|
|
}
|
|
return a64wordLE(spec.base | uint32(esize-int(sh))<<16 | qBit | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
}
|
|
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
|
}
|
|
|
|
// encodeARM64DP1 encodes a data-processing (1 source) instruction:
|
|
// RBIT, REV, CLZ and friends take (Rn, Rd), word = base | Rn<<5 | Rd.
|
|
func encodeARM64DP1(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[0]))
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64ADR encodes ADR/ADRP: (label, Rd), the byte distance to the
|
|
// target split into immlo (bits 30:29) and immhi (bits 23:5), with bit 31
|
|
// selecting the page form. An n(PC) operand resolves to the instruction's
|
|
// own address: the toolchain rewrites it away and encodes displacement 0.
|
|
func encodeARM64ADR(mnem string, page uint32, ops []*ast.Operand, pc int, offsets map[string]int) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
var rel int64
|
|
if _, pcRel := arm64PCRelOffset(ops[0]); !pcRel {
|
|
// The label resolves to its own statement: the toolchain's
|
|
// jump-to-jump collapse rewrites p.To alone (obj/pass.go), so an
|
|
// ADR/ADRP's From-side label is never chased through a chain.
|
|
target := arm64Label(ops[0])
|
|
targetOff, ok := offsets[target]
|
|
if !ok {
|
|
return nil, fmt.Errorf("undefined label %q", target)
|
|
}
|
|
rel = int64(targetOff - pc)
|
|
if rel < -(1<<20) || rel >= 1<<20 {
|
|
return nil, fmt.Errorf("%s to %q too far (21-bit range)", mnem, target)
|
|
}
|
|
}
|
|
return a64wordLE(a64ADR(page, int32(rel>>2), uint32(rel)&3, uint32(rd))), nil
|
|
}
|
|
|
|
// encodeARM64Bitfield2 encodes UBFX/SBFX ($lsb, Rn, $width, Rd): immr
|
|
// carries the lsb (six bits with the N flag on the X forms) and imms the
|
|
// lsb plus width minus one. A sum beyond the register width is the
|
|
// toolchain's "illegal bit number" error.
|
|
func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
|
|
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
|
|
}
|
|
lsb := arm64Imm64(ops[0])
|
|
width := arm64Imm64(ops[2])
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
bits := int64(32) << (baseOp >> 31 & 1)
|
|
if lsb < 0 || lsb >= bits {
|
|
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
|
|
}
|
|
if width < 1 || lsb+width > bits {
|
|
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
|
|
}
|
|
immr := uint32(lsb)
|
|
n := uint32(0)
|
|
if immr >= 32 {
|
|
n = 1 << 21
|
|
immr &^= 32
|
|
}
|
|
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
|
|
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
|
|
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
|
|
// with imms = lsb+width-1.
|
|
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
|
|
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
|
|
}
|
|
lsb := arm64Imm64(ops[0])
|
|
width := arm64Imm64(ops[2])
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
bits := int64(32) << (baseOp >> 31 & 1)
|
|
if lsb < 0 || lsb >= bits {
|
|
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
|
|
}
|
|
if width < 1 || width > bits || lsb+width > bits {
|
|
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
|
|
}
|
|
var immr, imms int64
|
|
switch mnem {
|
|
case "BFXIL", "BFXILW":
|
|
immr, imms = lsb, lsb+width-1
|
|
default: // BFI, SBFIZ, UBFIZ
|
|
// immr = (bits - lsb) mod bits, the toolchain's 64-r form with the
|
|
// zero lsb folding to zero (Go's % keeps the sign, so the operand
|
|
// order matters here).
|
|
immr, imms = (bits-lsb)%bits, width-1
|
|
}
|
|
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
|
|
// third field carries Rm or a 5-bit immediate in the same bits, at the
|
|
// toolchain's choice of register or immediate operand.
|
|
func encodeARM64CondCmp(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 4 || !isImmOperand(ops[3]) {
|
|
return nil, fmt.Errorf("%s expects 4 operands (cond, Rn, Rm|$imm, $nzcv)", mnem)
|
|
}
|
|
condName := operandRegName(ops[0])
|
|
cond, ok := arm64CondMap[condName]
|
|
if !ok {
|
|
return nil, fmt.Errorf("invalid condition code %q in %s", condName, mnem)
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
var v2 uint32
|
|
if isImmOperand(ops[2]) {
|
|
v := arm64Imm64(ops[2])
|
|
if v < 0 || v > 31 {
|
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..31)", mnem, v)
|
|
}
|
|
v2 = uint32(v)
|
|
} else {
|
|
rm := arm64RegNum(operandRegName(ops[2]))
|
|
if rm < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
v2 = uint32(rm)
|
|
}
|
|
nzcv := arm64Imm64(ops[3])
|
|
if nzcv < 0 || nzcv > 15 {
|
|
return nil, fmt.Errorf("%s: nzcv %d out of range (0..15)", mnem, nzcv)
|
|
}
|
|
// Bit 11 carries the immediate-vs-register choice for the third field.
|
|
op2 := uint32(0)
|
|
if isImmOperand(ops[2]) {
|
|
op2 = 1 << 11
|
|
}
|
|
return a64wordLE(baseOp | v2<<16 | cond<<12 | op2 | uint32(rn)<<5 | uint32(nzcv)&0xF), nil
|
|
}
|
|
|
|
// encodeARM64Branch19 encodes CBZ/CBNZ: (Rt, label),
|
|
// word = base | imm19<<5 | Rt with imm19 = (target - pc) >> 2.
|
|
// arm64PCRelOffset recognises the toolchain's forward branch spelling
|
|
// n(PC): the number counts INSTRUCTIONS from the branch itself. It reports
|
|
// ok for operands spelled that way and leaves everything else alone.
|
|
func arm64PCRelOffset(op *ast.Operand) (int, bool) {
|
|
if op.Addr.Base != "PC" && !(op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
|
return 0, false
|
|
}
|
|
off := op.Addr.Offset
|
|
if !op.Addr.HasOff {
|
|
off = 0
|
|
}
|
|
return int(off), true
|
|
}
|
|
|
|
func encodeARM64Branch19(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
rt := arm64RegNum(operandRegName(ops[0]))
|
|
if rt < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
rel, pcRel := arm64PCRelOffset(ops[1])
|
|
if !pcRel {
|
|
target := resolve(arm64Label(ops[1]))
|
|
targetOff, ok := offsets[target]
|
|
if !ok {
|
|
return nil, fmt.Errorf("undefined label %q", target)
|
|
}
|
|
rel = (targetOff - pc) >> 2
|
|
}
|
|
if rel < -(1<<18) || rel >= (1<<18) {
|
|
return nil, fmt.Errorf("%s: branch offset %d out of 19-bit range", mnem, rel)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rel)&0x7FFFF<<5 | uint32(rt)), nil
|
|
}
|
|
|
|
// encodeARM64TestBranch encodes TBZ/TBNZ: ($bit, Rt, label). Bits 32 to 63
|
|
// set the b5 flag at bit 31; there is one mnemonic per polarity, no width
|
|
// suffix.
|
|
func encodeARM64TestBranch(mnem string, baseOp uint32, ops []*ast.Operand, pc int, offsets map[string]int, resolve func(string) string) ([]byte, error) {
|
|
if len(ops) != 3 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects 3 operands ($bit, Rt, label)", mnem)
|
|
}
|
|
bit := arm64Imm64(ops[0])
|
|
if bit < 0 || bit > 63 {
|
|
return nil, fmt.Errorf("%s: bit number %d out of range (0..63)", mnem, bit)
|
|
}
|
|
rt := arm64RegNum(operandRegName(ops[1]))
|
|
if rt < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
rel, pcRel := arm64PCRelOffset(ops[2])
|
|
if !pcRel {
|
|
target := resolve(arm64Label(ops[2]))
|
|
targetOff, ok := offsets[target]
|
|
if !ok {
|
|
return nil, fmt.Errorf("undefined label %q", target)
|
|
}
|
|
rel = (targetOff - pc) >> 2
|
|
}
|
|
if rel < -(1<<13) || rel >= (1<<13) {
|
|
return nil, fmt.Errorf("%s: branch offset %d out of 14-bit range", mnem, rel)
|
|
}
|
|
return a64wordLE(baseOp | uint32(bit>>5)<<31 | uint32(bit&31)<<19 | uint32(rel)&0x3FFF<<5 | uint32(rt)), nil
|
|
}
|
|
|
|
// encodeARM64Pair encodes load/store pair instructions. Loads spell
|
|
// (mem, (Rt1, Rt2)), stores (Rt1, Rt2), mem; the scaled immediate rides
|
|
// imm7 at bits 21:15 and must fit -64..63 after division by the access
|
|
// size (8 bytes for the D forms, 4 for the W forms).
|
|
// encodeARM64Pair encodes the load/store pair family. wb selects the
|
|
// addressing mode: "" the signed-offset form, "P" post-index, "W" pre-index;
|
|
// in the writeback forms the immediate is the amount added to the base
|
|
// register around the access.
|
|
func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, pc int, fi arm64FrameInfo, wb string, relocs *[]Reloc, pool *arm64Pool, poolBase int) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
scale := arm64PairScale(mnem)
|
|
load := strings.Contains(mnem, "LDP")
|
|
memOp, pairOp := ops[0], ops[1]
|
|
if !load {
|
|
memOp, pairOp = ops[1], ops[0]
|
|
}
|
|
if !arm64IsMemOperand(memOp) {
|
|
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
|
|
}
|
|
rt1, rt2, ok := arm64PairOf(pairOp)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s expects a register pair (Rt1, Rt2)", mnem)
|
|
}
|
|
// Constrained unpredictable: the pair registers differ (the ZR pair is
|
|
// the toolchain's own idiom), and a writeback base rides no pair member.
|
|
if rt1 == rt2 && rt1 != 31 {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the pair registers match", mnem)
|
|
}
|
|
if wb != "" && rt1 != 31 {
|
|
if strings.EqualFold(operandRegName(memOp), fmt.Sprintf("R%d", rt1)) || strings.EqualFold(operandRegName(memOp), fmt.Sprintf("R%d", rt2)) {
|
|
return nil, fmt.Errorf("%s: constrained unpredictable behavior: the base rides a pair register", mnem)
|
|
}
|
|
}
|
|
// The FP pairs take FP registers against a GP base (asm7.go: invalid
|
|
// register pair). The integer pairs are the mirror image: an F pair on
|
|
// LDP/STP is an invalid register pair too, the toolchain having no class
|
|
// for it.
|
|
if strings.HasPrefix(mnem, "FLDP") || strings.HasPrefix(mnem, "FSTP") {
|
|
first := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(pairOp.Raw), "("))
|
|
if i := strings.IndexAny(first, ",)"); i >= 0 {
|
|
first = strings.TrimSpace(first[:i])
|
|
}
|
|
if !strings.HasPrefix(first, "F") {
|
|
return nil, fmt.Errorf("%s: invalid register pair %s", mnem, pairOp.Raw)
|
|
}
|
|
if strings.HasPrefix(strings.TrimLeft(operandRegName(memOp), "( "), "F") {
|
|
return nil, fmt.Errorf("%s: invalid register pair: the base must be a general register", mnem)
|
|
}
|
|
} else {
|
|
first := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(pairOp.Raw), "("))
|
|
if i := strings.IndexAny(first, ",)"); i >= 0 {
|
|
first = strings.TrimSpace(first[:i])
|
|
}
|
|
if strings.HasPrefix(first, "F") {
|
|
return nil, fmt.Errorf("%s: invalid register pair %s", mnem, pairOp.Raw)
|
|
}
|
|
}
|
|
|
|
// Pair access against a static symbol: ADRP R27, sym; ADD R27, R27, #lo;
|
|
// LDP/STP (R27), (Rt1, Rt2), with the R_ADDRARM64 pair riding the first
|
|
// two words, exactly like the toolchain lays it out.
|
|
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" {
|
|
if wb != "" {
|
|
return nil, fmt.Errorf("%s: writeback is not supported on a symbol operand", mnem)
|
|
}
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 0, Name: memOp.Addr.Sym.Name, Kind: RelArm64Addr, Addend: memOp.Addr.Sym.Offset},
|
|
Reloc{Off: 4, After: 4, Name: memOp.Addr.Sym.Name, Kind: RelArm64Addr, Addend: memOp.Addr.Sym.Offset},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, 27), // ADRP R27, 0
|
|
a64AddSub(1, 0, 0, 0, 0, 27, 27), // ADD $0, R27, R27
|
|
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (Rt1, Rt2)
|
|
), nil
|
|
}
|
|
|
|
rn, off := arm64MemWithFrame(memOp, fi)
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
|
|
}
|
|
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" {
|
|
return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem)
|
|
}
|
|
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
|
|
imm7 := off / scale
|
|
// The signed-offset form carries bits 24:23 = 10; post-index drops
|
|
// bit 24 and pre-index sets both, with the base carrying the opc, V
|
|
// and L halves.
|
|
switch wb {
|
|
case "P":
|
|
baseOp = baseOp&^(1<<24) | 1<<23
|
|
case "W":
|
|
baseOp |= 1 << 23
|
|
}
|
|
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
|
}
|
|
if wb != "" {
|
|
return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale)
|
|
}
|
|
// Offsets within ±4095 the imm7 field cannot carry move the whole
|
|
// distance into REGTMP first (asm7.go cases 74/76: add/sub + ldp/stp).
|
|
// A store refuses a REGTMP pair member here (case 76: the add would
|
|
// clobber it before the store reads it); a load allows one.
|
|
if off >= -4095 && off <= 4095 {
|
|
if !load && (rt1 == 27 || rt2 == 27) {
|
|
return nil, fmt.Errorf("%s: cannot use REGTMP as source", mnem)
|
|
}
|
|
op, v := uint32(0), off // ADD
|
|
if v < 0 {
|
|
op, v = 1, -v // SUB
|
|
}
|
|
return a64WordsLE(
|
|
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
|
|
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
|
), nil
|
|
}
|
|
// Positive offsets up to 16 MiB split into two ADDs: the low imm12 bits
|
|
// from the base register into REGTMP, the high multiple of 0x1000 on top
|
|
// (asm7.go cases 75/77). Beyond the band the offset reaches the pool,
|
|
// which refuses a REGTMP base outright and, for the stores, a REGTMP
|
|
// pair member as well.
|
|
if off < 0 || off > 0xffffff {
|
|
if rn == 27 {
|
|
kind := "load"
|
|
if !load {
|
|
kind = "store"
|
|
}
|
|
return nil, fmt.Errorf("%s: REGTMP used in large offset %s", mnem, kind)
|
|
}
|
|
if !load && (rt1 == 27 || rt2 == 27) {
|
|
return nil, fmt.Errorf("%s: REGTMP used in large offset store", mnem)
|
|
}
|
|
if pool == nil {
|
|
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
|
}
|
|
entryOff, w := pool.add(off)
|
|
dist := (poolBase + entryOff - pc) >> 2
|
|
if !pool.probe && (dist < -(1<<18) || dist >= 1<<18) {
|
|
return nil, fmt.Errorf("%s: literal pool %d out of 19-bit reach", mnem, dist<<2)
|
|
}
|
|
return a64WordsLE(
|
|
w<<30|3<<27|uint32(dist)&0x7FFFF<<5|27, // LDR R27, pool
|
|
a64AddSubReg(1, 27, uint32(rn), 27), // ADD R27, Rn, R27
|
|
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
|
), nil
|
|
}
|
|
hi, lo, ok := arm64SplitImm24(off, 0)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
|
}
|
|
return a64WordsLE(
|
|
arm64AddImmWord(lo, uint32(rn), 27), // ADD $lo, Rn, R27
|
|
arm64AddImmWord(hi, 27, 27), // ADD $hi[<<12], R27, R27
|
|
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
|
), nil
|
|
}
|
|
|
|
// a64AddSubReg encodes ADD/SUB (extended register) with the identity
|
|
// extend: sf | 0xB<<24 | 1<<21 | Rm<<16 | UXTX<<13 | Rn<<5 | Rd, the shape
|
|
// opxrrr emits for the pool's base addition.
|
|
func a64AddSubReg(sf, rm, rn, rd uint32) uint32 {
|
|
return sf<<31 | 0xB<<24 | 1<<21 | rm<<16 | 3<<13 | rn<<5 | rd
|
|
}
|
|
|
|
// arm64PairScale returns the byte width a pair access's imm7 offset divides
|
|
// by: four for the 32-bit pairs (integer W, signed W and the FP S pairs),
|
|
// sixteen for the 128-bit FP pairs, eight for everything else.
|
|
func arm64PairScale(mnem string) int64 {
|
|
switch {
|
|
case strings.HasSuffix(mnem, "W"), mnem == "FLDPS", mnem == "FSTPS":
|
|
return 4
|
|
case strings.HasSuffix(mnem, "Q"):
|
|
return 16
|
|
}
|
|
return 8
|
|
}
|
|
|
|
// arm64SpExtendOpt reports the extended-register encoding an ADD/SUB-family
|
|
// instruction takes when the stack pointer sits among the base and
|
|
// destination registers: the identity extend UXTX (UXTW in the 32-bit
|
|
// forms), how the toolchain spells a plain register operand against SP.
|
|
// ok is false for the logical and carry groups, whose register form has no
|
|
// extend field and admits no SP operand.
|
|
func arm64SpExtendOpt(mnem string, ops []*ast.Operand) (uint32, bool) {
|
|
sp := false
|
|
for _, op := range ops {
|
|
if n := operandRegName(op); n == "SP" || n == "RSP" {
|
|
sp = true
|
|
}
|
|
}
|
|
if !sp {
|
|
return 0, false
|
|
}
|
|
switch {
|
|
case strings.HasPrefix(mnem, "ADD"), strings.HasPrefix(mnem, "SUB"),
|
|
mnem == "CMP", mnem == "CMPW", mnem == "CMN", mnem == "CMNW":
|
|
opt := uint32(3) // UXTX
|
|
if strings.HasSuffix(mnem, "W") {
|
|
opt = 2 // UXTW
|
|
}
|
|
return opt, true
|
|
}
|
|
return 0, false
|
|
}
|
|
|
|
// arm64AddShift returns the ADD-immediate shift bit: a non-zero value whose
|
|
// low twelve bits are zero encodes shifted left by twelve.
|
|
func arm64AddShift(v int64) uint32 {
|
|
if v != 0 && v&0xfff == 0 {
|
|
return 1
|
|
}
|
|
return 0
|
|
}
|
|
|
|
// arm64SplitImm24 decomposes an offset the way the toolchain's
|
|
// splitImm24uScaled does for the two-ADD sequences: hi either fits imm12
|
|
// alone or is a multiple of 0x1000 up to 0xfff000, and the low half always
|
|
// fits imm12 after division by the access scale. ok is false when the value
|
|
// is negative or beyond the 24-bit reach, where the toolchain pools.
|
|
func arm64SplitImm24(v int64, shift int) (hi, lo int64, ok bool) {
|
|
if v < 0 || v > 0xfff000+0xfff<<uint(shift) {
|
|
return 0, 0, false
|
|
}
|
|
h := max(v-(0xfff<<uint(shift)), v&((1<<uint(shift))-1))
|
|
if h <= 0xfff {
|
|
l := (v - h) >> uint(shift)
|
|
if l <= 0xfff {
|
|
return h, l, true
|
|
}
|
|
}
|
|
l := (v >> uint(shift)) & 0xfff
|
|
h = v - l<<uint(shift)
|
|
if h > 0xfff000 {
|
|
h = 0xfff000
|
|
l = (v - h) >> uint(shift)
|
|
}
|
|
if h&^0xfff000 == 0 && h+l<<uint(shift) == v {
|
|
return h, l, true
|
|
}
|
|
return 0, 0, false
|
|
}
|
|
|
|
// encodeARM64AcqRel encodes the acquire/release loads and stores. Loads
|
|
// spell (Rn), Rt; stores Rt, (Rn). Both lay the base register at bits 9:5
|
|
// and the data register at bits 4:0 over a base that carries no offset
|
|
// field, so a displaced operand is reported the way arm64ExclMem reports
|
|
// one for the exclusive family.
|
|
func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
memOp, regOp := ops[0], ops[1]
|
|
if !strings.HasPrefix(mnem, "LD") {
|
|
memOp, regOp = ops[1], ops[0]
|
|
}
|
|
rn, err := arm64ExclMem(mnem, memOp)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
rt := arm64RegNum(operandRegName(regOp))
|
|
if rt < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rn)<<5 | uint32(rt)), nil
|
|
}
|
|
|
|
// encodeARM64Sys encodes the system operations:
|
|
//
|
|
// BRK [$imm16] SVC $imm16
|
|
// DMB|DSB|ISB $imm4 DC <op>, Rn
|
|
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
|
// PRFM (Rn), $imm|<op> RPRFM (Rn), Rm, <op|$imm6>
|
|
// SYS $imm[, Rn] SYSL $imm, Rd
|
|
// TLBI <op>[, Rn] SB, PACIASP, PACIBSP
|
|
//
|
|
// The system registers, TLBI and DC aliases and the range-prefetch operations
|
|
// come from the toolchain's own data tables in arm64_sysregs.go.
|
|
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|
// Operand-less returns and pointer-authentication hints.
|
|
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
|
|
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, "PACIASP": 0xd503233f, "PACIBSP": 0xd503237f,
|
|
"AUTIA1716": 0xd503219f, "AUTIB1716": 0xd50321df,
|
|
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
|
|
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
|
|
if len(ops) != 0 {
|
|
return nil, fmt.Errorf("%s expects no operand", mnem)
|
|
}
|
|
return a64wordLE(w), nil
|
|
}
|
|
switch mnem {
|
|
case "BRK", "SVC":
|
|
base := uint32(0xd4200000)
|
|
if mnem == "SVC" {
|
|
base = 0xd4000001
|
|
}
|
|
if len(ops) == 0 {
|
|
return a64wordLE(base), nil
|
|
}
|
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects no operand or $immediate", mnem)
|
|
}
|
|
v := arm64Imm64(ops[0])
|
|
if v < 0 || v > 0xFFFF {
|
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
|
}
|
|
return a64wordLE(base | uint32(v)<<5), nil
|
|
case "DMB", "DSB", "ISB", "CLREX":
|
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
|
}
|
|
v := arm64Imm64(ops[0])
|
|
if v < 0 || v > 15 {
|
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
|
|
}
|
|
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
|
|
return a64wordLE(base | uint32(v)<<8), nil
|
|
case "SB":
|
|
// Speculation barrier: DSB with a fixed barrier domain.
|
|
if len(ops) != 0 {
|
|
return nil, fmt.Errorf("%s expects no operand", mnem)
|
|
}
|
|
return a64wordLE(0xd50330ff), nil
|
|
case "HINT":
|
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
|
}
|
|
v := arm64Imm64(ops[0])
|
|
if v < 0 || v > 127 {
|
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
|
|
}
|
|
return a64wordLE(0xd503201f | uint32(v)<<5), nil
|
|
case "BTI":
|
|
// The toolchain requires the landing-pad kind: bare BTI is
|
|
// rejected ("missing operand"), and only the uppercase C/J/JC
|
|
// spellings assemble (0xd503245f/49f/4df).
|
|
if len(ops) != 1 {
|
|
return nil, fmt.Errorf("%s expects C, J, or JC", mnem)
|
|
}
|
|
op := operandRegName(ops[0])
|
|
base, ok := map[string]uint32{"C": 0xd503245f, "J": 0xd503249f, "JC": 0xd50324df}[op]
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
|
|
}
|
|
return a64wordLE(base), nil
|
|
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
|
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
|
}
|
|
v := arm64Imm64(ops[0])
|
|
if v < 0 || v > 0xFFFF {
|
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
|
}
|
|
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
|
|
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
|
|
"DCPS3": 0xd4a00003}[mnem]
|
|
return a64wordLE(base | uint32(v)<<5), nil
|
|
case "DC":
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("DC expects <op>, Rn")
|
|
}
|
|
inst, ok := a64DCOps2[operandRegName(ops[0])]
|
|
if !ok {
|
|
return nil, fmt.Errorf("DC: unknown cache operation %q", operandRegName(ops[0]))
|
|
}
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("DC: invalid register operand")
|
|
}
|
|
w := 0xd5080000 | inst.op1<<16 | 7<<12 | inst.cm<<8 | inst.op2<<5
|
|
return a64wordLE(w | uint32(rn)&31), nil
|
|
case "TLBI":
|
|
// The register operand's arity follows the operation (asm7.go's
|
|
// sysInstFields): the by-address spellings take the Xt and the
|
|
// whole-entry ones (VMALL*, ALL**) take none, an explicit register
|
|
// being extraneous.
|
|
if len(ops) != 1 && len(ops) != 2 {
|
|
return nil, fmt.Errorf("TLBI expects <op>[, Rn]")
|
|
}
|
|
name := operandRegName(ops[0])
|
|
inst, ok := a64TLBIOps[name]
|
|
if !ok {
|
|
return nil, fmt.Errorf("TLBI: unknown operation %q", name)
|
|
}
|
|
rt := 31
|
|
if len(ops) == 2 {
|
|
if !arm64TLBITakesReg(name) {
|
|
return nil, fmt.Errorf("TLBI %s: extraneous register at operand 2", name)
|
|
}
|
|
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
|
|
return nil, fmt.Errorf("TLBI: invalid register operand")
|
|
}
|
|
} else if arm64TLBITakesReg(name) {
|
|
return nil, fmt.Errorf("TLBI %s: missing register at operand 2", name)
|
|
}
|
|
w := 0xd5080000 | inst.op1<<16 | 8<<12 | inst.cm<<8 | inst.op2<<5
|
|
return a64wordLE(w | uint32(rt)&31), nil
|
|
case "SYS", "SYSL":
|
|
// SYS $imm[, Rn] / SYSL $imm, Rd: the immediate packs
|
|
// op1<<16 | CRn<<12 | CRm<<8 | op2<<5, the register defaults to ZR.
|
|
if len(ops) != 1 && len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
|
|
}
|
|
if len(ops) == 1 && mnem == "SYSL" {
|
|
return nil, fmt.Errorf("SYSL expects $immediate, Rd")
|
|
}
|
|
if !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects $immediate[, Rn]", mnem)
|
|
}
|
|
imm := arm64Imm64(ops[0])
|
|
if imm < 0 || imm&^0x7FFE0 != 0 {
|
|
return nil, fmt.Errorf("%s: illegal SYS argument %d", mnem, imm)
|
|
}
|
|
rt := 31
|
|
if len(ops) == 2 {
|
|
if rt = arm64RegNum(operandRegName(ops[1])); rt < 0 {
|
|
return nil, fmt.Errorf("%s: invalid register operand", mnem)
|
|
}
|
|
}
|
|
base := uint32(0xd5080000)
|
|
if mnem == "SYSL" {
|
|
base = 0xd5280000
|
|
}
|
|
return a64wordLE(base | uint32(imm) | uint32(rt)&31), nil
|
|
case "MRS":
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("MRS expects <sysreg>, Rd")
|
|
}
|
|
reg, ok := a64SysRegs[operandRegName(ops[0])]
|
|
if !ok {
|
|
return nil, fmt.Errorf("MRS: unknown system register %q", operandRegName(ops[0]))
|
|
}
|
|
if !reg.read {
|
|
return nil, fmt.Errorf("MRS: system register is not readable: %q", operandRegName(ops[0]))
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("MRS: invalid register operand")
|
|
}
|
|
return a64wordLE(0xd5300000 | reg.v | uint32(rd)&31), nil
|
|
case "MSR":
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
|
|
}
|
|
if isImmOperand(ops[0]) {
|
|
v := arm64Imm64(ops[0])
|
|
// The PSTATE fields keep their dedicated immediate form.
|
|
if base, ok := a64MSROps[operandRegName(ops[1])]; ok {
|
|
if v < 0 || v > 15 {
|
|
return nil, fmt.Errorf("MSR: immediate %d out of range (0..15)", v)
|
|
}
|
|
return a64wordLE(base | uint32(v)<<8 | 31), nil
|
|
}
|
|
// A $0 against a full system register writes it from ZR, exactly
|
|
// the way the toolchain preprocesses the constant away; any other
|
|
// immediate is the PSTATE-form error.
|
|
if v != 0 {
|
|
return nil, fmt.Errorf("MSR: illegal PSTATE field for immediate move: %q", operandRegName(ops[1]))
|
|
}
|
|
reg, ok := a64SysRegs[operandRegName(ops[1])]
|
|
if !ok {
|
|
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
|
}
|
|
if !reg.write {
|
|
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
|
|
}
|
|
return a64wordLE(0xd5100000 | reg.v | 31), nil
|
|
}
|
|
// Register form: MSR Rn, <sysreg>.
|
|
reg, ok := a64SysRegs[operandRegName(ops[1])]
|
|
if !ok {
|
|
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
|
}
|
|
if !reg.write {
|
|
return nil, fmt.Errorf("MSR: system register is not writable: %q", operandRegName(ops[1]))
|
|
}
|
|
rs := arm64RegNum(operandRegName(ops[0]))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("MSR: invalid source register")
|
|
}
|
|
return a64wordLE(0xd5100000 | reg.v | uint32(rs)&31), nil
|
|
case "PRFM":
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("PRFM expects (Rn), $immediate|<op>")
|
|
}
|
|
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
|
|
if rn < 0 || off < 0 || off%8 != 0 || off/8 >= 4096 {
|
|
return nil, fmt.Errorf("PRFM: invalid memory operand")
|
|
}
|
|
var prfop int64
|
|
if isImmOperand(ops[1]) {
|
|
prfop = arm64Imm64(ops[1])
|
|
if prfop < 0 || prfop > 31 {
|
|
return nil, fmt.Errorf("PRFM: immediate %d out of range (0..31)", prfop)
|
|
}
|
|
} else {
|
|
p, ok := a64PRFOps[operandRegName(ops[1])]
|
|
if !ok {
|
|
return nil, fmt.Errorf("PRFM: unknown prefetch operation %q", operandRegName(ops[1]))
|
|
}
|
|
prfop = int64(p)
|
|
}
|
|
return a64wordLE(0xf9800000 | uint32(off/8)<<10 | uint32(rn)<<5 | uint32(prfop)), nil
|
|
case "RPRFM":
|
|
// RPRFM (Rn), Rm, <op|$imm6>: the 6-bit operation scatters across
|
|
// bits 15, 13, 12 and 2:0 (asm7.go case 110).
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("RPRFM expects (Rn), Rm, <op|$immediate>")
|
|
}
|
|
rn, off := arm64MemWithFrame(ops[0], arm64FrameInfo{})
|
|
if rn < 0 || off != 0 {
|
|
return nil, fmt.Errorf("RPRFM: invalid memory operand")
|
|
}
|
|
rm := arm64RegNum(operandRegName(ops[1]))
|
|
if rm < 0 {
|
|
return nil, fmt.Errorf("RPRFM: invalid register operand")
|
|
}
|
|
// The toolchain's class ladder takes a plain R register alone: RSP
|
|
// and ZR are illegal combinations (asm7.go case 110).
|
|
if name := operandRegName(ops[1]); strings.EqualFold(name, "RSP") || strings.EqualFold(name, "SP") || strings.EqualFold(name, "ZR") {
|
|
return nil, fmt.Errorf("RPRFM: illegal combination: %s is not a general register", name)
|
|
}
|
|
var op uint64
|
|
if isImmOperand(ops[2]) {
|
|
op = uint64(arm64Imm64(ops[2]))
|
|
if op > 63 {
|
|
return nil, fmt.Errorf("RPRFM: range prefetch immediate %d out of range (0..63)", op)
|
|
}
|
|
} else {
|
|
v, ok := a64RPRFOps[operandRegName(ops[2])]
|
|
if !ok {
|
|
return nil, fmt.Errorf("RPRFM: unknown range prefetch operation %q", operandRegName(ops[2]))
|
|
}
|
|
op = uint64(v)
|
|
}
|
|
scatter := (op&(1<<5))<<10 | (op&(1<<4))<<9 | (op&(1<<3))<<9 | op&7
|
|
return a64wordLE(0xf8a04818 | uint32(rm)<<16 | uint32(rn)<<5 | uint32(scatter)), nil
|
|
}
|
|
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
|
}
|
|
|
|
// encodeARM64Crypto encodes the crypto instructions. Two-register forms
|
|
// spell (Rn, Rd), three-register forms (Rm, Rn, Rd); the arrangements, when
|
|
// spelled, must match the instruction's own (B16 for AES, S4 for the SHA1
|
|
// and SHA256 families, D2 for SHA512).
|
|
func encodeARM64Crypto(mnem string, baseOp uint32, ops []*ast.Operand, n int) ([]byte, error) {
|
|
if len(ops) != n {
|
|
return nil, fmt.Errorf("%s expects %d operands, got %d", mnem, n, len(ops))
|
|
}
|
|
vs := make([]a64Vec, n)
|
|
for i, op := range ops {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
if v.arr != "" && a64ArrIndex(v.arr) != a64CryptoArr[mnem] {
|
|
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, v.arr)
|
|
}
|
|
vs[i] = v
|
|
}
|
|
if n == 2 {
|
|
return a64wordLE(baseOp | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
|
}
|
|
return a64wordLE(baseOp | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
|
}
|
|
|
|
// encodeARM64MoveWide encodes a standalone MOVK: ($value, Rd) with the value
|
|
// sitting in one 16-bit chunk, the chunk's position becoming the hw field.
|
|
func encodeARM64MoveWide(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects $value, Rd", mnem)
|
|
}
|
|
rd := arm64RegNum(operandRegName(ops[1]))
|
|
if rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// The opcode (MOVN 0, MOVZ 2, MOVK 3) and the width ride the table
|
|
// base, so MOVZ and MOVN come along for free.
|
|
opc := baseOp >> 29 & 3
|
|
sf := baseOp >> 31 & 1
|
|
// The toolchain's optab case 33, shared by the whole family in both
|
|
// widths: the immediate is one unsigned 64-bit pattern (a high-lane
|
|
// constant such as $(40000<<48) arrives negative through int64
|
|
// folding), it must occupy exactly one 16-bit lane, zero is rejected,
|
|
// and the W forms cannot reach the top half.
|
|
u := uint64(arm64Imm64(ops[0]))
|
|
if u == 0 {
|
|
return nil, fmt.Errorf("%s: zero immediate cannot be handled", mnem)
|
|
}
|
|
hw := -1
|
|
for lane := range 4 {
|
|
if u&^(uint64(0xFFFF)<<(lane*16)) == 0 {
|
|
hw = lane
|
|
break
|
|
}
|
|
}
|
|
if hw < 0 {
|
|
return nil, fmt.Errorf("%s: immediate %#x does not fit one 16-bit chunk", mnem, u)
|
|
}
|
|
if sf == 0 && hw > 1 {
|
|
return nil, fmt.Errorf("%s: immediate %#x out of range for the 32-bit form", mnem, u)
|
|
}
|
|
return a64wordLE(a64MoveWide(sf, opc, uint32(hw), uint32(u>>uint(hw*16)&0xFFFF), uint32(rd))), nil
|
|
}
|
|
|
|
// ---- Bitfield/EXTR encoding ----
|
|
|
|
// encodeARM64Bitfield encodes a bitfield instruction.
|
|
// BFI/BFXIL/SBFM/UBFM $immr, Rn, $imms, Rd → 4 operands
|
|
func encodeARM64Bitfield(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
// BFI/BFXIL/SBFM/UBFM: 4 operands ($immr, Rn, $imms, Rd)
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
immr := arm64Imm64(ops[0])
|
|
rn := arm64RegNum(operandRegName(ops[1]))
|
|
imms := arm64Imm64(ops[2])
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// The toolchain rejects bit numbers at or above the operand width, which
|
|
// sf (bit 31 of the base) selects: 64 when set, 32 otherwise.
|
|
width := uint32(32) << (baseOp >> 31 & 1)
|
|
if immr < 0 || uint32(immr) >= width || imms < 0 || uint32(imms) >= width {
|
|
return nil, fmt.Errorf("%s: bit number out of range (immr=%d imms=%d, width=%d)", mnem, immr, imms, width)
|
|
}
|
|
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// encodeARM64Extr encodes an EXTR instruction.
|
|
// EXTR $lsb, Rm, Rn, Rd → 4 operands
|
|
func encodeARM64Extr(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
lsb := arm64Imm64(ops[0])
|
|
rm := arm64RegNum(operandRegName(ops[1]))
|
|
rn := arm64RegNum(operandRegName(ops[2]))
|
|
rd := arm64RegNum(operandRegName(ops[3]))
|
|
if rm < 0 || rn < 0 || rd < 0 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
// The imms field is 6 bits and must stay below the operand width, which
|
|
// sf (bit 31 of the base) selects: 64 when set, 32 otherwise.
|
|
width := int64(32) << (baseOp >> 31 & 1)
|
|
if lsb < 0 || lsb >= width {
|
|
return nil, fmt.Errorf("%s: bit number %d out of range (width=%d)", mnem, lsb, width)
|
|
}
|
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(lsb)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
|
}
|
|
|
|
// ---- SIMD/NEON encoding ----
|
|
|
|
// arm64VecOf parses a vector operand. The element suffix of V13.S[0] does
|
|
// not survive into the symbol name, so the verbatim operand text is tried
|
|
// first and the register name second.
|
|
func arm64VecOf(op *ast.Operand) (a64Vec, bool) {
|
|
if v, ok := a64VecReg(op.Raw); ok {
|
|
return v, true
|
|
}
|
|
return a64VecReg(operandRegName(op))
|
|
}
|
|
|
|
// arm64SimdHasElement reports whether any operand carries a lane index such
|
|
// as V13.S[0].
|
|
func arm64SimdHasElement(ops []*ast.Operand) bool {
|
|
for _, op := range ops {
|
|
if v, ok := arm64VecOf(op); ok && v.hasIdx {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// arm64SimdArrs validates that a SIMD operand run spells one arrangement,
|
|
// that it is the same on every operand that spells one, and that the table
|
|
// admits it. It returns the arrangement's index, with a64Arr8B for a bare
|
|
// V/F spelling.
|
|
func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
|
|
sel := -1
|
|
for _, a := range arrs {
|
|
if a == "" {
|
|
continue
|
|
}
|
|
i := a64ArrIndex(a)
|
|
if i < 0 || specBit(i)&allowed == 0 {
|
|
return 0, fmt.Errorf("%s: invalid arrangement %q", mnem, a)
|
|
}
|
|
if sel >= 0 && sel != i {
|
|
return 0, fmt.Errorf("%s: mixed arrangements", mnem)
|
|
}
|
|
sel = i
|
|
}
|
|
if sel < 0 {
|
|
sel = a64Arr8B
|
|
}
|
|
return sel, nil
|
|
}
|
|
|
|
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
|
|
func specBit(i int) uint16 { return 1 << uint(i) }
|
|
|
|
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
|
|
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
|
|
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
|
|
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
|
|
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
|
|
return true
|
|
}
|
|
if !strings.HasPrefix(mnem, "VFCM") {
|
|
return false
|
|
}
|
|
s := strings.Join(strings.Fields(op.Raw), "")
|
|
s = strings.TrimPrefix(s, "$")
|
|
s = strings.Trim(s, "()")
|
|
return s == "0" || s == "0.0"
|
|
}
|
|
|
|
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
|
|
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
|
|
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
|
|
// their compare-against-zero form instead, and the polynomial multiplies read
|
|
// the arrangement from their source operands alone, the result spelling
|
|
// (H8, Q1) riding no encoding bits.
|
|
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
|
if mnem == "VPMULL" || mnem == "VPMULL2" {
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
vs := make([]a64Vec, 2)
|
|
arrs := make([]string, 2)
|
|
for i, op := range ops[:2] {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
vs[i], arrs[i] = v, v.arr
|
|
}
|
|
if _, ok := arm64VecOf(ops[2]); !ok {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
rd, _ := arm64VecOf(ops[2])
|
|
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
|
|
}
|
|
if len(ops) == 3 && isImmOperand(ops[0]) {
|
|
base, ok := a64SimdVZero[mnem]
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
|
}
|
|
if !arm64SimdZeroImm(mnem, ops[0]) {
|
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
|
}
|
|
vn, ok1 := arm64VecOf(ops[1])
|
|
vd, ok2 := arm64VecOf(ops[2])
|
|
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
allowed := uint16(0x7f)
|
|
if strings.HasPrefix(mnem, "VFCM") {
|
|
allowed = fpSimdArrs
|
|
}
|
|
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
}
|
|
// The toolchain's two-operand spellings VADD/VSUB Vm, Vn accumulate Vn
|
|
// with Vm in place (asm7.go case 89, r defaulting to rt). They exist
|
|
// for bare V registers alone: the arranged forms and every other
|
|
// three-register mnemonic are rejected outright.
|
|
if len(ops) == 2 && (mnem == "VADD" || mnem == "VSUB") {
|
|
vm, ok1 := arm64VecOf(ops[0])
|
|
vn, ok2 := arm64VecOf(ops[1])
|
|
if !ok1 || !ok2 || vm.hasIdx || vn.hasIdx || vm.arr != "" || vn.arr != "" {
|
|
return nil, fmt.Errorf("%s: two-operand form takes bare V registers", mnem)
|
|
}
|
|
base := uint32(0x5ee08400) // VADD
|
|
if mnem == "VSUB" {
|
|
base = 0x7ee08400
|
|
}
|
|
return a64wordLE(base | uint32(vm.reg)<<16 | uint32(vn.reg)<<5 | uint32(vn.reg)), nil
|
|
}
|
|
if len(ops) != 3 {
|
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
|
}
|
|
vs := make([]a64Vec, 3)
|
|
arrs := make([]string, 3)
|
|
for i, op := range ops {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
vs[i], arrs[i] = v, v.arr
|
|
}
|
|
// The bare three-register spellings of VADD/VSUB are the scalar D forms
|
|
// (the toolchain's ADD/SUB scalar rows), not the 8B vector rows.
|
|
if (mnem == "VADD" || mnem == "VSUB") &&
|
|
arrs[0] == "" && arrs[1] == "" && arrs[2] == "" {
|
|
base := uint32(0x5ee08400) // VADD scalar
|
|
if mnem == "VSUB" {
|
|
base = 0x7ee08400
|
|
}
|
|
return a64wordLE(base | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
|
}
|
|
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
arrBits := a64ArrBits[arr]
|
|
if spec.fp {
|
|
// The FP rows carry a one-bit size field (S=0, D=1) instead of the
|
|
// integer size, and no Q-only masking applies to them.
|
|
arrBits = a64FPArrBits[arr]
|
|
} else if spec.fixed {
|
|
arrBits = 0
|
|
} else if a64SimdQOnly[mnem] {
|
|
arrBits &= 1 << 30
|
|
}
|
|
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
|
}
|
|
|
|
// encodeARM64SimdV2 encodes an arrangement-aware two-register SIMD
|
|
// instruction: word = base | arrBits | Rn<<5 | Rd. VMOV is the exception:
|
|
// its register pair spelling ORRs the source with itself into the
|
|
// destination (base | Rm<<16 | Rn<<5 | Rd with Rm = Rn = source).
|
|
func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
vs := make([]a64Vec, 2)
|
|
arrs := make([]string, 2)
|
|
for i, op := range ops {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
vs[i], arrs[i] = v, v.arr
|
|
}
|
|
if mnem == "VMOV" {
|
|
if (vs[0].arr != "" && vs[1].arr != "") && vs[0].arr != vs[1].arr {
|
|
return nil, fmt.Errorf("%s: mixed arrangements", mnem)
|
|
}
|
|
arr, err := arm64SimdArrs(mnem, arrs, spec.arrs)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
arrBits := a64ArrBits[arr]
|
|
if a64SimdQOnly[mnem] {
|
|
arrBits &= 1 << 30
|
|
}
|
|
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
|
}
|
|
// VUADDLV spells its arrangement on the source alone; the rest take it
|
|
// on both.
|
|
vn, ok1 := arm64VecOf(ops[0])
|
|
vd, ok2 := arm64VecOf(ops[1])
|
|
if !ok1 || !ok2 {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, spec.arrs)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
arrBits := a64ArrBits[arr]
|
|
if spec.fp {
|
|
// The FP rows carry the one-bit FP size field instead of the
|
|
// integer size, and no Q-only masking applies to them.
|
|
arrBits = a64FPArrBits[arr]
|
|
} else if a64SimdQOnly[mnem] {
|
|
arrBits &= 1 << 30
|
|
}
|
|
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
}
|
|
|
|
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
|
|
// ops ride Rm, Sa, Rn, Rd at 16, 10, 5 and 0) and its immediate relatives
|
|
// (VXAR with a 6-bit rotation at bits 15:10, VEXT with the index at bits
|
|
// 15:11 and a B16 flag at bit 30).
|
|
func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, error) {
|
|
switch mnem {
|
|
case "VEOR3", "VBCAX":
|
|
if len(ops) != 4 {
|
|
return nil, fmt.Errorf("%s expects 4 operands, got %d", mnem, len(ops))
|
|
}
|
|
vs := make([]a64Vec, 4)
|
|
arrs := make([]string, 4)
|
|
for i, op := range ops {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
vs[i], arrs[i] = v, v.arr
|
|
}
|
|
if _, err := arm64SimdArrs(mnem, arrs, 1<<a64Arr16B|1<<a64Arr8B); err != nil {
|
|
return nil, err
|
|
}
|
|
// The first source rides the opcode's Sa field at bits 15:10, the
|
|
// second Rm at bits 20:16, then Rn and Rd.
|
|
return a64wordLE(base | uint32(vs[1].reg)<<16 | uint32(vs[0].reg)<<10 | uint32(vs[2].reg)<<5 | uint32(vs[3].reg)), nil
|
|
case "VXAR":
|
|
if len(ops) != 4 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects 4 operands ($rotation, Vn, Vm, Vd)", mnem)
|
|
}
|
|
rot := arm64Imm64(ops[0])
|
|
if rot < 0 || rot > 63 {
|
|
return nil, fmt.Errorf("%s: rotation %d out of range (0..63)", mnem, rot)
|
|
}
|
|
vs := make([]a64Vec, 3)
|
|
arrs := make([]string, 3)
|
|
for i, op := range ops[1:] {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
vs[i], arrs[i] = v, v.arr
|
|
}
|
|
if _, err := arm64SimdArrs(mnem, arrs, 1<<a64Arr2D|1<<a64ArrD1); err != nil {
|
|
return nil, err
|
|
}
|
|
return a64wordLE(base | uint32(rot)<<10 | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
|
case "VEXT":
|
|
if len(ops) != 4 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects 4 operands ($index, Vn, Vm, Vd)", mnem)
|
|
}
|
|
idx := arm64Imm64(ops[0])
|
|
vs := make([]a64Vec, 3)
|
|
arrs := make([]string, 3)
|
|
for i, op := range ops[1:] {
|
|
v, ok := arm64VecOf(op)
|
|
if !ok || v.hasIdx {
|
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
|
}
|
|
vs[i], arrs[i] = v, v.arr
|
|
}
|
|
arr, err := arm64SimdArrs(mnem, arrs, 1<<a64Arr8B|1<<a64Arr16B)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
max := 7
|
|
b16 := uint32(0)
|
|
if arr == a64Arr16B {
|
|
max = 15
|
|
b16 = 1 << 30
|
|
}
|
|
if idx < 0 || idx > int64(max) {
|
|
return nil, fmt.Errorf("%s: index %d out of range (0..%d)", mnem, idx, max)
|
|
}
|
|
return a64wordLE(base | b16 | uint32(idx)<<11 | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
|
}
|
|
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
|
}
|
|
|
|
// encodeARM64VTBL encodes VTBL Vidx.arr, [Vt1.arr, ...], Vdest.arr: the
|
|
// index register rides bits 19:16, the first table register bits 9:5, the
|
|
// destination bits 4:0 and the table length (registers minus one) bits
|
|
// 14:13. The table registers must be consecutive.
|
|
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) < 3 {
|
|
return nil, fmt.Errorf("VTBL expects index, table list and destination")
|
|
}
|
|
vi, ok := arm64VecOf(ops[0])
|
|
if !ok || vi.hasIdx {
|
|
return nil, fmt.Errorf("invalid index register in VTBL")
|
|
}
|
|
ts, end, ok := a64VecListOf(ops, 1)
|
|
if !ok || len(ts) < 1 || len(ts) > 4 {
|
|
return nil, fmt.Errorf("VTBL expects a table of one to four registers")
|
|
}
|
|
// The list may close on the last operand, leaving no destination: bound
|
|
// the index before reading it.
|
|
if end+1 >= len(ops) {
|
|
return nil, fmt.Errorf("%s expects a destination register after the table list", mnem)
|
|
}
|
|
vd, ok := a64VecReg(operandRegName(ops[end+1]))
|
|
if !ok || end+2 != len(ops) || vd.hasIdx {
|
|
return nil, fmt.Errorf("invalid destination register in VTBL")
|
|
}
|
|
for i, t := range ts {
|
|
if t.hasIdx || (ts[0].reg+i)&31 != t.reg {
|
|
return nil, fmt.Errorf("VTBL table registers must be consecutive")
|
|
}
|
|
}
|
|
q := uint32(0)
|
|
switch vi.arr {
|
|
case "B16":
|
|
q = 1 << 30
|
|
case "B8", "":
|
|
default:
|
|
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
|
|
}
|
|
base := uint32(0x0e000000)
|
|
if mnem == "VTBX" {
|
|
base |= 1 << 12
|
|
}
|
|
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
|
}
|
|
|
|
// encodeARM64GPToVec encodes the whole-vector move VMOV/VDUP Rs, Vd.<T>: a
|
|
// general register into an arranged vector, the spelling asm7.go's case 82
|
|
// calls vmov/vdup Rn, Vd.<T>. ok is false for anything that is not that
|
|
// shape, so the caller falls through to the arrangement and element paths;
|
|
// the toolchain rejects the bare spellings outright, and the reverse
|
|
// Vd.<T>, Rs with them.
|
|
func encodeARM64GPToVec(mnem string, ops []*ast.Operand) ([]byte, bool, error) {
|
|
if len(ops) != 2 {
|
|
return nil, false, nil
|
|
}
|
|
if ops[0].Addr.Base != "" || isImmOperand(ops[0]) {
|
|
return nil, false, nil
|
|
}
|
|
rs := arm64RegNum(operandRegName(ops[0]))
|
|
if rs < 0 {
|
|
return nil, false, nil
|
|
}
|
|
dst, ok := arm64VecOf(ops[1])
|
|
if !ok || dst.hasIdx || dst.arr == "" {
|
|
return nil, false, nil
|
|
}
|
|
b, err := a64GPVecWhole(mnem, rs, dst)
|
|
return b, true, err
|
|
}
|
|
|
|
// a64GPVecWhole lays down the general-register-into-a-whole-vector move:
|
|
// word = Q | 7<<25 | imm5<<16 | 3<<10 | rs<<5 | rd, with imm5 naming the
|
|
// lane width and Q the vector length. Both VMOV and VDUP take this form
|
|
// (asm7.go case 82); INS-into-one-lane is encoded elsewhere.
|
|
func a64GPVecWhole(mnem string, rs int, dst a64Vec) ([]byte, error) {
|
|
var imm5, q uint32
|
|
switch dst.arr {
|
|
case "B8":
|
|
imm5, q = 1, 0
|
|
case "B16":
|
|
imm5, q = 1, 1<<30
|
|
case "H4":
|
|
imm5, q = 2, 0
|
|
case "H8":
|
|
imm5, q = 2, 1<<30
|
|
case "S2":
|
|
imm5, q = 4, 0
|
|
case "S4":
|
|
imm5, q = 4, 1<<30
|
|
case "D2":
|
|
imm5, q = 8, 1<<30
|
|
default:
|
|
// D1 rides no case-82 row: the toolchain rejects the one-doubleword
|
|
// spelling for this form, so the encoder refuses it too.
|
|
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
|
}
|
|
return a64wordLE(q | 0x0e000c00 | imm5<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
|
}
|
|
|
|
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
|
// lane indices:
|
|
//
|
|
// Vn.<T>[i], Rd UMOV, element to general register
|
|
// Vn.<T>[i], Vd.arr DUP, element across a vector
|
|
// Vn.<T>[i], Vd.<T>[j] INS, element to element
|
|
// Rs, Vd.<T>[i] INS, general register into an element
|
|
func encodeARM64Dup(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
|
}
|
|
dst, dstVec := arm64VecOf(ops[1])
|
|
dstGP := false
|
|
if !dstVec {
|
|
// A general-register spelling as destination (UMOV forms).
|
|
if rd := arm64RegNum(operandRegName(ops[1])); rd >= 0 {
|
|
dst, dstGP, dstVec = a64Vec{reg: rd}, true, true
|
|
}
|
|
}
|
|
if !dstVec {
|
|
return nil, fmt.Errorf("%s: invalid destination operand", mnem)
|
|
}
|
|
src, ok1 := arm64VecOf(ops[0])
|
|
if (!ok1 || !src.hasIdx) && !dst.hasIdx {
|
|
return nil, fmt.Errorf("%s expects an element operand Vn.<T>[i]", mnem)
|
|
}
|
|
if !ok1 || !src.hasIdx {
|
|
// General register into a vector. An arranged destination without a
|
|
// lane index duplicates the register across every lane (DUP Vd.T,
|
|
// Rn); an indexed one is an INS into that single lane.
|
|
rs := arm64RegNum(operandRegName(ops[0]))
|
|
if rs < 0 {
|
|
return nil, fmt.Errorf("%s: source must be a general register", mnem)
|
|
}
|
|
if !dst.hasIdx {
|
|
// Duplicates the register across every lane (DUP Vd.T, Rn).
|
|
return a64GPVecWhole(mnem, rs, dst)
|
|
}
|
|
f, ok := a64ElemField(dst.arr, dst.idx)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
|
}
|
|
return a64wordLE(0x4e001c00 | f<<16 | uint32(rs)<<5 | uint32(dst.reg)), nil
|
|
}
|
|
sf, ok := a64ElemField(src.arr, src.idx)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
|
}
|
|
if dst.hasIdx {
|
|
// Element to element: the toolchain requires the two element letters
|
|
// to match (asm7.go case 92), packs the destination index into imm5
|
|
// and the source index, in units of the element size, into imm4.
|
|
if src.arr != dst.arr {
|
|
return nil, fmt.Errorf("%s: operand mismatch: %s and %s elements", mnem, src.arr, dst.arr)
|
|
}
|
|
df, ok := a64ElemField(dst.arr, dst.idx)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
|
}
|
|
var imm4 uint32
|
|
switch src.arr {
|
|
case "B":
|
|
imm4 = uint32(src.idx)
|
|
case "H":
|
|
imm4 = uint32(src.idx) << 1
|
|
case "S":
|
|
imm4 = uint32(src.idx) << 2
|
|
case "D":
|
|
imm4 = uint32(src.idx) << 3
|
|
default:
|
|
return nil, fmt.Errorf("%s: invalid element operand", mnem)
|
|
}
|
|
return a64wordLE(0x6e000400 | df<<16 | imm4&0xf<<11 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
|
}
|
|
if dstGP {
|
|
// Element to a general register: UMOV, with the D form setting bit
|
|
// 30.
|
|
base := uint32(0x0e003c00)
|
|
if src.arr == "D" {
|
|
base = 0x4e003c00
|
|
}
|
|
return a64wordLE(base | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
|
}
|
|
if dst.arr == "" {
|
|
// Element across a bare V register.
|
|
return a64wordLE(0x5e000400 | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
|
}
|
|
// Element across an arranged vector; the 128-bit arrangements set bit
|
|
// 30.
|
|
q := uint32(0)
|
|
switch dst.arr {
|
|
case "B16", "H8", "S4", "D2":
|
|
q = 1 << 30
|
|
case "B8", "H4", "S2", "D1":
|
|
default:
|
|
return nil, fmt.Errorf("%s: invalid destination arrangement %q", mnem, dst.arr)
|
|
}
|
|
return a64wordLE(0x0e000400 | q | sf<<16 | uint32(src.reg)<<5 | uint32(dst.reg)), nil
|
|
}
|
|
|
|
// encodeARM64VLDST encodes the SIMD structure loads and stores:
|
|
//
|
|
// VLD1|2|3|4 (Rn), [Vt.arr, ...] VST1|2|3|4 [Vt.arr, ...], (Rn)
|
|
// VLD1|2|3|4.P off(Rn), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], off(Rn)
|
|
// VLD1|2|3|4.P (Rn)(Rm), [Vt.arr, ...] VST1|2|3|4.P [Vt.arr, ...], (Rn)(Rm)
|
|
// VLD1|2|3|4R (Rn), [Vt.arr, ...] (replicating loads)
|
|
// VLD1 off(Rn), Vt.T[i] VST1 Vt.T[i], off(Rn) (one lane)
|
|
//
|
|
// The post-index forms set the post bit and carry Rm: 11111 for an immediate
|
|
// increment, the spelled register for (Rn)(Rm). A register list may wrap
|
|
// around V31: the toolchain checks only (first+i) mod 32. A spelled offset
|
|
// rides along on the one-lane forms (the toolchain only checks that it
|
|
// matches the access size).
|
|
func encodeARM64VLDST(mnem string, post uint32, ops []*ast.Operand) ([]byte, error) {
|
|
load := strings.HasPrefix(mnem, "VLD")
|
|
|
|
// One-lane forms spell a single Vt.T[i] operand, not a bracketed list.
|
|
laneIdx := 1
|
|
if !load {
|
|
laneIdx = 0
|
|
}
|
|
if len(ops) > laneIdx {
|
|
if v, ok := arm64VecOf(ops[laneIdx]); ok && v.hasIdx {
|
|
return encodeARM64VLDSTLane(mnem, post, ops, load, laneIdx, v)
|
|
}
|
|
}
|
|
|
|
listStart, memAt := 0, 1
|
|
if load {
|
|
// Every load spells the memory operand first.
|
|
listStart, memAt = 1, 0
|
|
}
|
|
vs, end, ok := a64VecListOf(ops, listStart)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid register list", mnem)
|
|
}
|
|
memIdx := end + 1
|
|
if memAt == 0 {
|
|
memIdx = 0
|
|
}
|
|
if memIdx >= len(ops) {
|
|
return nil, fmt.Errorf("%s expects a (Rn) memory operand", mnem)
|
|
}
|
|
rn, off := arm64MemWithFrame(ops[memIdx], arm64FrameInfo{})
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
|
|
}
|
|
if off != 0 && post == 0 {
|
|
return nil, fmt.Errorf("%s: offset %d not supported, plain and list accesses take a plain (Rn) operand", mnem, off)
|
|
}
|
|
// The toolchain's addressing contract for the structure forms (asm7.go's
|
|
// class ladder): an index register exists only as the post-index
|
|
// increment, so it is illegal without .P, and the immediate increment
|
|
// must be exactly the bytes the whole list transfers.
|
|
idx := ops[memIdx].Addr.Index
|
|
if idx != "" && post == 0 {
|
|
return nil, fmt.Errorf("%s: illegal combination: the register index is a post-index, it needs the .P spelling", mnem)
|
|
}
|
|
if idx != "" && (ops[memIdx].Addr.Shift != "" || ops[memIdx].Addr.HasOff) {
|
|
return nil, fmt.Errorf("%s: invalid extended register op: the post-index register takes no offset, extend or shift", mnem)
|
|
}
|
|
// The post-index increment: 11111 for an immediate offset, else the
|
|
// spelled (Rn)(Rm) register.
|
|
rm := 31
|
|
if post != 0 {
|
|
if idx != "" {
|
|
if rm = arm64RegNum(idx); rm < 0 {
|
|
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
|
|
}
|
|
}
|
|
}
|
|
|
|
// The replicating loads: VLD1R through VLD4R load one register and
|
|
// replicate it across the whole list.
|
|
if base := strings.TrimSuffix(mnem, ".P"); load && strings.HasSuffix(base, "R") && len(base) == 5 {
|
|
n := int(base[3] - '0')
|
|
if len(vs) != n {
|
|
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
|
|
}
|
|
size, q, ok := a64ArrSizeQ(vs[0].arr)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
|
|
}
|
|
// The immediate post-increment transfers one element per register,
|
|
// not a whole register (VLD3R.P 6(R15), [V15.H4,V16.H4,V17.H4]).
|
|
// An unspelled offset (or a spelled zero, the same class) is the
|
|
// implicit by-size increment the Rm=11111 encoding carries; any
|
|
// other spelled value must match.
|
|
if post != 0 && idx == "" && off != 0 && off != int64(n)*(1<<uint(size)) {
|
|
return nil, fmt.Errorf("%s: invalid post-increment offset %d, want %d", mnem, off, n*(1<<uint(size)))
|
|
}
|
|
w := a64VLDNReplicate[n] | q<<30 | size<<10 | uint32(rn)<<5 | uint32(vs[0].reg)
|
|
if post != 0 {
|
|
w |= 1<<23 | uint32(rm)<<16
|
|
}
|
|
return a64wordLE(w), nil
|
|
}
|
|
|
|
if len(vs) < 1 || len(vs) > 4 {
|
|
return nil, fmt.Errorf("%s expects a list of one to four registers", mnem)
|
|
}
|
|
for i, v := range vs {
|
|
if v.hasIdx || (vs[0].reg+i)&31 != v.reg {
|
|
return nil, fmt.Errorf("%s: register list must be consecutive", mnem)
|
|
}
|
|
_, _, okArr := a64ArrSizeQ(v.arr)
|
|
if !okArr || (i > 0 && v.arr != vs[0].arr) {
|
|
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, v.arr)
|
|
}
|
|
}
|
|
size, q, ok := a64ArrSizeQ(vs[0].arr)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vs[0].arr)
|
|
}
|
|
n := len(vs)
|
|
// The immediate post-increment moves the whole list: count times the
|
|
// register width (16 bytes in the Q forms, 8 otherwise). An unspelled
|
|
// offset (or a spelled zero) is the implicit by-size increment
|
|
// (Rm=11111); any other spelled value must match.
|
|
if post != 0 && idx == "" && off != 0 {
|
|
regBytes := 8
|
|
if q != 0 {
|
|
regBytes = 16
|
|
}
|
|
if off != int64(n)*int64(regBytes) {
|
|
return nil, fmt.Errorf("%s: invalid post-increment offset %d, want %d", mnem, off, n*regBytes)
|
|
}
|
|
}
|
|
base := a64VLD1Base[n]
|
|
if !load {
|
|
base = a64VST1Base[n]
|
|
}
|
|
// VLD2/VLD3/VLD4 and VST2/VST3/VST4 name the register count in the
|
|
// mnemonic and carry their own opcode fields. The count digit sits at
|
|
// index 3 of the mnemonic (VLD2, VST3.P, ...), before any .P suffix.
|
|
if stem := strings.TrimSuffix(mnem, ".P"); len(stem) >= 4 && stem[3] >= '2' && stem[3] <= '4' {
|
|
n := int(stem[3] - '0')
|
|
if n != len(vs) {
|
|
return nil, fmt.Errorf("%s expects a list of %d registers", mnem, n)
|
|
}
|
|
if load {
|
|
base = a64VLDNBase[n]
|
|
} else {
|
|
base = a64VSTNBase[n]
|
|
}
|
|
}
|
|
postBits := uint32(0)
|
|
if post != 0 {
|
|
postBits = 1<<23 | uint32(rm)<<16
|
|
}
|
|
return a64wordLE(base | q<<30 | size<<10 | postBits | uint32(rn)<<5 | uint32(vs[0].reg)), nil
|
|
}
|
|
|
|
// encodeARM64VLDSTLane encodes the one-lane structure forms:
|
|
// VLD1 off(Rn), Vt.T[i] and VST1 Vt.T[i], off(Rn); the post-index spellings
|
|
// add the post bit and Rm: 11111 for an immediate increment, the spelled
|
|
// register for (Rn)(Rm).
|
|
func encodeARM64VLDSTLane(mnem string, post uint32, ops []*ast.Operand, load bool, laneIdx int, v a64Vec) ([]byte, error) {
|
|
if len(ops) != 2 {
|
|
return nil, fmt.Errorf("%s expects a memory operand and one Vt.T[i] lane operand", mnem)
|
|
}
|
|
memIdx := laneIdx ^ 1
|
|
rn, _ := arm64MemWithFrame(ops[memIdx], arm64FrameInfo{})
|
|
if rn < 0 {
|
|
return nil, fmt.Errorf("%s: invalid memory operand", mnem)
|
|
}
|
|
rm := 31
|
|
if post != 0 {
|
|
if idx := ops[memIdx].Addr.Index; idx != "" {
|
|
if rm = arm64RegNum(idx); rm < 0 {
|
|
return nil, fmt.Errorf("%s: invalid post-index register %q", mnem, idx)
|
|
}
|
|
}
|
|
}
|
|
w := uint32(0x0d400000)
|
|
switch strings.ToUpper(v.arr) {
|
|
case "B":
|
|
// Index<3> rides bit 30, index<2:0> the size field at bits 12:10.
|
|
w |= uint32(v.idx&7)<<10 | uint32(v.idx>>3&1)<<30
|
|
case "H":
|
|
// Index<2> at bit 30, index<1> at bit 12, index<0> at bit 11.
|
|
w |= 1<<14 | uint32(v.idx&1)<<11 | uint32(v.idx>>1&1)<<12 | uint32(v.idx>>2&1)<<30
|
|
case "S":
|
|
// Index<0> at bit 12, index<1> at bit 30.
|
|
w |= 4<<13 | uint32(v.idx&1)<<12 | uint32(v.idx>>1&1)<<30
|
|
case "D":
|
|
// Index<0> at bit 30, fixed size field 01.
|
|
w |= 4<<13 | 1<<10 | uint32(v.idx&1)<<30
|
|
default:
|
|
return nil, fmt.Errorf("%s: invalid lane arrangement %q", mnem, v.arr)
|
|
}
|
|
// The base carries bit 22 (L) set; a store clears it. The post-index
|
|
// forms add bit 23 and Rm.
|
|
if !load {
|
|
w &^= 1 << 22
|
|
}
|
|
if post != 0 {
|
|
w |= 1<<23 | uint32(rm)<<16
|
|
}
|
|
return a64wordLE(w | uint32(rn)<<5 | uint32(v.reg)), nil
|
|
}
|
|
|
|
// a64ArrSizeQ maps an arrangement to its size code (bits 11:10) and 128-bit
|
|
// flag for the structure load/store words.
|
|
func a64ArrSizeQ(arr string) (size, q uint32, ok bool) {
|
|
switch arr {
|
|
case "B8":
|
|
return 0, 0, true
|
|
case "B16":
|
|
return 0, 1, true
|
|
case "H4":
|
|
return 1, 0, true
|
|
case "H8":
|
|
return 1, 1, true
|
|
case "S2":
|
|
return 2, 0, true
|
|
case "S4":
|
|
return 2, 1, true
|
|
case "D1":
|
|
return 3, 0, true
|
|
case "D2":
|
|
return 3, 1, true
|
|
}
|
|
return 0, 0, false
|
|
}
|
|
|
|
// encodeARM64ShiftImm encodes a SIMD shift by immediate:
|
|
// word = base | Q<<30 | immh:immb<<16 | Rn<<5 | Rd, where immh:immb is the
|
|
// element size plus the shift for a left shift (VSHL) and twice the element
|
|
// size minus the shift for right shifts (VUSHR, VSRI).
|
|
func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte, error) {
|
|
if len(ops) != 3 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects 3 operands ($shift, Vn.arr, Vd.arr)", mnem)
|
|
}
|
|
sh := arm64Imm64(ops[0])
|
|
vn, ok1 := arm64VecOf(ops[1])
|
|
vd, ok2 := arm64VecOf(ops[2])
|
|
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx || vn.arr != vd.arr {
|
|
return nil, fmt.Errorf("%s: operands must share one arrangement", mnem)
|
|
}
|
|
var esize int64
|
|
q := uint32(0)
|
|
switch vn.arr {
|
|
case "B8", "B":
|
|
esize = 8
|
|
case "B16":
|
|
esize, q = 8, 1
|
|
case "H4", "H":
|
|
esize = 16
|
|
case "H8":
|
|
esize, q = 16, 1
|
|
case "S2", "S":
|
|
esize = 32
|
|
case "S4":
|
|
esize, q = 32, 1
|
|
case "D2":
|
|
esize, q = 64, 1
|
|
default:
|
|
return nil, fmt.Errorf("%s: invalid arrangement %q", mnem, vn.arr)
|
|
}
|
|
var immval int64
|
|
switch mnem {
|
|
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
|
|
if sh < 0 || sh >= esize {
|
|
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
|
}
|
|
immval = esize + sh
|
|
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
|
|
if sh < 1 || sh > esize {
|
|
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
|
|
}
|
|
immval = 2*esize - sh
|
|
}
|
|
return a64wordLE(base | q<<30 | uint32(immval)<<16 | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
|
}
|
|
|
|
// encodeARM64MoviLit loads a large vector constant the way the toolchain
|
|
// does: ADRP R27 and ADD materialise the literal's address, then FMOVS,
|
|
// FMOVD or the 128-bit FMOVQ form loads it, with R_ADDRARM64 relocations
|
|
// against a read-only literal the file assembler lays out.
|
|
func encodeARM64MoviLit(mnem string, ldr uint32, ops []*ast.Operand, relocs *[]Reloc, lits *arm64Literals) ([]byte, error) {
|
|
var vd a64Vec
|
|
var ok bool
|
|
var data []byte
|
|
switch mnem {
|
|
case "VMOVS", "VMOVD":
|
|
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
|
return nil, fmt.Errorf("%s expects $value, Vd", mnem)
|
|
}
|
|
v := arm64Imm64(ops[0])
|
|
if mnem == "VMOVS" {
|
|
if v < -2147483648 || v > 0xFFFFFFFF {
|
|
return nil, fmt.Errorf("%s: constant does not fit 32 bits", mnem)
|
|
}
|
|
data = a64wordLE(uint32(v))
|
|
} else {
|
|
data = a64WordsLE(uint32(v), uint32(v>>32))
|
|
}
|
|
vd, ok = arm64VecOf(ops[1])
|
|
if !ok || vd.hasIdx || vd.arr != "" {
|
|
return nil, fmt.Errorf("%s: destination must be a bare V register", mnem)
|
|
}
|
|
case "VMOVQ":
|
|
if len(ops) != 3 || !isImmOperand(ops[0]) || !isImmOperand(ops[1]) {
|
|
return nil, fmt.Errorf("VMOVQ expects $lo, $hi, Vd")
|
|
}
|
|
lo, hi := arm64Imm64(ops[0]), arm64Imm64(ops[1])
|
|
data = a64WordsLE(uint32(lo), uint32(lo>>32), uint32(hi), uint32(hi>>32))
|
|
vd, ok = arm64VecOf(ops[2])
|
|
if !ok || vd.hasIdx || vd.arr != "" {
|
|
return nil, fmt.Errorf("VMOVQ: destination must be a bare V register")
|
|
}
|
|
default:
|
|
return nil, fmt.Errorf("unsupported arm64 instruction %q", mnem)
|
|
}
|
|
name := lits.add(moviLitName(mnem, data), data)
|
|
if relocs != nil {
|
|
*relocs = append(*relocs,
|
|
Reloc{Off: 0, After: 0, Name: name, Kind: RelArm64Addr},
|
|
Reloc{Off: 4, After: 4, Name: name, Kind: RelArm64Addr},
|
|
)
|
|
}
|
|
return a64WordsLE(
|
|
a64ADR(1, 0, 0, 27),
|
|
a64AddSub(1, 0, 0, 0, 0, 27, 27),
|
|
ldr|0<<10|27<<5|uint32(vd.reg),
|
|
), nil
|
|
}
|
|
|
|
// moviLitName mirrors the toolchain's literal naming: $i32/$i64/$i128
|
|
// followed by the constant's value in hex.
|
|
func moviLitName(mnem string, data []byte) string {
|
|
switch mnem {
|
|
case "VMOVS":
|
|
v := uint32(data[0]) | uint32(data[1])<<8 | uint32(data[2])<<16 | uint32(data[3])<<24
|
|
return "$i32." + strconv.FormatUint(uint64(v), 16)
|
|
case "VMOVD":
|
|
v := uint64(data[0]) | uint64(data[1])<<8 | uint64(data[2])<<16 | uint64(data[3])<<24 |
|
|
uint64(data[4])<<32 | uint64(data[5])<<40 | uint64(data[6])<<48 | uint64(data[7])<<56
|
|
return "$i64." + strconv.FormatUint(v, 16)
|
|
default:
|
|
hi := uint64(data[8]) | uint64(data[9])<<8 | uint64(data[10])<<16 | uint64(data[11])<<24 |
|
|
uint64(data[12])<<32 | uint64(data[13])<<40 | uint64(data[14])<<48 | uint64(data[15])<<56
|
|
lo := uint64(data[0]) | uint64(data[1])<<8 | uint64(data[2])<<16 | uint64(data[3])<<24 |
|
|
uint64(data[4])<<32 | uint64(data[5])<<40 | uint64(data[6])<<48 | uint64(data[7])<<56
|
|
return "$i128." + strings.Repeat("0", max(0, 16-len(strconv.FormatUint(hi, 16)))) +
|
|
strconv.FormatUint(hi, 16) + strings.Repeat("0", max(0, 16-len(strconv.FormatUint(lo, 16)))) +
|
|
strconv.FormatUint(lo, 16)
|
|
}
|
|
}
|
|
|
|
// arm64Literals collects the read-only constants the VMOVS/VMOVD/VMOVQ
|
|
// loads refer to. Names follow the toolchain's $i32/$i64/$i128 spellings so
|
|
// equal constants deduplicate to one literal.
|
|
type arm64Literals struct {
|
|
order []Arm64Literal
|
|
seen map[string]bool
|
|
}
|
|
|
|
// Arm64Literal is one pooled vector constant.
|
|
type Arm64Literal struct {
|
|
Name string
|
|
Data []byte
|
|
}
|
|
|
|
// add registers a literal under its name and returns it.
|
|
func (l *arm64Literals) add(name string, data []byte) string {
|
|
if l.seen == nil {
|
|
l.seen = map[string]bool{}
|
|
}
|
|
if !l.seen[name] {
|
|
l.seen[name] = true
|
|
l.order = append(l.order, Arm64Literal{Name: name, Data: data})
|
|
}
|
|
return name
|
|
}
|
|
|
|
// list returns the literals in first-use order.
|
|
func (l *arm64Literals) list() []Arm64Literal { return l.order }
|
|
|
|
// arm64Pool collects the out-of-range load/store offsets a function pools.
|
|
// The toolchain appends them after the last instruction (asm7.go addpool and
|
|
// flushpool) and reaches them with PC-relative literal loads into REGTMP, and
|
|
// when a reference would leave the displacement bound it drains the pool
|
|
// mid-function behind a branch. The pool therefore holds segments: each
|
|
// holds the entries drained together, an entry resolves against the segment
|
|
// open at its referrer. Entries deduplicate by value alone, whatever width
|
|
// the first referrer selected, and concatenate in first-use order with no
|
|
// alignment padding: the toolchain's roundUp touches its size accounting
|
|
// alone, never the byte stream.
|
|
type arm64Pool struct {
|
|
segs []*arm64PoolSeg // the drained segments in layout order, the open one last
|
|
active int // the segment the current statement's references resolve against
|
|
probe bool // the plan pass probes: harvest the requests, suppress the reach check
|
|
}
|
|
|
|
// arm64PoolSeg is one drained pool segment: the words with their offsets
|
|
// from the segment start and their literal-load widths (0 = LDR W
|
|
// zero-extended, 1 = LDR X for an 8-byte entry), the accounting size the
|
|
// flush condition measures, the image position and the guard word, and the
|
|
// body index the flush follows.
|
|
type arm64PoolSeg struct {
|
|
order []arm64PoolEntry
|
|
seen map[int64]int // pooled value → entry index
|
|
size int // bytes the words occupy
|
|
acct int // the toolchain's size accounting: an 8-byte entry rounds the total up to 8 first, the byte stream never
|
|
start int // the pc of the statement that opened the segment
|
|
base int // image offset of the segment's first byte
|
|
guard int // bytes of the guard word before the words (0 or 4)
|
|
branch bool // the guard word is a branch over the words, not the UNDEF
|
|
line int // the source line the words carry, the flushing statement's
|
|
after int // the body index the flush follows
|
|
}
|
|
|
|
// arm64PoolEntry is one pooled constant: its bytes, its offset from the pool
|
|
// start and the literal-load width its bytes select (0 = LDR W zero-extended,
|
|
// 1 = LDR X for an 8-byte entry). omovlit reads the width off the entry
|
|
// itself, so every referrer of a value loads with the first referrer's
|
|
// width; negative values always take the 8-byte entry, which makes the
|
|
// sign-extended LDRSW load unreachable for this pool.
|
|
type arm64PoolEntry struct {
|
|
data []byte
|
|
off int
|
|
w uint32
|
|
}
|
|
|
|
// open starts a fresh segment and makes it the active one. The fresh
|
|
// segment's after is unassigned: a flush and the closing segment set it, and
|
|
// a segment left open but empty carries no words and no flush.
|
|
func (p *arm64Pool) open() {
|
|
p.segs = append(p.segs, &arm64PoolSeg{after: -1})
|
|
p.active = len(p.segs) - 1
|
|
}
|
|
|
|
// activeSeg returns the segment open now, nil when it holds no entries: the
|
|
// toolchain's checkpool runs only while the pool is open, blitrl non-nil.
|
|
func (p *arm64Pool) activeSeg() *arm64PoolSeg {
|
|
if p.active < len(p.segs) && len(p.segs[p.active].order) > 0 {
|
|
return p.segs[p.active]
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// wordsBase returns the image offset of the active segment's first word, the
|
|
// base the PC-relative literal loads resolve against.
|
|
func (p *arm64Pool) wordsBase() int {
|
|
if p.active >= len(p.segs) {
|
|
return 0
|
|
}
|
|
s := p.segs[p.active]
|
|
return s.base + s.guard
|
|
}
|
|
|
|
// add interns a pooled load/store offset in the active segment and returns
|
|
// its offset from the segment start and the literal-load width (asm7.go
|
|
// addpool): a value inside [0, 0x7FFFFFFF] takes a four-byte word loaded
|
|
// zero-extended, anything else the eight-byte slot a full LDR X reads.
|
|
func (p *arm64Pool) add(v int64) (int, uint32) {
|
|
return p.seg().addEntry(v, false)
|
|
}
|
|
|
|
// add64 interns a pooled displacement of the MOVD $con(R) lowering (asm7.go
|
|
// case 34): the entry takes the eight-byte slot even when the value fits a
|
|
// word, but an existing entry of the same value is shared as it stands, the
|
|
// toolchain's value-only dedup.
|
|
func (p *arm64Pool) add64(v int64) (int, uint32) {
|
|
return p.seg().addEntry(v, true)
|
|
}
|
|
|
|
// seg returns the active segment, opening one on demand: the first reference
|
|
// of a function opens the first segment.
|
|
func (p *arm64Pool) seg() *arm64PoolSeg {
|
|
if p.active >= len(p.segs) {
|
|
p.open()
|
|
}
|
|
return p.segs[p.active]
|
|
}
|
|
|
|
// addEntry creates or reuses the segment's entry for v. Reuse is by value
|
|
// alone; at creation, lacon forces the eight-byte slot and every other
|
|
// requestor takes it only for a value no 32-bit load can carry: omovlit's
|
|
// ADWORD rule `lit != int32(lit) || uint64(lit) != uint32(lit)`.
|
|
func (s *arm64PoolSeg) addEntry(v int64, lacon bool) (int, uint32) {
|
|
if i, ok := s.seen[v]; ok {
|
|
return s.order[i].off, s.order[i].w
|
|
}
|
|
off := s.size
|
|
var data []byte
|
|
var w uint32
|
|
if lacon || v < 0 || v > 0x7FFFFFFF {
|
|
// The toolchain's roundUp before a DWORD: the accounting rounds the
|
|
// total up to eight, the byte stream stays unpadded.
|
|
s.acct = (s.acct + 7) &^ 7
|
|
w = 1 // LDR X
|
|
data = a64WordsLE(uint32(v), uint32(v>>32))
|
|
s.size = off + 8
|
|
s.acct += 8
|
|
} else {
|
|
w = 0 // LDR W, zero-extended
|
|
data = a64wordLE(uint32(v))
|
|
s.size = off + 4
|
|
s.acct += 4
|
|
}
|
|
if s.seen == nil {
|
|
s.seen = map[int64]int{}
|
|
}
|
|
s.seen[v] = len(s.order)
|
|
s.order = append(s.order, arm64PoolEntry{data: data, off: off, w: w})
|
|
return off, w
|
|
}
|
|
|
|
// AssembleFileARM64 assembles every TEXT function of a parsed arm64 file
|
|
// and lays out its static symbols (GLOBL/DATA) in a data section behind the
|
|
// code. SB references in the code are encoded as ADRP pairs with zero
|
|
// immediates; the object-file emitters record R_ADDRARM64 relocations for
|
|
// the linker.
|
|
func AssembleFileARM64(f *ast.File) (*Image, error) {
|
|
// Resolve the file's own simple #define aliases (RARG0 → R0, NR → R9,
|
|
// TEB_error → 0x68, B0 → V0) the way the toolchain's preprocessor does
|
|
// textually. Parameterised macros and multi-line bodies are beyond
|
|
// token substitution and stay untouched.
|
|
arm64ResolveAliases(f)
|
|
|
|
dataSyms, err := collectData(f)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
// The TLS symbols: a GLOBL marked TLSBSS makes the toolchain's aclass
|
|
// key its load off objabi.STLSBSS (asm7.go C_TLSIE), so the loads of
|
|
// these names encode as the local-exec MOVZ with the TLS_LE relocation.
|
|
tlsSyms := map[string]bool{}
|
|
for _, d := range f.Decls {
|
|
g, ok := d.(*ast.Globl)
|
|
if !ok || g.Name == nil {
|
|
continue
|
|
}
|
|
for _, fl := range g.Flags {
|
|
if fl == "TLSBSS" {
|
|
tlsSyms[g.Name.Name] = true
|
|
}
|
|
}
|
|
}
|
|
|
|
img := &Image{Symbols: map[string]int{}, SourcePath: f.Path}
|
|
var pendingLits []Arm64Literal
|
|
litSeen := map[string]bool{}
|
|
for _, d := range f.Decls {
|
|
t, ok := d.(*ast.Text)
|
|
if !ok {
|
|
continue
|
|
}
|
|
code, labels, relocs, lines, spadj, lits, err := assembleARM64(t, tlsSyms)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("%s: %w", t.Name.Name, err)
|
|
}
|
|
// The literals this function's constant loads refer to join the
|
|
// data section once, deduplicated by name.
|
|
for _, lit := range lits {
|
|
if _, seen := litSeen[lit.Name]; seen {
|
|
continue
|
|
}
|
|
litSeen[lit.Name] = true
|
|
pendingLits = append(pendingLits, lit)
|
|
}
|
|
fl := FuncLayout{
|
|
Name: t.Name.Name,
|
|
Pkg: t.Name.Pkg,
|
|
Static: t.Name.Static,
|
|
Offset: len(img.Code),
|
|
Size: len(code),
|
|
Frame: frameSize(t),
|
|
Args: argsSize(t),
|
|
Line: t.Pos().Line,
|
|
Labels: labels,
|
|
Lines: lines,
|
|
Spadj: spadj,
|
|
Relocs: relocs,
|
|
}
|
|
for _, f := range t.Flags {
|
|
switch f {
|
|
case "NOSPLIT":
|
|
fl.NoSplit = true
|
|
case "SPWRITE":
|
|
fl.SPWrite = true
|
|
}
|
|
}
|
|
img.Funcs = append(img.Funcs, fl)
|
|
img.Code = append(img.Code, code...)
|
|
}
|
|
|
|
// Lay out the data section behind the code, 16-aligned.
|
|
dataStart := len(img.Code)
|
|
for _, d := range dataSyms {
|
|
pos := dataStart + len(img.Data)
|
|
for pos%16 != 0 {
|
|
img.Data = append(img.Data, 0)
|
|
pos++
|
|
}
|
|
img.Symbols[d.name] = pos
|
|
img.Data = append(img.Data, d.buf...)
|
|
img.DataSyms = append(img.DataSyms, DataSymbol{
|
|
Name: d.name,
|
|
Pkg: d.pkg,
|
|
Offset: len(img.Data) - len(d.buf),
|
|
Size: d.size,
|
|
Static: d.static,
|
|
Rodata: d.rodata,
|
|
Noptr: d.noptr,
|
|
Dupok: d.dupok,
|
|
})
|
|
}
|
|
// The read-only literals the VMOVS/VMOVD/VMOVQ constant loads refer to
|
|
// follow the declared data, deduplicated across the file.
|
|
for _, lit := range pendingLits {
|
|
pos := dataStart + len(img.Data)
|
|
for pos%16 != 0 {
|
|
img.Data = append(img.Data, 0)
|
|
pos++
|
|
}
|
|
img.Symbols[lit.Name] = pos
|
|
img.Data = append(img.Data, lit.Data...)
|
|
img.DataSyms = append(img.DataSyms, DataSymbol{
|
|
Name: lit.Name,
|
|
Offset: len(img.Data) - len(lit.Data),
|
|
Size: len(lit.Data),
|
|
Rodata: true,
|
|
Dupok: true,
|
|
})
|
|
}
|
|
|
|
markExternals(img, dataSyms)
|
|
return img, nil
|
|
}
|
|
|
|
// arm64ResolveAliases applies the file's own simple #define aliases to every
|
|
// instruction operand, the way the toolchain's preprocessor substitutes them
|
|
// textually. Only single-line, non-parameterised bodies whose value is a
|
|
// register name or an integer constant are resolved: anything else
|
|
// (parameterised macros, multi-instruction bodies, header-supplied names)
|
|
// stays as written and surfaces as a normal operand error.
|
|
func arm64ResolveAliases(f *ast.File) {
|
|
type alias struct {
|
|
raw string // replacement text
|
|
reg bool // the body is a register name
|
|
value int64 // the body as an integer (when !reg)
|
|
isInt bool // the body parsed as an integer
|
|
mem bool // the body is a frame-relative memory reference
|
|
sym *ast.Symbol // the parsed frame-relative reference (when mem)
|
|
}
|
|
aliases := map[string]alias{}
|
|
raws := map[string]string{}
|
|
// The directive lines the parser records include the dead branches of
|
|
// every conditional, and a define inside a disabled region must not
|
|
// become an alias (go_tls.h's `#ifdef GOARCH_arm \n #define LR R14`
|
|
// must stay dead on arm64, where LR is R30). The liveness walk mirrors
|
|
// the preprocessor's: a conditional is live when its name was defined
|
|
// by a live define earlier in the file; every other name is undefined,
|
|
// the predefines (GOARCH_arm64 and friends) included, which the
|
|
// assembler never sees.
|
|
seen := map[string]bool{} // every live-defined name, any macro shape
|
|
cond := []bool{} // one entry per open #ifdef/#ifndef: its truth
|
|
live := func() bool {
|
|
for _, c := range cond {
|
|
if !c {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
for _, d := range f.Decls {
|
|
pre, ok := d.(*ast.Preproc)
|
|
if !ok {
|
|
continue
|
|
}
|
|
fields := strings.Fields(pre.Raw)
|
|
if len(fields) == 0 {
|
|
continue
|
|
}
|
|
switch fields[0] {
|
|
case "ifdef", "ifndef":
|
|
truth := false
|
|
if len(fields) >= 2 {
|
|
truth = seen[fields[1]] != (fields[0] == "ifndef")
|
|
}
|
|
cond = append(cond, truth)
|
|
case "else":
|
|
if len(cond) > 0 {
|
|
cond[len(cond)-1] = !cond[len(cond)-1]
|
|
}
|
|
case "endif":
|
|
if len(cond) > 0 {
|
|
cond = cond[:len(cond)-1]
|
|
}
|
|
case "define":
|
|
if !live() || len(fields) < 3 {
|
|
continue
|
|
}
|
|
name, body := fields[1], strings.Join(fields[2:], " ")
|
|
seen[name] = true
|
|
// A parameterised macro spells its parameter list right after
|
|
// the name; a multi-instruction body needs statement expansion.
|
|
if strings.ContainsAny(name, "(") || body == "" ||
|
|
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
|
|
continue
|
|
}
|
|
raws[name] = body
|
|
}
|
|
}
|
|
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
|
|
// transitively until nothing changes, bounded against cycles.
|
|
for range 8 {
|
|
changed := false
|
|
for name, body := range raws {
|
|
if next, ok := raws[body]; ok && next != body {
|
|
raws[name] = next
|
|
changed = true
|
|
}
|
|
}
|
|
if !changed {
|
|
break
|
|
}
|
|
}
|
|
for name, body := range raws {
|
|
isReg := func(s string) bool {
|
|
if arm64RegNum(s) >= 0 {
|
|
return true
|
|
}
|
|
if v, ok := a64VecReg(s); ok && !v.hasIdx {
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
if isReg(body) {
|
|
aliases[name] = alias{raw: body, reg: true}
|
|
continue
|
|
}
|
|
if sym, ok := arm64FrameAliasBody(body); ok {
|
|
aliases[name] = alias{raw: body, mem: true, sym: sym}
|
|
continue
|
|
}
|
|
v, err := strconv.ParseInt(body, 0, 64)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
aliases[name] = alias{raw: body, value: v, isInt: true}
|
|
}
|
|
if len(aliases) == 0 {
|
|
return
|
|
}
|
|
|
|
// replaceToken rewrites an operand whose whole text is one alias use
|
|
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
|
|
// ending at a non-identifier character.
|
|
replaceToken := func(s string) (string, bool) {
|
|
for name, a := range aliases {
|
|
if s == name {
|
|
return a.raw, true
|
|
}
|
|
if strings.HasPrefix(s, name) {
|
|
rest := s[len(name):]
|
|
if rest != "" && !isAliasWordByte(rest[0]) {
|
|
return a.raw + rest, true
|
|
}
|
|
}
|
|
}
|
|
return s, false
|
|
}
|
|
|
|
// replaceScan rewrites alias uses inside a composite operand (a
|
|
// parenthesised memory operand or a bracketed register list): every
|
|
// identifier run of word and dot characters is matched against the alias
|
|
// names, everything else copies verbatim. The whitespace-split replace
|
|
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
|
|
// punctuation attached.
|
|
replaceScan := func(s string) string {
|
|
var b strings.Builder
|
|
for i := 0; i < len(s); {
|
|
if isAliasWordByte(s[i]) || s[i] == '.' {
|
|
j := i
|
|
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
|
|
j++
|
|
}
|
|
if nn, ok := replaceToken(s[i:j]); ok {
|
|
b.WriteString(nn)
|
|
} else {
|
|
b.WriteString(s[i:j])
|
|
}
|
|
i = j
|
|
continue
|
|
}
|
|
b.WriteByte(s[i])
|
|
i++
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
for _, d := range f.Decls {
|
|
t, ok := d.(*ast.Text)
|
|
if !ok {
|
|
continue
|
|
}
|
|
for _, stmt := range t.Body {
|
|
in, ok := stmt.(*ast.Instr)
|
|
if !ok {
|
|
continue
|
|
}
|
|
for _, op := range in.Operands {
|
|
// Immediate aliases: $CLOCK_REALTIME → $0. The parser may
|
|
// leave the unevaluable name in Raw alone or carry it as an
|
|
// unevaluated symbol immediate; both shapes resolve here.
|
|
if op.Kind == ast.OpImmediate && !op.Imm.HasVal {
|
|
name := ""
|
|
if op.Imm.Sym != nil && op.Imm.Sym.Pseudo == "" {
|
|
name = op.Imm.Sym.Name
|
|
} else if op.Addr.Sym == nil && op.Addr.Base == "" {
|
|
name = strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(op.Raw), "$"))
|
|
}
|
|
if a, ok := aliases[name]; ok && a.isInt {
|
|
op.Imm.Val, op.Imm.HasVal = a.value, true
|
|
op.Imm.Sym = nil
|
|
op.Addr = ast.Address{}
|
|
op.Raw = "$" + a.raw
|
|
continue
|
|
}
|
|
}
|
|
// Memory operand whose displacement is an alias:
|
|
// TEB_error(R18_PLATFORM) with TEB_error → 0x68.
|
|
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base != "" {
|
|
if a, ok := aliases[op.Addr.Sym.Name]; ok && a.isInt {
|
|
op.Addr.Offset, op.Addr.HasOff = a.value, true
|
|
op.Addr.Sym = nil
|
|
op.Raw = a.raw + "(" + op.Addr.Base + ")"
|
|
continue
|
|
}
|
|
}
|
|
// Register alias as a memory base.
|
|
if op.Addr.Base != "" {
|
|
if a, ok := aliases[op.Addr.Base]; ok && a.reg {
|
|
op.Addr.Base = a.raw
|
|
}
|
|
}
|
|
// Bare and suffixed symbol tokens: registers, vector
|
|
// registers with arrangement or lane, branch labels.
|
|
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "" && op.Addr.Base == "" {
|
|
if nn, changed := replaceToken(op.Addr.Sym.Name); changed {
|
|
// A frame-relative body (ret+24(FP)) rebuilds the
|
|
// operand as a full memory reference.
|
|
if a, ok := aliases[op.Addr.Sym.Name]; ok && a.mem {
|
|
op.Addr.Sym, op.Addr.Base, op.Addr.Shift = a.sym, "", ""
|
|
op.Raw = a.raw
|
|
continue
|
|
}
|
|
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
|
|
// The span shape depends on what trailed the
|
|
// name: an element or arrangement selector
|
|
// (POLY.D[0], POLY.B16) rides in Shift and folds
|
|
// back onto the rewritten token; a shift
|
|
// operator stays in Shift while the span carries
|
|
// the bare register; a split list keeps its
|
|
// closing bracket, so the rewrite goes through
|
|
// the scan.
|
|
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
|
|
switch {
|
|
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
|
|
// Element or arrangement selectors and the
|
|
// closing bracket of a split list belong to
|
|
// the token text.
|
|
op.Raw = nn + sfx
|
|
op.Addr.Shift = ""
|
|
case op.Addr.Shift != "":
|
|
op.Raw = nn
|
|
default:
|
|
op.Raw = replaceScan(op.Raw)
|
|
}
|
|
continue
|
|
}
|
|
}
|
|
// Bracketed groups and lists travel in Raw: (RARG0, R1) and
|
|
// [V0.B16, V1.B16] with aliased members.
|
|
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
|
|
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
|
|
op.Raw = replaceScan(op.Raw)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// isAliasWordByte reports whether b can appear inside an identifier, so a
|
|
// substitution ending here would have merged two tokens.
|
|
func isAliasWordByte(b byte) bool {
|
|
return b == '_' || b >= '0' && b <= '9' || b >= 'a' && b <= 'z' || b >= 'A' && b <= 'Z'
|
|
}
|
|
|
|
// arm64FrameAliasBody parses an alias body of the shape NAME, NAME+off or
|
|
// NAME+off(PSEUDO) with PSEUDO one of FP/SP: the frame-relative memory
|
|
// references the runtime headers alias wholesale (LOCAL_RETVALID
|
|
// → ret+24(FP)). ok is false for anything else.
|
|
func arm64FrameAliasBody(body string) (*ast.Symbol, bool) {
|
|
s := strings.Join(strings.Fields(body), "")
|
|
i := strings.LastIndexByte(s, '(')
|
|
if i < 0 || !strings.HasSuffix(s, ")") {
|
|
return nil, false
|
|
}
|
|
pseudo := s[i+1 : len(s)-1]
|
|
if pseudo != "FP" && pseudo != "SP" {
|
|
return nil, false
|
|
}
|
|
head := s[:i]
|
|
name, offStr := head, ""
|
|
if j := strings.LastIndexByte(head, '+'); j >= 0 {
|
|
name, offStr = head[:j], head[j+1:]
|
|
}
|
|
if name == "" {
|
|
return nil, false
|
|
}
|
|
off, hasOff := int64(0), false
|
|
if offStr != "" {
|
|
v, err := strconv.ParseInt(offStr, 0, 64)
|
|
if err != nil {
|
|
return nil, false
|
|
}
|
|
off, hasOff = v, true
|
|
}
|
|
return &ast.Symbol{Name: name, Pseudo: pseudo, Offset: off, HasOff: hasOff, Raw: s}, true
|
|
}
|