feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+569
-108
@@ -5,6 +5,7 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/bits"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
||||
"FMOVS", "FMOVD":
|
||||
return arm64MovSize(mnem, ops, fi)
|
||||
case "ADD", "ADDW", "SUB", "SUBW":
|
||||
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
|
||||
"ADDS", "ADDSW", "SUBS", "SUBSW":
|
||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||
v := arm64Imm64(ops[0])
|
||||
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
|
||||
if v >= 0 && v <= 0xFFF {
|
||||
return 4
|
||||
// Size exactly as the encoder will emit: a single imm12 word, the
|
||||
// two-word ADDCON2 split, or a materialisation into REGTMP plus
|
||||
// the register form. Anything else would desynchronise the label
|
||||
// offsets of pass 1 from the bytes pass 2 lays down.
|
||||
if v, ok := arm64ImmOperandValue(ops[0]); ok {
|
||||
rn, rd := 0, 0
|
||||
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
|
||||
rd = n
|
||||
}
|
||||
if len(ops) == 3 {
|
||||
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
|
||||
rn = n
|
||||
}
|
||||
}
|
||||
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
|
||||
return 4 * len(ws)
|
||||
}
|
||||
}
|
||||
if v >= -2048 && v < 0 {
|
||||
return 4
|
||||
}
|
||||
// Larger immediates need MOV materialisation + op.
|
||||
return 8
|
||||
return 4
|
||||
}
|
||||
}
|
||||
return 4
|
||||
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64Bitfield(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
|
||||
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// EXTR.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
||||
return encodeARM64Extr(mnem, enc.op, ops)
|
||||
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
|
||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
||||
// this check precedes the plain SIMD3 path below.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok {
|
||||
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
|
||||
// only in the zero-immediate form (a64SimdVZero), so they route here with
|
||||
// an empty register-form spec.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
|
||||
return encodeARM64SimdV(mnem, spec, ops)
|
||||
}
|
||||
|
||||
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
}
|
||||
|
||||
// SIMD table lookup.
|
||||
if mnem == "VTBL" {
|
||||
return encodeARM64VTBL(ops)
|
||||
if mnem == "VTBL" || mnem == "VTBX" {
|
||||
return encodeARM64VTBL(mnem, ops)
|
||||
}
|
||||
|
||||
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
|
||||
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
}
|
||||
op := ops[0]
|
||||
|
||||
// Branch to the program counter itself: JMP (PC) spins forever. The
|
||||
// toolchain encodes it as an unconditional branch with a zero offset.
|
||||
// Branch to the program counter: JMP (PC) spins forever, and a spelled
|
||||
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
|
||||
// units. The toolchain encodes both as a plain branch of that offset.
|
||||
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
||||
if link {
|
||||
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
|
||||
rel := op.Addr.Offset
|
||||
if rel < -(1<<25) || rel >= (1<<25) {
|
||||
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
|
||||
}
|
||||
return a64wordLE(a64Branch(0, 0)), nil
|
||||
bop := uint32(0) // B
|
||||
if link {
|
||||
bop = 1 // BL
|
||||
}
|
||||
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
||||
}
|
||||
|
||||
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
|
||||
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
|
||||
// the ADD/SUB-with-flags family an add/sub immediate.
|
||||
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
|
||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
|
||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
|
||||
mnem == "MVN" || mnem == "MVNW" ||
|
||||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
|
||||
|
||||
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
|
||||
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
|
||||
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
var logical bool
|
||||
switch mnem {
|
||||
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
||||
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
|
||||
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
|
||||
"TST", "TSTW":
|
||||
logical = true
|
||||
}
|
||||
if logical {
|
||||
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
}
|
||||
inverted := false
|
||||
switch mnem {
|
||||
case "BIC", "BICW", "BICS", "BICSW":
|
||||
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
|
||||
inverted = true
|
||||
}
|
||||
if inverted {
|
||||
@@ -700,7 +727,37 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
}
|
||||
n, immr, imms, ok := a64LogicalImm(v, width)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
||||
// Beyond the bitmask immediates the toolchain materialises
|
||||
// the constant into REGTMP (R27) and uses the register form
|
||||
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
|
||||
// value, so the materialisation uses v before any inversion.
|
||||
written := v
|
||||
if inverted {
|
||||
written = ^v
|
||||
}
|
||||
width := mnem
|
||||
if strings.HasSuffix(mnem, "W") {
|
||||
width = "MOVW"
|
||||
} else {
|
||||
width = "MOVD"
|
||||
}
|
||||
mw, merr := encodeARM64LoadImm(27, written, width)
|
||||
var rn, rd int
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
rn = arm64RegNum(operandRegName(ops[1]))
|
||||
rd = arm64RegNum(operandRegName(ops[2]))
|
||||
default:
|
||||
rd = arm64RegNum(operandRegName(ops[1]))
|
||||
rn = rd
|
||||
}
|
||||
if isCmp {
|
||||
rd = 31
|
||||
}
|
||||
if merr != nil || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
||||
}
|
||||
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
|
||||
}
|
||||
opc := (baseOp >> 29) & 7
|
||||
sf := (baseOp >> 31) & 1
|
||||
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
|
||||
}
|
||||
if !isExtend {
|
||||
if amount < 0 || amount > 63 {
|
||||
// ROR rides the shifted-register field only for the logical
|
||||
// group; the toolchain reports "unsupported shift operator" for
|
||||
// the arithmetic forms, whose shift=11 encoding is unallocated.
|
||||
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
|
||||
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
|
||||
}
|
||||
// The imm6 field is 5 bits and truncates at the 32-bit width.
|
||||
limit := 63
|
||||
if strings.HasSuffix(mnem, "W") {
|
||||
limit = 31
|
||||
}
|
||||
if amount < 0 || amount > limit {
|
||||
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
|
||||
}
|
||||
// SP-based ADD/SUB have no shifted-register encoding: the
|
||||
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
// The carry family carries an immediate spelling in three operands
|
||||
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
|
||||
// operand, the same shape the two-operand form takes.
|
||||
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
|
||||
if v := arm64Imm64(ops[0]); v != 0 {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
// OP Rm, Rn, Rd
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
|
||||
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
|
||||
// shifted-register group, the only forms whose register operand accepts the
|
||||
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
|
||||
func arm64LogicalShifted(mnem string) bool {
|
||||
switch mnem {
|
||||
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
|
||||
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
|
||||
"TST", "TSTW", "MVN", "MVNW":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
|
||||
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
|
||||
// data-processing instructions the toolchain accepts an immediate $0
|
||||
// operand spelling for.
|
||||
func arm64CarryOp(mnem string) bool {
|
||||
switch mnem {
|
||||
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// arm64RegMod reports whether a register operand carries the shifted-register
|
||||
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
|
||||
// option (R0.UXTW, R3.SXTW<<2).
|
||||
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
|
||||
|
||||
// arm64RegModifier resolves a modified register operand: the register number,
|
||||
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
|
||||
// option (UXTB=0..SXTX=7) with its shift amount.
|
||||
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
|
||||
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
|
||||
// before the operator match.
|
||||
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
|
||||
name := operandRegName(op)
|
||||
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||
if before, after, ok0 := strings.Cut(name, "."); ok0 {
|
||||
switch strings.ToUpper(strings.TrimSpace(after)) {
|
||||
case "UXTB":
|
||||
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
||||
if rm < 0 {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
||||
amount, ok = arm64ShiftAmount(shift)
|
||||
if !ok || amount < 0 || amount > 4 {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
return rm, 0, extendOpt, true, amount, true
|
||||
}
|
||||
shiftKind = 0 // LSL
|
||||
shift := strings.TrimSpace(op.Addr.Shift)
|
||||
switch {
|
||||
case strings.HasPrefix(shift, "<<"):
|
||||
shiftKind = 0
|
||||
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
||||
default:
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
||||
amount, ok = arm64ShiftAmount(shift)
|
||||
if !ok {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
|
||||
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
|
||||
// MUL mnemonic, not to these.
|
||||
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
// The widening three-operand forms (SMULL, UMNEGL, …) read the
|
||||
// accumulate register as ZR, already preset in the table's base word.
|
||||
if len(ops) == 3 {
|
||||
switch mnem {
|
||||
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
|
||||
default:
|
||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
|
||||
}
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rm < 0 || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
if len(ops) != 4 {
|
||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
|
||||
}
|
||||
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
|
||||
// ---- ADD/SUB immediate ----
|
||||
|
||||
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
|
||||
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
|
||||
// following the toolchain's immediate classification (asm7.go conclass and
|
||||
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
|
||||
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
|
||||
// otherwise a constant materialisation into REGTMP (R27) followed by the
|
||||
// register form.
|
||||
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 2 && len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
v, ok := arm64ImmOperandValue(ops[0])
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
|
||||
}
|
||||
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
||||
rn := rd
|
||||
if len(ops) == 3 {
|
||||
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
|
||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
sf := uint32(1) // 64-bit
|
||||
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
|
||||
sf = 0 // 32-bit
|
||||
}
|
||||
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
|
||||
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
|
||||
// CMP/CMN discard the destination.
|
||||
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
|
||||
rd = 31 // ZR
|
||||
}
|
||||
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", mnem, err)
|
||||
}
|
||||
return a64WordsLE(ws...), nil
|
||||
}
|
||||
|
||||
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
|
||||
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
|
||||
// their W forms. rn and rd are resolved register numbers (a comparison
|
||||
// discards rd, so the caller passes 31).
|
||||
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
|
||||
w := strings.HasSuffix(mnem, "W")
|
||||
sf := uint32(1) // 64-bit
|
||||
d := v
|
||||
if w {
|
||||
sf = 0 // 32-bit
|
||||
// The W forms classify the 32-bit value (asm7.go con32class).
|
||||
d = int64(uint32(v))
|
||||
}
|
||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
op := uint32(0) // ADD
|
||||
S := uint32(0)
|
||||
if isSub {
|
||||
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if isS {
|
||||
S = 1
|
||||
}
|
||||
single := func(sh, imm12 uint32) []uint32 {
|
||||
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
|
||||
}
|
||||
|
||||
if v >= 0 && v <= 0xFFF {
|
||||
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
|
||||
// imm12: plain, then the one-shifted-by-12 form.
|
||||
if d >= 0 && d <= 0xFFF {
|
||||
return single(0, uint32(d)), nil
|
||||
}
|
||||
if v >= -2048 && v < 0 {
|
||||
// Encode as the opposite operation with positive immediate.
|
||||
opp := op ^ 1
|
||||
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
|
||||
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
|
||||
return single(1, uint32(d>>12)), nil
|
||||
}
|
||||
// Try with shift by 12.
|
||||
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
|
||||
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
|
||||
|
||||
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
|
||||
// split into two imm12 instructions, low half first (asm7.go case 48).
|
||||
// The encoding is complete in itself: no REGTMP, no register form. The S
|
||||
// forms must not break addition/subtraction, so the toolchain
|
||||
// reclassifies them and falls through to the materialisation below.
|
||||
dm := ^d
|
||||
if w {
|
||||
dm = ^d & 0xFFFFFFFF
|
||||
}
|
||||
// The imm12 field cannot carry the value; rejecting (rather than
|
||||
// truncating) matches the toolchain, which reports the same shape.
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
|
||||
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
|
||||
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
|
||||
return []uint32{
|
||||
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
|
||||
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Constant into REGTMP (R27), then the register form. The first word
|
||||
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
|
||||
// the complement form, the bitmask ORR otherwise, and the full
|
||||
// omovlconst sequence when no single word carries the value.
|
||||
var seq []uint32
|
||||
switch s := arm64Movcon(d); {
|
||||
case s >= 0:
|
||||
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
|
||||
case arm64Movcon(dm) >= 0:
|
||||
s := arm64Movcon(dm)
|
||||
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
|
||||
case isBitcon:
|
||||
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
|
||||
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
|
||||
default:
|
||||
seq = arm64MovLConst(d, sf)
|
||||
}
|
||||
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
|
||||
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
|
||||
for i := range seq[:len(seq)-1] {
|
||||
seq[i] |= 27 // REGTMP
|
||||
}
|
||||
return seq, nil
|
||||
}
|
||||
|
||||
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
|
||||
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
|
||||
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
|
||||
// returned with the destination field clear so the caller can OR its own
|
||||
// register in. movcon and movcon-of-complement must fail for d before this
|
||||
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
|
||||
func arm64MovLConst(d int64, sf uint32) []uint32 {
|
||||
if sf == 0 {
|
||||
// omovlconst AMOVW: both 16-bit halves, low first.
|
||||
return []uint32{
|
||||
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
|
||||
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
|
||||
}
|
||||
}
|
||||
dn := ^d
|
||||
var immh [4]uint64
|
||||
zero, neg := 0, 0
|
||||
for i := range immh {
|
||||
immh[i] = uint64(d>>(i*16)) & 0xFFFF
|
||||
switch immh[i] {
|
||||
case 0:
|
||||
zero++
|
||||
case 0xFFFF:
|
||||
neg++
|
||||
}
|
||||
}
|
||||
mw := func(opc uint32, val int64, chunk int) uint32 {
|
||||
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
|
||||
}
|
||||
var os []uint32
|
||||
switch {
|
||||
case zero == 2:
|
||||
// one MOVZ and one MOVK
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(2, d, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
case neg == 2:
|
||||
// one MOVN and one MOVK
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(0, dn, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
default:
|
||||
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
|
||||
// a single MOVK (constants from strength-reduced division).
|
||||
if zero == 0 && neg == 0 {
|
||||
for i := range 4 {
|
||||
mask := uint64(0xFFFF) << (i * 16)
|
||||
for period := 2; period <= 32; period *= 2 {
|
||||
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
|
||||
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
|
||||
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
|
||||
os = append(os, mw(3, d, i))
|
||||
return os
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case zero >= 1:
|
||||
// one MOVZ and up to three MOVKs
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(2, d, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
case neg >= 1:
|
||||
// one MOVN and up to three MOVKs
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(0, dn, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
default:
|
||||
// one MOVZ and three MOVKs
|
||||
os = append(os, mw(2, d, 0))
|
||||
for i := 1; i < 4; i++ {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
}
|
||||
return os
|
||||
}
|
||||
|
||||
// ---- MOV pseudo-instruction ----
|
||||
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
|
||||
var ws []uint32
|
||||
first := true
|
||||
for i := range 4 {
|
||||
chunk := (d >> uint(i*16)) & 0xFFFF
|
||||
if chunk == 0 {
|
||||
continue
|
||||
}
|
||||
if first {
|
||||
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
|
||||
first = false
|
||||
} else {
|
||||
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
|
||||
}
|
||||
}
|
||||
if len(ws) == 0 {
|
||||
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
|
||||
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
||||
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
|
||||
// for the first special 16-bit chunk, then MOVK per remaining one, with
|
||||
// the bitmask-plus-fixup shortcut for strength-reduced constants).
|
||||
ws := arm64MovLConst(d, sf)
|
||||
for i := range ws {
|
||||
ws[i] |= uint32(rd)
|
||||
}
|
||||
return a64WordsLE(ws...), nil
|
||||
}
|
||||
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
|
||||
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
|
||||
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
|
||||
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
|
||||
// with imms = lsb+width-1.
|
||||
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
|
||||
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
|
||||
}
|
||||
lsb := arm64Imm64(ops[0])
|
||||
width := arm64Imm64(ops[2])
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[3]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
bits := int64(32) << (baseOp >> 31 & 1)
|
||||
if lsb < 0 || lsb >= bits {
|
||||
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
|
||||
}
|
||||
if width < 1 || width > bits || lsb+width > bits {
|
||||
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
|
||||
}
|
||||
var immr, imms int64
|
||||
switch mnem {
|
||||
case "BFXIL", "BFXILW":
|
||||
immr, imms = lsb, lsb+width-1
|
||||
default: // BFI, SBFIZ, UBFIZ
|
||||
immr, imms = (-lsb)%bits, width-1
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
|
||||
// third field carries Rm or a 5-bit immediate in the same bits, at the
|
||||
// toolchain's choice of register or immediate operand.
|
||||
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
||||
// PRFM (Rn), $imm|<op>
|
||||
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
// Operand-less returns and pointer-authentication hints.
|
||||
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
|
||||
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
|
||||
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
|
||||
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
|
||||
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("%s expects no operand", mnem)
|
||||
}
|
||||
return a64wordLE(w), nil
|
||||
}
|
||||
switch mnem {
|
||||
case "BRK", "SVC":
|
||||
base := uint32(0xd4200000)
|
||||
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||
}
|
||||
return a64wordLE(base | uint32(v)<<5), nil
|
||||
case "DMB", "DSB", "ISB":
|
||||
case "DMB", "DSB", "ISB", "CLREX":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if v < 0 || v > 15 {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
|
||||
}
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<8), nil
|
||||
case "HINT":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 127 {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
|
||||
}
|
||||
return a64wordLE(0xd503201f | uint32(v)<<5), nil
|
||||
case "BTI":
|
||||
op := operandRegName(ops[0])
|
||||
base, ok := map[string]uint32{"C": 0xd503245f}[op]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
|
||||
}
|
||||
return a64wordLE(base), nil
|
||||
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 0xFFFF {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||
}
|
||||
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
|
||||
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
|
||||
"DCPS3": 0xd4a00003}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<5), nil
|
||||
case "DC":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("DC expects <op>, Rn")
|
||||
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
}
|
||||
return a64wordLE(base | uint32(rd)&31), nil
|
||||
case "MSR":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
|
||||
}
|
||||
if !isImmOperand(ops[0]) {
|
||||
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
|
||||
base, ok := a64MSRRegOps[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("MSR: invalid source register")
|
||||
}
|
||||
return a64wordLE(base | uint32(rs)&31), nil
|
||||
}
|
||||
base, ok := a64MSROps[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
|
||||
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
|
||||
func specBit(i int) uint16 { return 1 << uint(i) }
|
||||
|
||||
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
|
||||
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
|
||||
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
|
||||
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
|
||||
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
|
||||
return true
|
||||
}
|
||||
if !strings.HasPrefix(mnem, "VFCM") {
|
||||
return false
|
||||
}
|
||||
s := strings.Join(strings.Fields(op.Raw), "")
|
||||
s = strings.TrimPrefix(s, "$")
|
||||
s = strings.Trim(s, "()")
|
||||
return s == "0" || s == "0.0"
|
||||
}
|
||||
|
||||
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
|
||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
|
||||
// zero immediate takes its compare-against-zero form instead, and the
|
||||
// polynomial multiplies read the arrangement from their source operands
|
||||
// alone, the result spelling (H8, Q1) riding no encoding bits.
|
||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
|
||||
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
|
||||
// their compare-against-zero form instead, and the polynomial multiplies read
|
||||
// the arrangement from their source operands alone, the result spelling
|
||||
// (H8, Q1) riding no encoding bits.
|
||||
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
||||
if mnem == "VPMULL" || mnem == "VPMULL2" {
|
||||
if len(ops) != 3 {
|
||||
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
rd, _ := arm64VecOf(ops[2])
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
|
||||
}
|
||||
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
|
||||
if arm64Imm64(ops[0]) != 0 {
|
||||
if len(ops) == 3 && isImmOperand(ops[0]) {
|
||||
base, ok := a64SimdVZero[mnem]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
if !arm64SimdZeroImm(mnem, ops[0]) {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
vn, ok1 := arm64VecOf(ops[1])
|
||||
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
|
||||
allowed := uint16(0x7f)
|
||||
if strings.HasPrefix(mnem, "VFCM") {
|
||||
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
if spec.fixed {
|
||||
arrBits = 0
|
||||
}
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
}
|
||||
|
||||
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||
}
|
||||
// VUADDLV spells its arrangement on the source alone; the rest take it
|
||||
// on both.
|
||||
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
|
||||
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
|
||||
// index register rides bits 19:16, the first table register bits 9:5, the
|
||||
// destination bits 4:0 and the table length (registers minus one) bits
|
||||
// 14:13. The table registers must be consecutive.
|
||||
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
||||
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) < 3 {
|
||||
return nil, fmt.Errorf("VTBL expects index, table list and destination")
|
||||
}
|
||||
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
||||
default:
|
||||
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
|
||||
}
|
||||
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
base := uint32(0x0e000000)
|
||||
if mnem == "VTBX" {
|
||||
base |= 1 << 12
|
||||
}
|
||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
|
||||
}
|
||||
var immval int64
|
||||
switch mnem {
|
||||
case "VSHL":
|
||||
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
|
||||
if sh < 0 || sh >= esize {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
||||
}
|
||||
immval = esize + sh
|
||||
default: // VUSHR, VSRI
|
||||
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
|
||||
if sh < 1 || sh > esize {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
|
||||
}
|
||||
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
sym *ast.Symbol // the parsed frame-relative reference (when mem)
|
||||
}
|
||||
aliases := map[string]alias{}
|
||||
raws := map[string]string{}
|
||||
for _, d := range f.Decls {
|
||||
pre, ok := d.(*ast.Preproc)
|
||||
if !ok {
|
||||
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
|
||||
continue
|
||||
}
|
||||
raws[name] = body
|
||||
}
|
||||
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
|
||||
// transitively until nothing changes, bounded against cycles.
|
||||
for range 8 {
|
||||
changed := false
|
||||
for name, body := range raws {
|
||||
if next, ok := raws[body]; ok && next != body {
|
||||
raws[name] = next
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
break
|
||||
}
|
||||
}
|
||||
for name, body := range raws {
|
||||
isReg := func(s string) bool {
|
||||
if arm64RegNum(s) >= 0 {
|
||||
return true
|
||||
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
return
|
||||
}
|
||||
|
||||
// replace rewrites whole-word occurrences of the alias names in s.
|
||||
replace := func(s string) string {
|
||||
if s == "" {
|
||||
return s
|
||||
}
|
||||
out := strings.Fields(s)
|
||||
for i, w := range out {
|
||||
if a, ok := aliases[w]; ok {
|
||||
out[i] = a.raw
|
||||
}
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return s
|
||||
}
|
||||
return strings.Join(out, " ")
|
||||
}
|
||||
// replaceToken rewrites an operand whose whole text is one alias use
|
||||
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
|
||||
// ending at a non-identifier character.
|
||||
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
return s, false
|
||||
}
|
||||
|
||||
// replaceScan rewrites alias uses inside a composite operand (a
|
||||
// parenthesised memory operand or a bracketed register list): every
|
||||
// identifier run of word and dot characters is matched against the alias
|
||||
// names, everything else copies verbatim. The whitespace-split replace
|
||||
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
|
||||
// punctuation attached.
|
||||
replaceScan := func(s string) string {
|
||||
var b strings.Builder
|
||||
for i := 0; i < len(s); {
|
||||
if isAliasWordByte(s[i]) || s[i] == '.' {
|
||||
j := i
|
||||
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
|
||||
j++
|
||||
}
|
||||
if nn, ok := replaceToken(s[i:j]); ok {
|
||||
b.WriteString(nn)
|
||||
} else {
|
||||
b.WriteString(s[i:j])
|
||||
}
|
||||
i = j
|
||||
continue
|
||||
}
|
||||
b.WriteByte(s[i])
|
||||
i++
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
op.Raw = a.raw
|
||||
continue
|
||||
}
|
||||
op.Addr.Sym.Name = nn
|
||||
op.Addr.Sym.Raw = nn
|
||||
op.Raw = nn
|
||||
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
|
||||
// The span shape depends on what trailed the
|
||||
// name: an element or arrangement selector
|
||||
// (POLY.D[0], POLY.B16) rides in Shift and folds
|
||||
// back onto the rewritten token; a shift
|
||||
// operator stays in Shift while the span carries
|
||||
// the bare register; a split list keeps its
|
||||
// closing bracket, so the rewrite goes through
|
||||
// the scan.
|
||||
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||
switch {
|
||||
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
|
||||
// Element or arrangement selectors and the
|
||||
// closing bracket of a split list belong to
|
||||
// the token text.
|
||||
op.Raw = nn + sfx
|
||||
op.Addr.Shift = ""
|
||||
case op.Addr.Shift != "":
|
||||
op.Raw = nn
|
||||
default:
|
||||
op.Raw = replaceScan(op.Raw)
|
||||
}
|
||||
continue
|
||||
}
|
||||
}
|
||||
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
// [V0.B16, V1.B16] with aliased members.
|
||||
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
|
||||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
|
||||
op.Raw = replace(op.Raw)
|
||||
op.Raw = replaceScan(op.Raw)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user