feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+569
-108
@@ -5,6 +5,7 @@ package asm
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"math/bits"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
|||||||
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
||||||
"FMOVS", "FMOVD":
|
"FMOVS", "FMOVD":
|
||||||
return arm64MovSize(mnem, ops, fi)
|
return arm64MovSize(mnem, ops, fi)
|
||||||
case "ADD", "ADDW", "SUB", "SUBW":
|
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
|
||||||
|
"ADDS", "ADDSW", "SUBS", "SUBSW":
|
||||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||||
v := arm64Imm64(ops[0])
|
// Size exactly as the encoder will emit: a single imm12 word, the
|
||||||
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
|
// two-word ADDCON2 split, or a materialisation into REGTMP plus
|
||||||
if v >= 0 && v <= 0xFFF {
|
// the register form. Anything else would desynchronise the label
|
||||||
return 4
|
// offsets of pass 1 from the bytes pass 2 lays down.
|
||||||
|
if v, ok := arm64ImmOperandValue(ops[0]); ok {
|
||||||
|
rn, rd := 0, 0
|
||||||
|
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
|
||||||
|
rd = n
|
||||||
|
}
|
||||||
|
if len(ops) == 3 {
|
||||||
|
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
|
||||||
|
rn = n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
|
||||||
|
return 4 * len(ws)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if v >= -2048 && v < 0 {
|
return 4
|
||||||
return 4
|
|
||||||
}
|
|
||||||
// Larger immediates need MOV materialisation + op.
|
|
||||||
return 8
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return 4
|
return 4
|
||||||
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
|||||||
return encodeARM64Bitfield(mnem, enc.op, ops)
|
return encodeARM64Bitfield(mnem, enc.op, ops)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
|
||||||
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
|
||||||
|
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
|
||||||
|
}
|
||||||
|
|
||||||
// EXTR.
|
// EXTR.
|
||||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
||||||
return encodeARM64Extr(mnem, enc.op, ops)
|
return encodeARM64Extr(mnem, enc.op, ops)
|
||||||
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
|||||||
|
|
||||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||||
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
||||||
// this check precedes the plain SIMD3 path below.
|
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
|
||||||
if spec, ok := a64SimdVTable[mnem]; ok {
|
// only in the zero-immediate form (a64SimdVZero), so they route here with
|
||||||
|
// an empty register-form spec.
|
||||||
|
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
|
||||||
return encodeARM64SimdV(mnem, spec, ops)
|
return encodeARM64SimdV(mnem, spec, ops)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
|||||||
}
|
}
|
||||||
|
|
||||||
// SIMD table lookup.
|
// SIMD table lookup.
|
||||||
if mnem == "VTBL" {
|
if mnem == "VTBL" || mnem == "VTBX" {
|
||||||
return encodeARM64VTBL(ops)
|
return encodeARM64VTBL(mnem, ops)
|
||||||
}
|
}
|
||||||
|
|
||||||
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
|
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
|
||||||
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
|||||||
}
|
}
|
||||||
op := ops[0]
|
op := ops[0]
|
||||||
|
|
||||||
// Branch to the program counter itself: JMP (PC) spins forever. The
|
// Branch to the program counter: JMP (PC) spins forever, and a spelled
|
||||||
// toolchain encodes it as an unconditional branch with a zero offset.
|
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
|
||||||
|
// units. The toolchain encodes both as a plain branch of that offset.
|
||||||
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
||||||
if link {
|
rel := op.Addr.Offset
|
||||||
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
|
if rel < -(1<<25) || rel >= (1<<25) {
|
||||||
|
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
|
||||||
}
|
}
|
||||||
return a64wordLE(a64Branch(0, 0)), nil
|
bop := uint32(0) // B
|
||||||
|
if link {
|
||||||
|
bop = 1 // BL
|
||||||
|
}
|
||||||
|
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
|
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
|
||||||
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
|
|||||||
// the ADD/SUB-with-flags family an add/sub immediate.
|
// the ADD/SUB-with-flags family an add/sub immediate.
|
||||||
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||||
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
|
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
|
||||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
|
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
|
||||||
|
mnem == "MVN" || mnem == "MVNW" ||
|
||||||
|
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
|
||||||
|
|
||||||
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
|
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
|
||||||
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
|
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
|
||||||
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
var logical bool
|
var logical bool
|
||||||
switch mnem {
|
switch mnem {
|
||||||
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
||||||
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
|
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
|
||||||
|
"TST", "TSTW":
|
||||||
logical = true
|
logical = true
|
||||||
}
|
}
|
||||||
if logical {
|
if logical {
|
||||||
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
}
|
}
|
||||||
inverted := false
|
inverted := false
|
||||||
switch mnem {
|
switch mnem {
|
||||||
case "BIC", "BICW", "BICS", "BICSW":
|
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
|
||||||
inverted = true
|
inverted = true
|
||||||
}
|
}
|
||||||
if inverted {
|
if inverted {
|
||||||
@@ -700,7 +727,37 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
}
|
}
|
||||||
n, immr, imms, ok := a64LogicalImm(v, width)
|
n, immr, imms, ok := a64LogicalImm(v, width)
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
// Beyond the bitmask immediates the toolchain materialises
|
||||||
|
// the constant into REGTMP (R27) and uses the register form
|
||||||
|
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
|
||||||
|
// value, so the materialisation uses v before any inversion.
|
||||||
|
written := v
|
||||||
|
if inverted {
|
||||||
|
written = ^v
|
||||||
|
}
|
||||||
|
width := mnem
|
||||||
|
if strings.HasSuffix(mnem, "W") {
|
||||||
|
width = "MOVW"
|
||||||
|
} else {
|
||||||
|
width = "MOVD"
|
||||||
|
}
|
||||||
|
mw, merr := encodeARM64LoadImm(27, written, width)
|
||||||
|
var rn, rd int
|
||||||
|
switch len(ops) {
|
||||||
|
case 3:
|
||||||
|
rn = arm64RegNum(operandRegName(ops[1]))
|
||||||
|
rd = arm64RegNum(operandRegName(ops[2]))
|
||||||
|
default:
|
||||||
|
rd = arm64RegNum(operandRegName(ops[1]))
|
||||||
|
rn = rd
|
||||||
|
}
|
||||||
|
if isCmp {
|
||||||
|
rd = 31
|
||||||
|
}
|
||||||
|
if merr != nil || rn < 0 || rd < 0 {
|
||||||
|
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
||||||
|
}
|
||||||
|
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
|
||||||
}
|
}
|
||||||
opc := (baseOp >> 29) & 7
|
opc := (baseOp >> 29) & 7
|
||||||
sf := (baseOp >> 31) & 1
|
sf := (baseOp >> 31) & 1
|
||||||
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
|
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
|
||||||
}
|
}
|
||||||
if !isExtend {
|
if !isExtend {
|
||||||
if amount < 0 || amount > 63 {
|
// ROR rides the shifted-register field only for the logical
|
||||||
|
// group; the toolchain reports "unsupported shift operator" for
|
||||||
|
// the arithmetic forms, whose shift=11 encoding is unallocated.
|
||||||
|
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
|
||||||
|
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
|
||||||
|
}
|
||||||
|
// The imm6 field is 5 bits and truncates at the 32-bit width.
|
||||||
|
limit := 63
|
||||||
|
if strings.HasSuffix(mnem, "W") {
|
||||||
|
limit = 31
|
||||||
|
}
|
||||||
|
if amount < 0 || amount > limit {
|
||||||
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
|
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
|
||||||
}
|
}
|
||||||
// SP-based ADD/SUB have no shifted-register encoding: the
|
// SP-based ADD/SUB have no shifted-register encoding: the
|
||||||
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
|
|
||||||
switch len(ops) {
|
switch len(ops) {
|
||||||
case 3:
|
case 3:
|
||||||
|
// The carry family carries an immediate spelling in three operands
|
||||||
|
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
|
||||||
|
// operand, the same shape the two-operand form takes.
|
||||||
|
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
|
||||||
|
if v := arm64Imm64(ops[0]); v != 0 {
|
||||||
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||||
|
}
|
||||||
|
rn := arm64RegNum(operandRegName(ops[1]))
|
||||||
|
rd := arm64RegNum(operandRegName(ops[2]))
|
||||||
|
if rn < 0 || rd < 0 {
|
||||||
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||||
|
}
|
||||||
|
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||||
|
}
|
||||||
// OP Rm, Rn, Rd
|
// OP Rm, Rn, Rd
|
||||||
rm := arm64RegNum(operandRegName(ops[0]))
|
rm := arm64RegNum(operandRegName(ops[0]))
|
||||||
rn := arm64RegNum(operandRegName(ops[1]))
|
rn := arm64RegNum(operandRegName(ops[1]))
|
||||||
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
|
||||||
|
// shifted-register group, the only forms whose register operand accepts the
|
||||||
|
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
|
||||||
|
func arm64LogicalShifted(mnem string) bool {
|
||||||
|
switch mnem {
|
||||||
|
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
|
||||||
|
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
|
||||||
|
"TST", "TSTW", "MVN", "MVNW":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
|
||||||
|
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
|
||||||
|
// data-processing instructions the toolchain accepts an immediate $0
|
||||||
|
// operand spelling for.
|
||||||
|
func arm64CarryOp(mnem string) bool {
|
||||||
|
switch mnem {
|
||||||
|
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
// arm64RegMod reports whether a register operand carries the shifted-register
|
// arm64RegMod reports whether a register operand carries the shifted-register
|
||||||
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
|
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
|
||||||
// option (R0.UXTW, R3.SXTW<<2).
|
// option (R0.UXTW, R3.SXTW<<2).
|
||||||
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
|
|||||||
|
|
||||||
// arm64RegModifier resolves a modified register operand: the register number,
|
// arm64RegModifier resolves a modified register operand: the register number,
|
||||||
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
|
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
|
||||||
// option (UXTB=0..SXTX=7) with its shift amount.
|
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
|
||||||
|
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
|
||||||
|
// before the operator match.
|
||||||
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
|
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
|
||||||
name := operandRegName(op)
|
name := operandRegName(op)
|
||||||
|
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||||
if before, after, ok0 := strings.Cut(name, "."); ok0 {
|
if before, after, ok0 := strings.Cut(name, "."); ok0 {
|
||||||
switch strings.ToUpper(strings.TrimSpace(after)) {
|
switch strings.ToUpper(strings.TrimSpace(after)) {
|
||||||
case "UXTB":
|
case "UXTB":
|
||||||
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
|||||||
if rm < 0 {
|
if rm < 0 {
|
||||||
return 0, 0, 0, false, 0, false
|
return 0, 0, 0, false, 0, false
|
||||||
}
|
}
|
||||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
amount, ok = arm64ShiftAmount(shift)
|
||||||
if !ok || amount < 0 || amount > 4 {
|
if !ok || amount < 0 || amount > 4 {
|
||||||
return 0, 0, 0, false, 0, false
|
return 0, 0, 0, false, 0, false
|
||||||
}
|
}
|
||||||
return rm, 0, extendOpt, true, amount, true
|
return rm, 0, extendOpt, true, amount, true
|
||||||
}
|
}
|
||||||
shiftKind = 0 // LSL
|
shiftKind = 0 // LSL
|
||||||
shift := strings.TrimSpace(op.Addr.Shift)
|
|
||||||
switch {
|
switch {
|
||||||
case strings.HasPrefix(shift, "<<"):
|
case strings.HasPrefix(shift, "<<"):
|
||||||
shiftKind = 0
|
shiftKind = 0
|
||||||
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
|||||||
default:
|
default:
|
||||||
return 0, 0, 0, false, 0, false
|
return 0, 0, 0, false, 0, false
|
||||||
}
|
}
|
||||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
amount, ok = arm64ShiftAmount(shift)
|
||||||
if !ok {
|
if !ok {
|
||||||
return 0, 0, 0, false, 0, false
|
return 0, 0, 0, false, 0, false
|
||||||
}
|
}
|
||||||
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
|
|||||||
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
|
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
|
||||||
// MUL mnemonic, not to these.
|
// MUL mnemonic, not to these.
|
||||||
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||||
|
// The widening three-operand forms (SMULL, UMNEGL, …) read the
|
||||||
|
// accumulate register as ZR, already preset in the table's base word.
|
||||||
|
if len(ops) == 3 {
|
||||||
|
switch mnem {
|
||||||
|
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
|
||||||
|
default:
|
||||||
|
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
|
||||||
|
}
|
||||||
|
rm := arm64RegNum(operandRegName(ops[0]))
|
||||||
|
rn := arm64RegNum(operandRegName(ops[1]))
|
||||||
|
rd := arm64RegNum(operandRegName(ops[2]))
|
||||||
|
if rm < 0 || rn < 0 || rd < 0 {
|
||||||
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||||
|
}
|
||||||
|
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||||
|
}
|
||||||
if len(ops) != 4 {
|
if len(ops) != 4 {
|
||||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
|||||||
|
|
||||||
// ---- ADD/SUB immediate ----
|
// ---- ADD/SUB immediate ----
|
||||||
|
|
||||||
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
|
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
|
||||||
|
// following the toolchain's immediate classification (asm7.go conclass and
|
||||||
|
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
|
||||||
|
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
|
||||||
|
// otherwise a constant materialisation into REGTMP (R27) followed by the
|
||||||
|
// register form.
|
||||||
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||||
if len(ops) != 2 && len(ops) != 3 {
|
if len(ops) != 2 && len(ops) != 3 {
|
||||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||||
}
|
}
|
||||||
v := arm64Imm64(ops[0])
|
v, ok := arm64ImmOperandValue(ops[0])
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
|
||||||
|
}
|
||||||
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
||||||
rn := rd
|
rn := rd
|
||||||
if len(ops) == 3 {
|
if len(ops) == 3 {
|
||||||
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
if rn < 0 || rd < 0 {
|
if rn < 0 || rd < 0 {
|
||||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||||
}
|
}
|
||||||
|
// CMP/CMN discard the destination.
|
||||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
|
||||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
|
||||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
|
||||||
sf := uint32(1) // 64-bit
|
|
||||||
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
|
|
||||||
sf = 0 // 32-bit
|
|
||||||
}
|
|
||||||
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
|
|
||||||
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
|
|
||||||
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
|
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
|
||||||
rd = 31 // ZR
|
rd = 31 // ZR
|
||||||
}
|
}
|
||||||
|
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("%s: %w", mnem, err)
|
||||||
|
}
|
||||||
|
return a64WordsLE(ws...), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
|
||||||
|
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
|
||||||
|
// their W forms. rn and rd are resolved register numbers (a comparison
|
||||||
|
// discards rd, so the caller passes 31).
|
||||||
|
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
|
||||||
|
w := strings.HasSuffix(mnem, "W")
|
||||||
|
sf := uint32(1) // 64-bit
|
||||||
|
d := v
|
||||||
|
if w {
|
||||||
|
sf = 0 // 32-bit
|
||||||
|
// The W forms classify the 32-bit value (asm7.go con32class).
|
||||||
|
d = int64(uint32(v))
|
||||||
|
}
|
||||||
|
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||||
|
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||||
|
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||||
op := uint32(0) // ADD
|
op := uint32(0) // ADD
|
||||||
S := uint32(0)
|
S := uint32(0)
|
||||||
if isSub {
|
if isSub {
|
||||||
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
if isS {
|
if isS {
|
||||||
S = 1
|
S = 1
|
||||||
}
|
}
|
||||||
|
single := func(sh, imm12 uint32) []uint32 {
|
||||||
|
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
|
||||||
|
}
|
||||||
|
|
||||||
if v >= 0 && v <= 0xFFF {
|
// imm12: plain, then the one-shifted-by-12 form.
|
||||||
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
|
if d >= 0 && d <= 0xFFF {
|
||||||
|
return single(0, uint32(d)), nil
|
||||||
}
|
}
|
||||||
if v >= -2048 && v < 0 {
|
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
|
||||||
// Encode as the opposite operation with positive immediate.
|
return single(1, uint32(d>>12)), nil
|
||||||
opp := op ^ 1
|
|
||||||
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
|
|
||||||
}
|
}
|
||||||
// Try with shift by 12.
|
|
||||||
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
|
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
|
||||||
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
|
// split into two imm12 instructions, low half first (asm7.go case 48).
|
||||||
|
// The encoding is complete in itself: no REGTMP, no register form. The S
|
||||||
|
// forms must not break addition/subtraction, so the toolchain
|
||||||
|
// reclassifies them and falls through to the materialisation below.
|
||||||
|
dm := ^d
|
||||||
|
if w {
|
||||||
|
dm = ^d & 0xFFFFFFFF
|
||||||
}
|
}
|
||||||
// The imm12 field cannot carry the value; rejecting (rather than
|
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
|
||||||
// truncating) matches the toolchain, which reports the same shape.
|
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
|
||||||
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
|
return []uint32{
|
||||||
|
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
|
||||||
|
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Constant into REGTMP (R27), then the register form. The first word
|
||||||
|
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
|
||||||
|
// the complement form, the bitmask ORR otherwise, and the full
|
||||||
|
// omovlconst sequence when no single word carries the value.
|
||||||
|
var seq []uint32
|
||||||
|
switch s := arm64Movcon(d); {
|
||||||
|
case s >= 0:
|
||||||
|
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
|
||||||
|
case arm64Movcon(dm) >= 0:
|
||||||
|
s := arm64Movcon(dm)
|
||||||
|
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
|
||||||
|
case isBitcon:
|
||||||
|
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
|
||||||
|
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
|
||||||
|
default:
|
||||||
|
seq = arm64MovLConst(d, sf)
|
||||||
|
}
|
||||||
|
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
|
||||||
|
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
|
||||||
|
for i := range seq[:len(seq)-1] {
|
||||||
|
seq[i] |= 27 // REGTMP
|
||||||
|
}
|
||||||
|
return seq, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
|
||||||
|
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
|
||||||
|
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
|
||||||
|
// returned with the destination field clear so the caller can OR its own
|
||||||
|
// register in. movcon and movcon-of-complement must fail for d before this
|
||||||
|
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
|
||||||
|
func arm64MovLConst(d int64, sf uint32) []uint32 {
|
||||||
|
if sf == 0 {
|
||||||
|
// omovlconst AMOVW: both 16-bit halves, low first.
|
||||||
|
return []uint32{
|
||||||
|
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
|
||||||
|
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dn := ^d
|
||||||
|
var immh [4]uint64
|
||||||
|
zero, neg := 0, 0
|
||||||
|
for i := range immh {
|
||||||
|
immh[i] = uint64(d>>(i*16)) & 0xFFFF
|
||||||
|
switch immh[i] {
|
||||||
|
case 0:
|
||||||
|
zero++
|
||||||
|
case 0xFFFF:
|
||||||
|
neg++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
mw := func(opc uint32, val int64, chunk int) uint32 {
|
||||||
|
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
|
||||||
|
}
|
||||||
|
var os []uint32
|
||||||
|
switch {
|
||||||
|
case zero == 2:
|
||||||
|
// one MOVZ and one MOVK
|
||||||
|
i := 0
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0 {
|
||||||
|
os = append(os, mw(2, d, i))
|
||||||
|
i++
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0 {
|
||||||
|
os = append(os, mw(3, d, i))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
case neg == 2:
|
||||||
|
// one MOVN and one MOVK
|
||||||
|
i := 0
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0xFFFF {
|
||||||
|
os = append(os, mw(0, dn, i))
|
||||||
|
i++
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0xFFFF {
|
||||||
|
os = append(os, mw(3, d, i))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
|
||||||
|
// a single MOVK (constants from strength-reduced division).
|
||||||
|
if zero == 0 && neg == 0 {
|
||||||
|
for i := range 4 {
|
||||||
|
mask := uint64(0xFFFF) << (i * 16)
|
||||||
|
for period := 2; period <= 32; period *= 2 {
|
||||||
|
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
|
||||||
|
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
|
||||||
|
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
|
||||||
|
os = append(os, mw(3, d, i))
|
||||||
|
return os
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case zero >= 1:
|
||||||
|
// one MOVZ and up to three MOVKs
|
||||||
|
i := 0
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0 {
|
||||||
|
os = append(os, mw(2, d, i))
|
||||||
|
i++
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0 {
|
||||||
|
os = append(os, mw(3, d, i))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
case neg >= 1:
|
||||||
|
// one MOVN and up to three MOVKs
|
||||||
|
i := 0
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0xFFFF {
|
||||||
|
os = append(os, mw(0, dn, i))
|
||||||
|
i++
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for ; i < 4; i++ {
|
||||||
|
if immh[i] != 0xFFFF {
|
||||||
|
os = append(os, mw(3, d, i))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
// one MOVZ and three MOVKs
|
||||||
|
os = append(os, mw(2, d, 0))
|
||||||
|
for i := 1; i < 4; i++ {
|
||||||
|
os = append(os, mw(3, d, i))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return os
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- MOV pseudo-instruction ----
|
// ---- MOV pseudo-instruction ----
|
||||||
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
|
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
|
||||||
var ws []uint32
|
// for the first special 16-bit chunk, then MOVK per remaining one, with
|
||||||
first := true
|
// the bitmask-plus-fixup shortcut for strength-reduced constants).
|
||||||
for i := range 4 {
|
ws := arm64MovLConst(d, sf)
|
||||||
chunk := (d >> uint(i*16)) & 0xFFFF
|
for i := range ws {
|
||||||
if chunk == 0 {
|
ws[i] |= uint32(rd)
|
||||||
continue
|
|
||||||
}
|
|
||||||
if first {
|
|
||||||
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
|
|
||||||
first = false
|
|
||||||
} else {
|
|
||||||
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(ws) == 0 {
|
|
||||||
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
|
|
||||||
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
|
||||||
}
|
}
|
||||||
return a64WordsLE(ws...), nil
|
return a64WordsLE(ws...), nil
|
||||||
}
|
}
|
||||||
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
|
|||||||
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
|
||||||
|
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
|
||||||
|
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
|
||||||
|
// with imms = lsb+width-1.
|
||||||
|
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||||
|
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
|
||||||
|
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
|
||||||
|
}
|
||||||
|
lsb := arm64Imm64(ops[0])
|
||||||
|
width := arm64Imm64(ops[2])
|
||||||
|
rn := arm64RegNum(operandRegName(ops[1]))
|
||||||
|
rd := arm64RegNum(operandRegName(ops[3]))
|
||||||
|
if rn < 0 || rd < 0 {
|
||||||
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||||
|
}
|
||||||
|
bits := int64(32) << (baseOp >> 31 & 1)
|
||||||
|
if lsb < 0 || lsb >= bits {
|
||||||
|
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
|
||||||
|
}
|
||||||
|
if width < 1 || width > bits || lsb+width > bits {
|
||||||
|
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
|
||||||
|
}
|
||||||
|
var immr, imms int64
|
||||||
|
switch mnem {
|
||||||
|
case "BFXIL", "BFXILW":
|
||||||
|
immr, imms = lsb, lsb+width-1
|
||||||
|
default: // BFI, SBFIZ, UBFIZ
|
||||||
|
immr, imms = (-lsb)%bits, width-1
|
||||||
|
}
|
||||||
|
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||||
|
}
|
||||||
|
|
||||||
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
|
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
|
||||||
// third field carries Rm or a 5-bit immediate in the same bits, at the
|
// third field carries Rm or a 5-bit immediate in the same bits, at the
|
||||||
// toolchain's choice of register or immediate operand.
|
// toolchain's choice of register or immediate operand.
|
||||||
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
|||||||
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
||||||
// PRFM (Rn), $imm|<op>
|
// PRFM (Rn), $imm|<op>
|
||||||
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||||
|
// Operand-less returns and pointer-authentication hints.
|
||||||
|
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
|
||||||
|
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
|
||||||
|
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
|
||||||
|
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
|
||||||
|
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
|
||||||
|
if len(ops) != 0 {
|
||||||
|
return nil, fmt.Errorf("%s expects no operand", mnem)
|
||||||
|
}
|
||||||
|
return a64wordLE(w), nil
|
||||||
|
}
|
||||||
switch mnem {
|
switch mnem {
|
||||||
case "BRK", "SVC":
|
case "BRK", "SVC":
|
||||||
base := uint32(0xd4200000)
|
base := uint32(0xd4200000)
|
||||||
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||||
}
|
}
|
||||||
return a64wordLE(base | uint32(v)<<5), nil
|
return a64wordLE(base | uint32(v)<<5), nil
|
||||||
case "DMB", "DSB", "ISB":
|
case "DMB", "DSB", "ISB", "CLREX":
|
||||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||||
}
|
}
|
||||||
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
if v < 0 || v > 15 {
|
if v < 0 || v > 15 {
|
||||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
|
||||||
}
|
}
|
||||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
|
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
|
||||||
return a64wordLE(base | uint32(v)<<8), nil
|
return a64wordLE(base | uint32(v)<<8), nil
|
||||||
|
case "HINT":
|
||||||
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||||
|
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||||
|
}
|
||||||
|
v := arm64Imm64(ops[0])
|
||||||
|
if v < 0 || v > 127 {
|
||||||
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
|
||||||
|
}
|
||||||
|
return a64wordLE(0xd503201f | uint32(v)<<5), nil
|
||||||
|
case "BTI":
|
||||||
|
op := operandRegName(ops[0])
|
||||||
|
base, ok := map[string]uint32{"C": 0xd503245f}[op]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
|
||||||
|
}
|
||||||
|
return a64wordLE(base), nil
|
||||||
|
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
|
||||||
|
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||||
|
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||||
|
}
|
||||||
|
v := arm64Imm64(ops[0])
|
||||||
|
if v < 0 || v > 0xFFFF {
|
||||||
|
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||||
|
}
|
||||||
|
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
|
||||||
|
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
|
||||||
|
"DCPS3": 0xd4a00003}[mnem]
|
||||||
|
return a64wordLE(base | uint32(v)<<5), nil
|
||||||
case "DC":
|
case "DC":
|
||||||
if len(ops) != 2 {
|
if len(ops) != 2 {
|
||||||
return nil, fmt.Errorf("DC expects <op>, Rn")
|
return nil, fmt.Errorf("DC expects <op>, Rn")
|
||||||
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
|||||||
}
|
}
|
||||||
return a64wordLE(base | uint32(rd)&31), nil
|
return a64wordLE(base | uint32(rd)&31), nil
|
||||||
case "MSR":
|
case "MSR":
|
||||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
if len(ops) != 2 {
|
||||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
|
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
|
||||||
|
}
|
||||||
|
if !isImmOperand(ops[0]) {
|
||||||
|
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
|
||||||
|
base, ok := a64MSRRegOps[operandRegName(ops[1])]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
||||||
|
}
|
||||||
|
rs := arm64RegNum(operandRegName(ops[0]))
|
||||||
|
if rs < 0 {
|
||||||
|
return nil, fmt.Errorf("MSR: invalid source register")
|
||||||
|
}
|
||||||
|
return a64wordLE(base | uint32(rs)&31), nil
|
||||||
}
|
}
|
||||||
base, ok := a64MSROps[operandRegName(ops[1])]
|
base, ok := a64MSROps[operandRegName(ops[1])]
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
|
|||||||
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
|
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
|
||||||
func specBit(i int) uint16 { return 1 << uint(i) }
|
func specBit(i int) uint16 { return 1 << uint(i) }
|
||||||
|
|
||||||
|
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
|
||||||
|
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
|
||||||
|
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
|
||||||
|
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
|
||||||
|
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if !strings.HasPrefix(mnem, "VFCM") {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
s := strings.Join(strings.Fields(op.Raw), "")
|
||||||
|
s = strings.TrimPrefix(s, "$")
|
||||||
|
s = strings.Trim(s, "()")
|
||||||
|
return s == "0" || s == "0.0"
|
||||||
|
}
|
||||||
|
|
||||||
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
|
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
|
||||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
|
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
|
||||||
// zero immediate takes its compare-against-zero form instead, and the
|
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
|
||||||
// polynomial multiplies read the arrangement from their source operands
|
// their compare-against-zero form instead, and the polynomial multiplies read
|
||||||
// alone, the result spelling (H8, Q1) riding no encoding bits.
|
// the arrangement from their source operands alone, the result spelling
|
||||||
|
// (H8, Q1) riding no encoding bits.
|
||||||
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
||||||
if mnem == "VPMULL" || mnem == "VPMULL2" {
|
if mnem == "VPMULL" || mnem == "VPMULL2" {
|
||||||
if len(ops) != 3 {
|
if len(ops) != 3 {
|
||||||
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
|||||||
rd, _ := arm64VecOf(ops[2])
|
rd, _ := arm64VecOf(ops[2])
|
||||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
|
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
|
||||||
}
|
}
|
||||||
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
|
if len(ops) == 3 && isImmOperand(ops[0]) {
|
||||||
if arm64Imm64(ops[0]) != 0 {
|
base, ok := a64SimdVZero[mnem]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||||
|
}
|
||||||
|
if !arm64SimdZeroImm(mnem, ops[0]) {
|
||||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||||
}
|
}
|
||||||
vn, ok1 := arm64VecOf(ops[1])
|
vn, ok1 := arm64VecOf(ops[1])
|
||||||
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
|||||||
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
|
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
|
||||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||||
}
|
}
|
||||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
|
allowed := uint16(0x7f)
|
||||||
|
if strings.HasPrefix(mnem, "VFCM") {
|
||||||
|
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
|
||||||
|
}
|
||||||
|
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||||
}
|
}
|
||||||
if len(ops) != 3 {
|
if len(ops) != 3 {
|
||||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||||
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
|||||||
if spec.fixed {
|
if spec.fixed {
|
||||||
arrBits = 0
|
arrBits = 0
|
||||||
}
|
}
|
||||||
|
if a64SimdQOnly[mnem] {
|
||||||
|
arrBits &= 1 << 30
|
||||||
|
}
|
||||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
arrBits := a64ArrBits[arr]
|
||||||
|
if a64SimdQOnly[mnem] {
|
||||||
|
arrBits &= 1 << 30
|
||||||
|
}
|
||||||
|
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||||
}
|
}
|
||||||
// VUADDLV spells its arrangement on the source alone; the rest take it
|
// VUADDLV spells its arrangement on the source alone; the rest take it
|
||||||
// on both.
|
// on both.
|
||||||
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
arrBits := a64ArrBits[arr]
|
||||||
|
if a64SimdQOnly[mnem] {
|
||||||
|
arrBits &= 1 << 30
|
||||||
|
}
|
||||||
|
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
|
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
|
||||||
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
|
|||||||
// index register rides bits 19:16, the first table register bits 9:5, the
|
// index register rides bits 19:16, the first table register bits 9:5, the
|
||||||
// destination bits 4:0 and the table length (registers minus one) bits
|
// destination bits 4:0 and the table length (registers minus one) bits
|
||||||
// 14:13. The table registers must be consecutive.
|
// 14:13. The table registers must be consecutive.
|
||||||
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||||
if len(ops) < 3 {
|
if len(ops) < 3 {
|
||||||
return nil, fmt.Errorf("VTBL expects index, table list and destination")
|
return nil, fmt.Errorf("VTBL expects index, table list and destination")
|
||||||
}
|
}
|
||||||
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
|||||||
default:
|
default:
|
||||||
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
|
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
|
||||||
}
|
}
|
||||||
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
base := uint32(0x0e000000)
|
||||||
|
if mnem == "VTBX" {
|
||||||
|
base |= 1 << 12
|
||||||
|
}
|
||||||
|
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||||
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
|
|||||||
}
|
}
|
||||||
var immval int64
|
var immval int64
|
||||||
switch mnem {
|
switch mnem {
|
||||||
case "VSHL":
|
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
|
||||||
if sh < 0 || sh >= esize {
|
if sh < 0 || sh >= esize {
|
||||||
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
||||||
}
|
}
|
||||||
immval = esize + sh
|
immval = esize + sh
|
||||||
default: // VUSHR, VSRI
|
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
|
||||||
if sh < 1 || sh > esize {
|
if sh < 1 || sh > esize {
|
||||||
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
|
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
|
||||||
}
|
}
|
||||||
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
|
|||||||
sym *ast.Symbol // the parsed frame-relative reference (when mem)
|
sym *ast.Symbol // the parsed frame-relative reference (when mem)
|
||||||
}
|
}
|
||||||
aliases := map[string]alias{}
|
aliases := map[string]alias{}
|
||||||
|
raws := map[string]string{}
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
pre, ok := d.(*ast.Preproc)
|
pre, ok := d.(*ast.Preproc)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
|
|||||||
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
|
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
raws[name] = body
|
||||||
|
}
|
||||||
|
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
|
||||||
|
// transitively until nothing changes, bounded against cycles.
|
||||||
|
for range 8 {
|
||||||
|
changed := false
|
||||||
|
for name, body := range raws {
|
||||||
|
if next, ok := raws[body]; ok && next != body {
|
||||||
|
raws[name] = next
|
||||||
|
changed = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !changed {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for name, body := range raws {
|
||||||
isReg := func(s string) bool {
|
isReg := func(s string) bool {
|
||||||
if arm64RegNum(s) >= 0 {
|
if arm64RegNum(s) >= 0 {
|
||||||
return true
|
return true
|
||||||
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
// replace rewrites whole-word occurrences of the alias names in s.
|
|
||||||
replace := func(s string) string {
|
|
||||||
if s == "" {
|
|
||||||
return s
|
|
||||||
}
|
|
||||||
out := strings.Fields(s)
|
|
||||||
for i, w := range out {
|
|
||||||
if a, ok := aliases[w]; ok {
|
|
||||||
out[i] = a.raw
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if len(out) == 0 {
|
|
||||||
return s
|
|
||||||
}
|
|
||||||
return strings.Join(out, " ")
|
|
||||||
}
|
|
||||||
// replaceToken rewrites an operand whose whole text is one alias use
|
// replaceToken rewrites an operand whose whole text is one alias use
|
||||||
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
|
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
|
||||||
// ending at a non-identifier character.
|
// ending at a non-identifier character.
|
||||||
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
|
|||||||
return s, false
|
return s, false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// replaceScan rewrites alias uses inside a composite operand (a
|
||||||
|
// parenthesised memory operand or a bracketed register list): every
|
||||||
|
// identifier run of word and dot characters is matched against the alias
|
||||||
|
// names, everything else copies verbatim. The whitespace-split replace
|
||||||
|
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
|
||||||
|
// punctuation attached.
|
||||||
|
replaceScan := func(s string) string {
|
||||||
|
var b strings.Builder
|
||||||
|
for i := 0; i < len(s); {
|
||||||
|
if isAliasWordByte(s[i]) || s[i] == '.' {
|
||||||
|
j := i
|
||||||
|
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
|
||||||
|
j++
|
||||||
|
}
|
||||||
|
if nn, ok := replaceToken(s[i:j]); ok {
|
||||||
|
b.WriteString(nn)
|
||||||
|
} else {
|
||||||
|
b.WriteString(s[i:j])
|
||||||
|
}
|
||||||
|
i = j
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
b.WriteByte(s[i])
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
for _, d := range f.Decls {
|
for _, d := range f.Decls {
|
||||||
t, ok := d.(*ast.Text)
|
t, ok := d.(*ast.Text)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
|
|||||||
op.Raw = a.raw
|
op.Raw = a.raw
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
op.Addr.Sym.Name = nn
|
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
|
||||||
op.Addr.Sym.Raw = nn
|
// The span shape depends on what trailed the
|
||||||
op.Raw = nn
|
// name: an element or arrangement selector
|
||||||
|
// (POLY.D[0], POLY.B16) rides in Shift and folds
|
||||||
|
// back onto the rewritten token; a shift
|
||||||
|
// operator stays in Shift while the span carries
|
||||||
|
// the bare register; a split list keeps its
|
||||||
|
// closing bracket, so the rewrite goes through
|
||||||
|
// the scan.
|
||||||
|
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||||
|
switch {
|
||||||
|
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
|
||||||
|
// Element or arrangement selectors and the
|
||||||
|
// closing bracket of a split list belong to
|
||||||
|
// the token text.
|
||||||
|
op.Raw = nn + sfx
|
||||||
|
op.Addr.Shift = ""
|
||||||
|
case op.Addr.Shift != "":
|
||||||
|
op.Raw = nn
|
||||||
|
default:
|
||||||
|
op.Raw = replaceScan(op.Raw)
|
||||||
|
}
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
|
|||||||
// [V0.B16, V1.B16] with aliased members.
|
// [V0.B16, V1.B16] with aliased members.
|
||||||
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
|
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
|
||||||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
|
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
|
||||||
op.Raw = replace(op.Raw)
|
op.Raw = replaceScan(op.Raw)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+292
-55
@@ -308,6 +308,8 @@ const (
|
|||||||
a64CondLT = 0xb
|
a64CondLT = 0xb
|
||||||
a64CondGT = 0xc
|
a64CondGT = 0xc
|
||||||
a64CondLE = 0xd
|
a64CondLE = 0xd
|
||||||
|
a64CondAL = 0xe
|
||||||
|
a64CondNV = 0xf
|
||||||
)
|
)
|
||||||
|
|
||||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||||
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
|
|||||||
"LT": a64CondLT,
|
"LT": a64CondLT,
|
||||||
"GT": a64CondGT,
|
"GT": a64CondGT,
|
||||||
"LE": a64CondLE,
|
"LE": a64CondLE,
|
||||||
|
"AL": a64CondAL,
|
||||||
|
"NV": a64CondNV,
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- instruction format tags ----
|
// ---- instruction format tags ----
|
||||||
@@ -335,46 +339,47 @@ var arm64CondMap = map[string]uint32{
|
|||||||
type a64Format uint8
|
type a64Format uint8
|
||||||
|
|
||||||
const (
|
const (
|
||||||
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
|
||||||
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
a64FMovWide // move wide: MOVZ, MOVN, MOVK
|
||||||
a64FBranch // unconditional branch (B/BL)
|
a64FBranch // unconditional branch (B/BL)
|
||||||
a64FBranchCond // conditional branch (B.cond)
|
a64FBranchCond // conditional branch (B.cond)
|
||||||
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
|
||||||
a64FADR // ADR/ADRP
|
a64FADR // ADR/ADRP
|
||||||
a64FEXTR // EXTR
|
a64FEXTR // EXTR
|
||||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
|
||||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||||
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||||
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
|
||||||
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
|
||||||
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
|
||||||
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
|
||||||
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
|
||||||
a64FCRC32 // CRC32
|
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
|
||||||
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
a64FCRC32 // CRC32
|
||||||
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
|
||||||
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
|
||||||
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
a64FLSE // LSE atomics: LDADD, CAS, SWP
|
||||||
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
|
||||||
a64FCondCmp // conditional compare: CCMP, CCMN
|
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
|
||||||
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
a64FCondCmp // conditional compare: CCMP, CCMN
|
||||||
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
|
||||||
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
|
||||||
a64FAcqRel // acquire/release: LDAR family, STLR family
|
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
|
||||||
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
a64FAcqRel // acquire/release: LDAR family, STLR family
|
||||||
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
|
||||||
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
|
||||||
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
|
||||||
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
|
||||||
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
|
||||||
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
|
||||||
a64FVTBL // SIMD table lookup: VTBL
|
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
|
||||||
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
a64FVTBL // SIMD table lookup: VTBL
|
||||||
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
|
||||||
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
|
||||||
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
|
||||||
|
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
|
||||||
)
|
)
|
||||||
|
|
||||||
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
// a64Enc is one instruction's encoding: its bit layout (format) and the
|
||||||
@@ -487,6 +492,16 @@ func init() {
|
|||||||
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||||
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||||
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||||
|
// The widening multiplies: a 64-bit result riding the same layout, the
|
||||||
|
// three-operand forms reading the accumulate register as ZR.
|
||||||
|
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
|
||||||
|
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
|
||||||
|
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
|
||||||
|
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
|
||||||
|
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
|
||||||
|
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
|
||||||
|
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
|
||||||
|
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
|
||||||
|
|
||||||
// ---- move wide ----
|
// ---- move wide ----
|
||||||
// MOVZ/MOVN/MOVK
|
// MOVZ/MOVN/MOVK
|
||||||
@@ -536,6 +551,15 @@ func init() {
|
|||||||
// ---- bitfield ----
|
// ---- bitfield ----
|
||||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||||
|
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
|
||||||
|
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||||
|
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||||
|
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||||
|
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||||
|
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
|
||||||
|
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
|
||||||
|
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
|
||||||
|
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
|
||||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||||
@@ -716,6 +740,13 @@ func init() {
|
|||||||
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
||||||
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
||||||
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
||||||
|
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
|
||||||
|
// the source width.
|
||||||
|
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
|
||||||
|
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
|
||||||
|
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
|
||||||
|
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
|
||||||
|
"REV16W": 0x5ac00400,
|
||||||
}
|
}
|
||||||
for m, op := range dp1 {
|
for m, op := range dp1 {
|
||||||
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
||||||
@@ -734,7 +765,7 @@ func init() {
|
|||||||
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||||
|
|
||||||
// ---- system operations ----
|
// ---- system operations ----
|
||||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
|
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
||||||
a64InstrTable[m] = a64Enc{format: a64FSys}
|
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -785,6 +816,73 @@ func init() {
|
|||||||
for m, op := range lse {
|
for m, op := range lse {
|
||||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||||
}
|
}
|
||||||
|
// The remaining width and ordering spellings of the same shapes, and the
|
||||||
|
// CAS compare-and-swap family, word-verified against go tool asm.
|
||||||
|
lseMore := map[string]uint32{
|
||||||
|
"LDADDAB": 0x38a00000,
|
||||||
|
"LDADDAH": 0x78a00000,
|
||||||
|
"LDADDALB": 0x38e00000,
|
||||||
|
"LDADDALH": 0x78e00000,
|
||||||
|
"LDADDLB": 0x38600000,
|
||||||
|
"LDADDLD": 0xf8600000,
|
||||||
|
"LDADDLH": 0x78600000,
|
||||||
|
"LDADDLW": 0xb8600000,
|
||||||
|
"LDCLRAB": 0x38a01000,
|
||||||
|
"LDCLRAH": 0x78a01000,
|
||||||
|
"LDCLRALH": 0x78e01000,
|
||||||
|
"LDCLRB": 0x38201000,
|
||||||
|
"LDCLRD": 0xf8201000,
|
||||||
|
"LDCLRH": 0x78201000,
|
||||||
|
"LDCLRLB": 0x38601000,
|
||||||
|
"LDCLRLD": 0xf8601000,
|
||||||
|
"LDCLRLH": 0x78601000,
|
||||||
|
"LDCLRLW": 0xb8601000,
|
||||||
|
"LDCLRW": 0xb8201000,
|
||||||
|
"LDEORAB": 0x38a02000,
|
||||||
|
"LDEORAD": 0xf8a02000,
|
||||||
|
"LDEORAH": 0x78a02000,
|
||||||
|
"LDEORALB": 0x38e02000,
|
||||||
|
"LDEORALH": 0x78e02000,
|
||||||
|
"LDEORAW": 0xb8a02000,
|
||||||
|
"LDEORB": 0x38202000,
|
||||||
|
"LDEORD": 0xf8202000,
|
||||||
|
"LDEORH": 0x78202000,
|
||||||
|
"LDEORLB": 0x38602000,
|
||||||
|
"LDEORLD": 0xf8602000,
|
||||||
|
"LDEORLH": 0x78602000,
|
||||||
|
"LDEORLW": 0xb8602000,
|
||||||
|
"LDEORW": 0xb8202000,
|
||||||
|
"LDORAB": 0x38a03000,
|
||||||
|
"LDORAD": 0xf8a03000,
|
||||||
|
"LDORAH": 0x78a03000,
|
||||||
|
"LDORALH": 0x78e03000,
|
||||||
|
"LDORAW": 0xb8a03000,
|
||||||
|
"LDORB": 0x38203000,
|
||||||
|
"LDORD": 0xf8203000,
|
||||||
|
"LDORH": 0x78203000,
|
||||||
|
"LDORLB": 0x38603000,
|
||||||
|
"LDORLD": 0xf8603000,
|
||||||
|
"LDORLH": 0x78603000,
|
||||||
|
"LDORLW": 0xb8603000,
|
||||||
|
"LDORW": 0xb8203000,
|
||||||
|
"SWPAB": 0x38a08000,
|
||||||
|
"SWPAD": 0xf8a08000,
|
||||||
|
"SWPAH": 0x78a08000,
|
||||||
|
"SWPALH": 0x78e08000,
|
||||||
|
"SWPAW": 0xb8a08000,
|
||||||
|
"SWPB": 0x38208000,
|
||||||
|
"SWPH": 0x78208000,
|
||||||
|
"SWPLB": 0x38608000,
|
||||||
|
"SWPLD": 0xf8608000,
|
||||||
|
"SWPLH": 0x78608000,
|
||||||
|
"SWPLW": 0xb8608000,
|
||||||
|
"CASAD": 0xc8e07c00,
|
||||||
|
"CASALB": 0x08e0fc00,
|
||||||
|
"CASLW": 0x88a0fc00,
|
||||||
|
}
|
||||||
|
for m, op := range lseMore {
|
||||||
|
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||||
|
}
|
||||||
|
|
||||||
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||||
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||||
@@ -794,7 +892,12 @@ func init() {
|
|||||||
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
||||||
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
||||||
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
||||||
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
|
||||||
|
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
|
||||||
|
"NGC": 0xda000000, "NGCW": 0x5a000000,
|
||||||
|
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
|
||||||
|
"NEGSW": 0x6b000000,
|
||||||
|
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||||
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
||||||
}
|
}
|
||||||
for m, op := range dpsrExtra {
|
for m, op := range dpsrExtra {
|
||||||
@@ -835,6 +938,12 @@ func init() {
|
|||||||
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||||
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||||
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||||
|
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
||||||
|
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
||||||
|
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
||||||
|
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
||||||
|
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
||||||
|
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
|
||||||
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
||||||
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||||
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||||
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
||||||
|
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
||||||
|
// the across-vector reductions.
|
||||||
|
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||||
|
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
||||||
|
|
||||||
|
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
||||||
|
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
||||||
|
// and pairwise compares among them. Word-verified against go tool asm.
|
||||||
|
var a64SimdQOnly = map[string]bool{
|
||||||
|
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
|
||||||
|
"VFABS": true, "VFNEG": true, "VFSQRT": true,
|
||||||
|
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
|
||||||
|
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
|
||||||
|
"VFMAXV": true, "VFMAXNMV": true,
|
||||||
|
}
|
||||||
|
|
||||||
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
||||||
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
||||||
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
||||||
@@ -918,19 +1044,96 @@ var a64ArrBits = [a64ArrCount]uint32{
|
|||||||
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
|
||||||
// word and arrangement bit was read off go tool asm.
|
// word and arrangement bit was read off go tool asm.
|
||||||
var a64SimdVTable = map[string]a64SimdVSpec{
|
var a64SimdVTable = map[string]a64SimdVSpec{
|
||||||
"VADD": {0x0e208400, 0x7f, false},
|
"VADD": {0x0e208400, 0x7f, false},
|
||||||
"VSUB": {0x2e208400, 0x7f, false},
|
"VSUB": {0x2e208400, 0x7f, false},
|
||||||
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
|
||||||
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
|
||||||
"VEOR": {0x2e201c00, 0x03, false},
|
"VEOR": {0x2e201c00, 0x03, false},
|
||||||
"VORR": {0x0ea01c00, 0x03, false},
|
"VORR": {0x0ea01c00, 0x03, false},
|
||||||
"VADDP": {0x0e20bc00, 0x7f, false},
|
"VADDP": {0x0e20bc00, 0x7f, false},
|
||||||
"VZIP1": {0x0e003800, 0x7f, false},
|
"VZIP1": {0x0e003800, 0x7f, false},
|
||||||
"VZIP2": {0x0e007800, 0x7f, false},
|
"VZIP2": {0x0e007800, 0x7f, false},
|
||||||
"VCMEQ": {0x2e208c00, 0x7f, false},
|
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
"VCMGE": {0x0e203c00, 0x7f, false},
|
||||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
"VCMGT": {0x0e203400, 0x7f, false},
|
||||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
"VCMHI": {0x2e203400, 0x7f, false},
|
||||||
|
"VCMHS": {0x2e203c00, 0x7f, false},
|
||||||
|
// FP compares take H, S and D arrangements only (the toolchain rejects
|
||||||
|
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
||||||
|
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
// FP arithmetic shares the same arrangement restriction.
|
||||||
|
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
||||||
|
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
||||||
|
// tool asm.
|
||||||
|
"VBIC": {0x0e601c00, 0x7f, false},
|
||||||
|
"VBIF": {0x2ee01c00, 0x7f, false},
|
||||||
|
"VBIT": {0x6ea01c00, 0x7f, false},
|
||||||
|
"VBSL": {0x6e601c00, 0x7f, false},
|
||||||
|
"VCMTST": {0x0e208c00, 0x7f, false},
|
||||||
|
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||||
|
"VMLA": {0x4ea09400, 0x7f, false},
|
||||||
|
"VMLS": {0x6ea09400, 0x7f, false},
|
||||||
|
"VORN": {0x4ee01c00, 0x7f, false},
|
||||||
|
"VSHADD": {0x4ea00400, 0x7f, false},
|
||||||
|
"VSRHADD": {0x4ea01400, 0x7f, false},
|
||||||
|
"VUHADD": {0x6ea00400, 0x7f, false},
|
||||||
|
"VURHADD": {0x6ea01400, 0x7f, false},
|
||||||
|
"VSMAX": {0x4ea06400, 0x7f, false},
|
||||||
|
"VSMIN": {0x4ea06c00, 0x7f, false},
|
||||||
|
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
||||||
|
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
||||||
|
"VUMAX": {0x2e206400, 0x7f, false},
|
||||||
|
"VUMIN": {0x2e206c00, 0x7f, false},
|
||||||
|
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
||||||
|
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
||||||
|
"VSQADD": {0x4ea00c00, 0x7f, false},
|
||||||
|
"VUQADD": {0x6ea00c00, 0x7f, false},
|
||||||
|
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
||||||
|
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
||||||
|
"VSSHL": {0x4ee04400, 0x7f, false},
|
||||||
|
"VUSHL": {0x6ee04400, 0x7f, false},
|
||||||
|
"VUZP1": {0x0e001800, 0x7f, false},
|
||||||
|
"VUZP2": {0x4ec05800, 0x7f, false},
|
||||||
|
"VTRN1": {0x4ec02800, 0x7f, false},
|
||||||
|
"VTRN2": {0x4ec06800, 0x7f, false},
|
||||||
|
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||||
|
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||||
|
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||||
|
}
|
||||||
|
|
||||||
|
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
||||||
|
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
|
||||||
|
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
|
||||||
|
// combination for them, so they stay out and the encoder rejects the shape.
|
||||||
|
var a64SimdVZero = map[string]uint32{
|
||||||
|
"VCMEQ": 0x0e209800,
|
||||||
|
"VCMGT": 0x0e208800,
|
||||||
|
"VCMGE": 0x2e208800,
|
||||||
|
"VCMLT": 0x0e20a800,
|
||||||
|
"VCMLE": 0x2e209800,
|
||||||
|
// FP compares against (0.0): the register forms above carry the U and op
|
||||||
|
// bits; the zero forms reshape them.
|
||||||
|
"VFCMEQ": 0x0ea0d800,
|
||||||
|
"VFCMGE": 0x2ea0c800,
|
||||||
|
"VFCMGT": 0x0ea0c800,
|
||||||
|
"VFCMLE": 0x2ea0d800,
|
||||||
|
"VFCMLT": 0x0ea0e800,
|
||||||
}
|
}
|
||||||
|
|
||||||
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
||||||
@@ -939,8 +1142,41 @@ var a64SimdVTable = map[string]a64SimdVSpec{
|
|||||||
var a64SimdV2Table = map[string]a64SimdVSpec{
|
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||||
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||||
"VREV64": {0x0e200800, 0x3f, false},
|
"VREV64": {0x0e200800, 0x3f, false},
|
||||||
|
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||||
"VUADDLV": {0x2e303800, 0x3f, false},
|
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||||
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||||
|
// Two-register data-processing across one arrangement.
|
||||||
|
"VABS": {0x0e20b800, 0x7f, false},
|
||||||
|
"VNEG": {0x2e20b800, 0x7f, false},
|
||||||
|
"VCLS": {0x0e204800, 0x7f, false},
|
||||||
|
"VCLZ": {0x2e204800, 0x7f, false},
|
||||||
|
"VCNT": {0x0e205800, 0x7f, false},
|
||||||
|
"VNOT": {0x2e205800, 0x7f, false},
|
||||||
|
"VSQABS": {0x0e207800, 0x7f, false},
|
||||||
|
"VSQNEG": {0x2e207800, 0x7f, false},
|
||||||
|
"VRBIT": {0x6e605800, 0x7f, false},
|
||||||
|
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
||||||
|
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
||||||
|
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
||||||
|
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
||||||
|
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
||||||
|
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
||||||
|
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
||||||
|
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
||||||
|
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
||||||
|
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
||||||
|
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
||||||
|
// Across-vector reductions: the operand arrangement rides as usual and
|
||||||
|
// the destination stays a bare V register.
|
||||||
|
"VADDV": {0x0e31b800, 0x3f, false},
|
||||||
|
"VSMAXV": {0x0e30a800, 0x3f, false},
|
||||||
|
"VSMINV": {0x0e31a800, 0x3f, false},
|
||||||
|
"VUMAXV": {0x2e30a800, 0x3f, false},
|
||||||
|
"VUMINV": {0x2e31a800, 0x3f, false},
|
||||||
|
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
||||||
|
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
||||||
|
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
||||||
|
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
||||||
}
|
}
|
||||||
|
|
||||||
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||||
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
|
|||||||
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||||
var a64MSRRegOps = map[string]uint32{
|
var a64MSRRegOps = map[string]uint32{
|
||||||
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||||
|
"ELR_EL1": 0xd5184020,
|
||||||
}
|
}
|
||||||
|
|
||||||
// a64MSROps maps the system register names GOROOT writes to their fixed
|
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||||
|
|||||||
+164
-13
@@ -4,6 +4,7 @@
|
|||||||
package asm
|
package asm
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||||
@@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
|
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
|
||||||
// rejected instead of wrapping through int32.
|
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
|
||||||
func TestArm64AddSubImmRange(t *testing.T) {
|
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
|
||||||
for _, body := range []string{
|
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
|
||||||
"\tADD $0x100000000, R0, R1\n",
|
// Comparisons never split, and the W forms classify the 32-bit value. Every
|
||||||
"\tSUB $-0x100000000, R0, R1\n",
|
// word is go tool asm's own for the same source.
|
||||||
"\tCMP $0x100000000, R0\n",
|
func TestArm64AddSubImmWide(t *testing.T) {
|
||||||
} {
|
got := arm64Words(t, strings.Join([]string{
|
||||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
"\tADD $0xaaaaaa, R2, R3",
|
||||||
if len(errs) > 0 {
|
"\tSUB $0xaaaaaa, R2",
|
||||||
t.Fatalf("parse: %v", errs)
|
"\tADD $0x186a0, R2, R5",
|
||||||
|
"\tADD $0x1ffe00, R2, R3",
|
||||||
|
"\tADD $0x3fffffffc000, R5",
|
||||||
|
"\tADD $-100000, R2, R3",
|
||||||
|
"\tADD $-2048, R2, R3",
|
||||||
|
"\tCMP $0xaaaaaa, R2",
|
||||||
|
"\tCMP $0xffffffffffa0, R3",
|
||||||
|
"\tCMPW $27745, R2",
|
||||||
|
"\tCMPW $0x60060, R2",
|
||||||
|
"\tADDS $0xaaaaaa, R2, R3",
|
||||||
|
"\tADD $0x12345678, R2, R3",
|
||||||
|
"\tADDW $0x60060, R2",
|
||||||
|
"\tSUB $0xe7791f700, R3, R1",
|
||||||
|
"\tADDW $0x12345678, R2, R3",
|
||||||
|
"\tCMN $0x1000000, R2",
|
||||||
|
}, "\n")+"\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
|
||||||
|
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
|
||||||
|
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
|
||||||
|
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
|
||||||
|
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
|
||||||
|
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
|
||||||
|
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
|
||||||
|
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
|
||||||
|
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
|
||||||
|
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
|
||||||
|
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
|
||||||
|
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
|
||||||
|
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
|
||||||
|
0x11018042, 0x11418042, // ADDW $0x60060: W split
|
||||||
|
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
|
||||||
|
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
|
||||||
|
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
}
|
}
|
||||||
if _, err := AssembleFileARM64(f); err == nil {
|
}
|
||||||
t.Errorf("%s: expected an error, got none", body)
|
}
|
||||||
|
|
||||||
|
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
|
||||||
|
// three operands, the ROR shift on the logical group (and its rejection for
|
||||||
|
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
|
||||||
|
// alias with an element selector. Words are go tool asm's own.
|
||||||
|
func TestArm64CarryShiftAlias(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
|
||||||
|
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
|
||||||
|
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
|
||||||
|
want := []uint32{
|
||||||
|
0x9a1f0294, // ADC ZR, R20, R20
|
||||||
|
0x9a1f0284, // ADC ZR, R20, R4
|
||||||
|
0xfa1f008c, // SBCS ZR, R4, R12
|
||||||
|
0xfa0f008c, // SBCS R15, R4, R12
|
||||||
|
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
|
||||||
|
0x8ac18443, // AND R1 ROR 33, R2, R3
|
||||||
|
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
|
||||||
|
0xda0203e7, // SBC ZR, R7, R2
|
||||||
|
0x9b0eff77, // MSUB ZR, R27, R14, R23
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ROR on an arithmetic form is unallocated: the toolchain reports an
|
||||||
|
// unsupported shift operator.
|
||||||
|
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
if _, err := AssembleFileARM64(f); err == nil {
|
||||||
|
t.Error("ADD R1@>33: expected an error, got none")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
|
||||||
|
// operand with an element selector and inside a split register list: the
|
||||||
|
// aliases resolve textually where the parser carries the selector apart from
|
||||||
|
// the name. Words are go tool asm's own.
|
||||||
|
func TestArm64VecAliasElement(t *testing.T) {
|
||||||
|
src := `#include "textflag.h"
|
||||||
|
|
||||||
|
#define POLY V15
|
||||||
|
#define ACC0 V8
|
||||||
|
#define ACC1 V9
|
||||||
|
|
||||||
|
TEXT ·f(SB), NOSPLIT, $0-0
|
||||||
|
VMOV R1, POLY.D[0]
|
||||||
|
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||||
|
VLD1 (R0), [ACC0.B16]
|
||||||
|
VLD1.P (R0), [ACC0.B16, ACC1.B16]
|
||||||
|
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
|
||||||
|
RET
|
||||||
|
`
|
||||||
|
f, errs := parser.Parse("test_arm64.s", src)
|
||||||
|
if len(errs) > 0 {
|
||||||
|
t.Fatalf("parse: %v", errs)
|
||||||
|
}
|
||||||
|
img, err := AssembleFileARM64(f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("AssembleFileARM64: %v", err)
|
||||||
|
}
|
||||||
|
got := leWords(img.Code)
|
||||||
|
want := []uint32{
|
||||||
|
0x4e081c2f, // INS V15.D[0], R1
|
||||||
|
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
|
||||||
|
0x4c407008, // VLD1 (R0), [V8.B16]
|
||||||
|
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
|
||||||
|
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
|
||||||
|
// once the value leaves every imm12 form: a constant sequence into REGTMP
|
||||||
|
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
|
||||||
|
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
|
||||||
|
// tool asm's own.
|
||||||
|
func TestArm64AddSubImmBeyond32(t *testing.T) {
|
||||||
|
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
|
||||||
|
want := []uint32{
|
||||||
|
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||||
|
0x8b1b0001, // ADD R27, R0, R1
|
||||||
|
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
|
||||||
|
0xcb1b0001, // SUB R27, R0, R1
|
||||||
|
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||||
|
0xeb1b001f, // CMP R27, R0
|
||||||
|
0xd65f03c0, // RET
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||||
|
}
|
||||||
|
for i := range want {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Vendored
+48
@@ -0,0 +1,48 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Carry arithmetic, logical shifts, register aliases with element selectors
|
||||||
|
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
|
||||||
|
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
#define acc0 V8
|
||||||
|
#define acc1 V9
|
||||||
|
#define const0 R15
|
||||||
|
#define POLY V15
|
||||||
|
|
||||||
|
// carry pins the ADC/SBC family: the $0 spellings in two and three
|
||||||
|
// operands, and the register-carry forms.
|
||||||
|
TEXT ·carry(SB), NOSPLIT, $0-0
|
||||||
|
ADC $0, R20
|
||||||
|
ADC $0, R20, R4
|
||||||
|
SBCS $0, R4
|
||||||
|
SBCS $0, R4, R12
|
||||||
|
SBCS R15, R4, R12
|
||||||
|
SBC $0, R1
|
||||||
|
ADCSW $0, R2, R3
|
||||||
|
RET
|
||||||
|
|
||||||
|
// shift pins the shifted-register forms including ROR, which only the
|
||||||
|
// logical family accepts.
|
||||||
|
TEXT ·shift(SB), NOSPLIT, $0-0
|
||||||
|
ANDW R9@>7, R19, R26
|
||||||
|
AND R1@>33, R2, R3
|
||||||
|
ADD R1<<11, R2, R3
|
||||||
|
SUB R1->33, R2
|
||||||
|
ORR R5<<2, R6, R7
|
||||||
|
RET
|
||||||
|
|
||||||
|
// vecalias pins the vector aliases with element selectors and the
|
||||||
|
// structure loads with aliased members.
|
||||||
|
TEXT ·vecalias(SB), NOSPLIT, $0-0
|
||||||
|
MOVD $0xC2, R1
|
||||||
|
VMOV R1, POLY.D[0]
|
||||||
|
VMOV R0, POLY.D[1]
|
||||||
|
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||||
|
VLD1 (R0), [acc0.B16]
|
||||||
|
VLD1.P (R0), [acc0.B16, acc1.B16]
|
||||||
|
VST1 [acc0.B16, acc1.B16], (R1)
|
||||||
|
VST1.P [acc0.B16, acc1.B16], 32(R1)
|
||||||
|
RET
|
||||||
Vendored
+70
@@ -0,0 +1,70 @@
|
|||||||
|
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||||
|
// SPDX-License-Identifier: BSD-3-Clause
|
||||||
|
|
||||||
|
// Wide-immediate arithmetic: every classification band of the ADD/SUB
|
||||||
|
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
|
||||||
|
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
|
||||||
|
// their materialised fallback. Byte-for-byte against go tool asm.
|
||||||
|
|
||||||
|
#include "textflag.h"
|
||||||
|
|
||||||
|
// imm12 covers the plain and shifted-by-12 imm12 forms.
|
||||||
|
TEXT ·imm12(SB), NOSPLIT, $0-0
|
||||||
|
ADD $1, R2, R3
|
||||||
|
ADD $0x000aaa, R2, R3
|
||||||
|
ADD $0xaaa000, R2
|
||||||
|
SUB $0x000aaa, R2, R3
|
||||||
|
SUB $0xaaa000, R2
|
||||||
|
ADDW $40960, R0
|
||||||
|
CMP $40960, R0
|
||||||
|
CMPW $40960, R0
|
||||||
|
RET
|
||||||
|
|
||||||
|
// split pins the ADDCON2 band: two imm12 instructions, low half first.
|
||||||
|
TEXT ·split(SB), NOSPLIT, $0-0
|
||||||
|
ADD $0xaaaaaa, R2, R3
|
||||||
|
SUB $0xaaaaaa, R2
|
||||||
|
ADD $0x186a0, R2, R5
|
||||||
|
SUB $0x186a0, R2, R3
|
||||||
|
ADDW $0x60060, R2
|
||||||
|
RET
|
||||||
|
|
||||||
|
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
|
||||||
|
// MOVN for the complement form, the bitmask ORR otherwise.
|
||||||
|
TEXT ·regtmp(SB), NOSPLIT, $0-0
|
||||||
|
ADD $0x1ffe00, R2, R3
|
||||||
|
ADD $0x3fffffffc000, R5
|
||||||
|
ADD $-2048, R2, R3
|
||||||
|
ADD $-100000, R2, R3
|
||||||
|
CMP $0x1000000, R2
|
||||||
|
CMP $0x100000000, R0
|
||||||
|
SUB $-0x100000000, R0, R1
|
||||||
|
RET
|
||||||
|
|
||||||
|
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
|
||||||
|
// compare forms that never split.
|
||||||
|
TEXT ·movseq(SB), NOSPLIT, $0-0
|
||||||
|
ADD $0x12345678, R2, R3
|
||||||
|
SUB $0xe7791f700, R3, R1
|
||||||
|
CMP $0xaaaaaa, R2
|
||||||
|
CMP $0xffffffffffa0, R3
|
||||||
|
CMPW $27745, R2
|
||||||
|
CMPW $0x60060, R2
|
||||||
|
ADDS $0xaaaaaa, R2, R3
|
||||||
|
CMN $0x1000000, R2
|
||||||
|
ADDW $0x12345678, R2, R3
|
||||||
|
RET
|
||||||
|
|
||||||
|
// logical covers the bitmask immediates of the logical family and the
|
||||||
|
// materialised fallback for the values a bitmask cannot carry.
|
||||||
|
TEXT ·logical(SB), NOSPLIT, $0-0
|
||||||
|
AND $0x3ff00000, R2, R3
|
||||||
|
BIC $0x22220000, R3, R4
|
||||||
|
ORR $0x3ff00000, R2
|
||||||
|
EOR $0x3ff00000, R2, R3
|
||||||
|
ANDS $0x3ff00000, R2
|
||||||
|
ORNW $0x3ff00000, R2
|
||||||
|
EONW $0x3ff00000, R2
|
||||||
|
BICSW $0x6006000060060, R5
|
||||||
|
TST $0x4900000049, R0
|
||||||
|
RET
|
||||||
Reference in New Issue
Block a user