feat(arm64): wide immediates, SIMD compare and system operand forms
Assisted-by: GLM 5.3 Flash
This commit is contained in:
+566
-105
@@ -5,6 +5,7 @@ package asm
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/bits"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
|
||||
"FMOVS", "FMOVD":
|
||||
return arm64MovSize(mnem, ops, fi)
|
||||
case "ADD", "ADDW", "SUB", "SUBW":
|
||||
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
|
||||
"ADDS", "ADDSW", "SUBS", "SUBSW":
|
||||
if len(ops) >= 2 && isImmOperand(ops[0]) {
|
||||
v := arm64Imm64(ops[0])
|
||||
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
|
||||
if v >= 0 && v <= 0xFFF {
|
||||
return 4
|
||||
// Size exactly as the encoder will emit: a single imm12 word, the
|
||||
// two-word ADDCON2 split, or a materialisation into REGTMP plus
|
||||
// the register form. Anything else would desynchronise the label
|
||||
// offsets of pass 1 from the bytes pass 2 lays down.
|
||||
if v, ok := arm64ImmOperandValue(ops[0]); ok {
|
||||
rn, rd := 0, 0
|
||||
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
|
||||
rd = n
|
||||
}
|
||||
if v >= -2048 && v < 0 {
|
||||
return 4
|
||||
if len(ops) == 3 {
|
||||
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
|
||||
rn = n
|
||||
}
|
||||
// Larger immediates need MOV materialisation + op.
|
||||
return 8
|
||||
}
|
||||
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
|
||||
return 4 * len(ws)
|
||||
}
|
||||
}
|
||||
return 4
|
||||
}
|
||||
}
|
||||
return 4
|
||||
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return encodeARM64Bitfield(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
|
||||
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
|
||||
}
|
||||
|
||||
// EXTR.
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
|
||||
return encodeARM64Extr(mnem, enc.op, ops)
|
||||
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
|
||||
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
|
||||
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
|
||||
// this check precedes the plain SIMD3 path below.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok {
|
||||
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
|
||||
// only in the zero-immediate form (a64SimdVZero), so they route here with
|
||||
// an empty register-form spec.
|
||||
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
|
||||
return encodeARM64SimdV(mnem, spec, ops)
|
||||
}
|
||||
|
||||
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
}
|
||||
|
||||
// SIMD table lookup.
|
||||
if mnem == "VTBL" {
|
||||
return encodeARM64VTBL(ops)
|
||||
if mnem == "VTBL" || mnem == "VTBX" {
|
||||
return encodeARM64VTBL(mnem, ops)
|
||||
}
|
||||
|
||||
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
|
||||
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
}
|
||||
op := ops[0]
|
||||
|
||||
// Branch to the program counter itself: JMP (PC) spins forever. The
|
||||
// toolchain encodes it as an unconditional branch with a zero offset.
|
||||
// Branch to the program counter: JMP (PC) spins forever, and a spelled
|
||||
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
|
||||
// units. The toolchain encodes both as a plain branch of that offset.
|
||||
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
|
||||
if link {
|
||||
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
|
||||
rel := op.Addr.Offset
|
||||
if rel < -(1<<25) || rel >= (1<<25) {
|
||||
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
|
||||
}
|
||||
return a64wordLE(a64Branch(0, 0)), nil
|
||||
bop := uint32(0) // B
|
||||
if link {
|
||||
bop = 1 // BL
|
||||
}
|
||||
return a64wordLE(a64Branch(bop, int32(rel))), nil
|
||||
}
|
||||
|
||||
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
|
||||
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
|
||||
// the ADD/SUB-with-flags family an add/sub immediate.
|
||||
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
|
||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
|
||||
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
|
||||
mnem == "MVN" || mnem == "MVNW" ||
|
||||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
|
||||
|
||||
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
|
||||
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
|
||||
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
var logical bool
|
||||
switch mnem {
|
||||
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
|
||||
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
|
||||
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
|
||||
"TST", "TSTW":
|
||||
logical = true
|
||||
}
|
||||
if logical {
|
||||
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
}
|
||||
inverted := false
|
||||
switch mnem {
|
||||
case "BIC", "BICW", "BICS", "BICSW":
|
||||
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
|
||||
inverted = true
|
||||
}
|
||||
if inverted {
|
||||
@@ -700,8 +727,38 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
}
|
||||
n, immr, imms, ok := a64LogicalImm(v, width)
|
||||
if !ok {
|
||||
// Beyond the bitmask immediates the toolchain materialises
|
||||
// the constant into REGTMP (R27) and uses the register form
|
||||
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
|
||||
// value, so the materialisation uses v before any inversion.
|
||||
written := v
|
||||
if inverted {
|
||||
written = ^v
|
||||
}
|
||||
width := mnem
|
||||
if strings.HasSuffix(mnem, "W") {
|
||||
width = "MOVW"
|
||||
} else {
|
||||
width = "MOVD"
|
||||
}
|
||||
mw, merr := encodeARM64LoadImm(27, written, width)
|
||||
var rn, rd int
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
rn = arm64RegNum(operandRegName(ops[1]))
|
||||
rd = arm64RegNum(operandRegName(ops[2]))
|
||||
default:
|
||||
rd = arm64RegNum(operandRegName(ops[1]))
|
||||
rn = rd
|
||||
}
|
||||
if isCmp {
|
||||
rd = 31
|
||||
}
|
||||
if merr != nil || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
|
||||
}
|
||||
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
|
||||
}
|
||||
opc := (baseOp >> 29) & 7
|
||||
sf := (baseOp >> 31) & 1
|
||||
var rn, rd int
|
||||
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
|
||||
}
|
||||
if !isExtend {
|
||||
if amount < 0 || amount > 63 {
|
||||
// ROR rides the shifted-register field only for the logical
|
||||
// group; the toolchain reports "unsupported shift operator" for
|
||||
// the arithmetic forms, whose shift=11 encoding is unallocated.
|
||||
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
|
||||
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
|
||||
}
|
||||
// The imm6 field is 5 bits and truncates at the 32-bit width.
|
||||
limit := 63
|
||||
if strings.HasSuffix(mnem, "W") {
|
||||
limit = 31
|
||||
}
|
||||
if amount < 0 || amount > limit {
|
||||
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
|
||||
}
|
||||
// SP-based ADD/SUB have no shifted-register encoding: the
|
||||
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
|
||||
switch len(ops) {
|
||||
case 3:
|
||||
// The carry family carries an immediate spelling in three operands
|
||||
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
|
||||
// operand, the same shape the two-operand form takes.
|
||||
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
|
||||
if v := arm64Imm64(ops[0]); v != 0 {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
// OP Rm, Rn, Rd
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
|
||||
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
|
||||
// shifted-register group, the only forms whose register operand accepts the
|
||||
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
|
||||
func arm64LogicalShifted(mnem string) bool {
|
||||
switch mnem {
|
||||
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
|
||||
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
|
||||
"TST", "TSTW", "MVN", "MVNW":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
|
||||
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
|
||||
// data-processing instructions the toolchain accepts an immediate $0
|
||||
// operand spelling for.
|
||||
func arm64CarryOp(mnem string) bool {
|
||||
switch mnem {
|
||||
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// arm64RegMod reports whether a register operand carries the shifted-register
|
||||
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
|
||||
// option (R0.UXTW, R3.SXTW<<2).
|
||||
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
|
||||
|
||||
// arm64RegModifier resolves a modified register operand: the register number,
|
||||
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
|
||||
// option (UXTB=0..SXTX=7) with its shift amount.
|
||||
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
|
||||
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
|
||||
// before the operator match.
|
||||
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
|
||||
name := operandRegName(op)
|
||||
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||
if before, after, ok0 := strings.Cut(name, "."); ok0 {
|
||||
switch strings.ToUpper(strings.TrimSpace(after)) {
|
||||
case "UXTB":
|
||||
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
||||
if rm < 0 {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
||||
amount, ok = arm64ShiftAmount(shift)
|
||||
if !ok || amount < 0 || amount > 4 {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
return rm, 0, extendOpt, true, amount, true
|
||||
}
|
||||
shiftKind = 0 // LSL
|
||||
shift := strings.TrimSpace(op.Addr.Shift)
|
||||
switch {
|
||||
case strings.HasPrefix(shift, "<<"):
|
||||
shiftKind = 0
|
||||
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
|
||||
default:
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
amount, ok = arm64ShiftAmount(op.Addr.Shift)
|
||||
amount, ok = arm64ShiftAmount(shift)
|
||||
if !ok {
|
||||
return 0, 0, 0, false, 0, false
|
||||
}
|
||||
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
|
||||
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
|
||||
// MUL mnemonic, not to these.
|
||||
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
// The widening three-operand forms (SMULL, UMNEGL, …) read the
|
||||
// accumulate register as ZR, already preset in the table's base word.
|
||||
if len(ops) == 3 {
|
||||
switch mnem {
|
||||
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
|
||||
default:
|
||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
|
||||
}
|
||||
rm := arm64RegNum(operandRegName(ops[0]))
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[2]))
|
||||
if rm < 0 || rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
if len(ops) != 4 {
|
||||
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
|
||||
}
|
||||
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
|
||||
// ---- ADD/SUB immediate ----
|
||||
|
||||
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
|
||||
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
|
||||
// following the toolchain's immediate classification (asm7.go conclass and
|
||||
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
|
||||
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
|
||||
// otherwise a constant materialisation into REGTMP (R27) followed by the
|
||||
// register form.
|
||||
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 2 && len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
v, ok := arm64ImmOperandValue(ops[0])
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
|
||||
}
|
||||
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
|
||||
rn := rd
|
||||
if len(ops) == 3 {
|
||||
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
|
||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
sf := uint32(1) // 64-bit
|
||||
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
|
||||
sf = 0 // 32-bit
|
||||
}
|
||||
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
|
||||
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
|
||||
// CMP/CMN discard the destination.
|
||||
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
|
||||
rd = 31 // ZR
|
||||
}
|
||||
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%s: %w", mnem, err)
|
||||
}
|
||||
return a64WordsLE(ws...), nil
|
||||
}
|
||||
|
||||
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
|
||||
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
|
||||
// their W forms. rn and rd are resolved register numbers (a comparison
|
||||
// discards rd, so the caller passes 31).
|
||||
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
|
||||
w := strings.HasSuffix(mnem, "W")
|
||||
sf := uint32(1) // 64-bit
|
||||
d := v
|
||||
if w {
|
||||
sf = 0 // 32-bit
|
||||
// The W forms classify the 32-bit value (asm7.go con32class).
|
||||
d = int64(uint32(v))
|
||||
}
|
||||
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
|
||||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
|
||||
op := uint32(0) // ADD
|
||||
S := uint32(0)
|
||||
if isSub {
|
||||
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if isS {
|
||||
S = 1
|
||||
}
|
||||
single := func(sh, imm12 uint32) []uint32 {
|
||||
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
|
||||
}
|
||||
|
||||
if v >= 0 && v <= 0xFFF {
|
||||
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
|
||||
// imm12: plain, then the one-shifted-by-12 form.
|
||||
if d >= 0 && d <= 0xFFF {
|
||||
return single(0, uint32(d)), nil
|
||||
}
|
||||
if v >= -2048 && v < 0 {
|
||||
// Encode as the opposite operation with positive immediate.
|
||||
opp := op ^ 1
|
||||
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
|
||||
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
|
||||
return single(1, uint32(d>>12)), nil
|
||||
}
|
||||
// Try with shift by 12.
|
||||
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
|
||||
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
|
||||
|
||||
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
|
||||
// split into two imm12 instructions, low half first (asm7.go case 48).
|
||||
// The encoding is complete in itself: no REGTMP, no register form. The S
|
||||
// forms must not break addition/subtraction, so the toolchain
|
||||
// reclassifies them and falls through to the materialisation below.
|
||||
dm := ^d
|
||||
if w {
|
||||
dm = ^d & 0xFFFFFFFF
|
||||
}
|
||||
// The imm12 field cannot carry the value; rejecting (rather than
|
||||
// truncating) matches the toolchain, which reports the same shape.
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
|
||||
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
|
||||
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
|
||||
return []uint32{
|
||||
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
|
||||
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Constant into REGTMP (R27), then the register form. The first word
|
||||
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
|
||||
// the complement form, the bitmask ORR otherwise, and the full
|
||||
// omovlconst sequence when no single word carries the value.
|
||||
var seq []uint32
|
||||
switch s := arm64Movcon(d); {
|
||||
case s >= 0:
|
||||
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
|
||||
case arm64Movcon(dm) >= 0:
|
||||
s := arm64Movcon(dm)
|
||||
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
|
||||
case isBitcon:
|
||||
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
|
||||
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
|
||||
default:
|
||||
seq = arm64MovLConst(d, sf)
|
||||
}
|
||||
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
|
||||
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
|
||||
for i := range seq[:len(seq)-1] {
|
||||
seq[i] |= 27 // REGTMP
|
||||
}
|
||||
return seq, nil
|
||||
}
|
||||
|
||||
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
|
||||
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
|
||||
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
|
||||
// returned with the destination field clear so the caller can OR its own
|
||||
// register in. movcon and movcon-of-complement must fail for d before this
|
||||
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
|
||||
func arm64MovLConst(d int64, sf uint32) []uint32 {
|
||||
if sf == 0 {
|
||||
// omovlconst AMOVW: both 16-bit halves, low first.
|
||||
return []uint32{
|
||||
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
|
||||
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
|
||||
}
|
||||
}
|
||||
dn := ^d
|
||||
var immh [4]uint64
|
||||
zero, neg := 0, 0
|
||||
for i := range immh {
|
||||
immh[i] = uint64(d>>(i*16)) & 0xFFFF
|
||||
switch immh[i] {
|
||||
case 0:
|
||||
zero++
|
||||
case 0xFFFF:
|
||||
neg++
|
||||
}
|
||||
}
|
||||
mw := func(opc uint32, val int64, chunk int) uint32 {
|
||||
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
|
||||
}
|
||||
var os []uint32
|
||||
switch {
|
||||
case zero == 2:
|
||||
// one MOVZ and one MOVK
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(2, d, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
case neg == 2:
|
||||
// one MOVN and one MOVK
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(0, dn, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
default:
|
||||
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
|
||||
// a single MOVK (constants from strength-reduced division).
|
||||
if zero == 0 && neg == 0 {
|
||||
for i := range 4 {
|
||||
mask := uint64(0xFFFF) << (i * 16)
|
||||
for period := 2; period <= 32; period *= 2 {
|
||||
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
|
||||
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
|
||||
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
|
||||
os = append(os, mw(3, d, i))
|
||||
return os
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case zero >= 1:
|
||||
// one MOVZ and up to three MOVKs
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(2, d, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0 {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
case neg >= 1:
|
||||
// one MOVN and up to three MOVKs
|
||||
i := 0
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(0, dn, i))
|
||||
i++
|
||||
break
|
||||
}
|
||||
}
|
||||
for ; i < 4; i++ {
|
||||
if immh[i] != 0xFFFF {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
default:
|
||||
// one MOVZ and three MOVKs
|
||||
os = append(os, mw(2, d, 0))
|
||||
for i := 1; i < 4; i++ {
|
||||
os = append(os, mw(3, d, i))
|
||||
}
|
||||
}
|
||||
}
|
||||
return os
|
||||
}
|
||||
|
||||
// ---- MOV pseudo-instruction ----
|
||||
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
|
||||
var ws []uint32
|
||||
first := true
|
||||
for i := range 4 {
|
||||
chunk := (d >> uint(i*16)) & 0xFFFF
|
||||
if chunk == 0 {
|
||||
continue
|
||||
}
|
||||
if first {
|
||||
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
|
||||
first = false
|
||||
} else {
|
||||
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
|
||||
}
|
||||
}
|
||||
if len(ws) == 0 {
|
||||
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
|
||||
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
|
||||
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
|
||||
// for the first special 16-bit chunk, then MOVK per remaining one, with
|
||||
// the bitmask-plus-fixup shortcut for strength-reduced constants).
|
||||
ws := arm64MovLConst(d, sf)
|
||||
for i := range ws {
|
||||
ws[i] |= uint32(rd)
|
||||
}
|
||||
return a64WordsLE(ws...), nil
|
||||
}
|
||||
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
|
||||
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
|
||||
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
|
||||
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
|
||||
// with imms = lsb+width-1.
|
||||
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
|
||||
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
|
||||
}
|
||||
lsb := arm64Imm64(ops[0])
|
||||
width := arm64Imm64(ops[2])
|
||||
rn := arm64RegNum(operandRegName(ops[1]))
|
||||
rd := arm64RegNum(operandRegName(ops[3]))
|
||||
if rn < 0 || rd < 0 {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
bits := int64(32) << (baseOp >> 31 & 1)
|
||||
if lsb < 0 || lsb >= bits {
|
||||
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
|
||||
}
|
||||
if width < 1 || width > bits || lsb+width > bits {
|
||||
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
|
||||
}
|
||||
var immr, imms int64
|
||||
switch mnem {
|
||||
case "BFXIL", "BFXILW":
|
||||
immr, imms = lsb, lsb+width-1
|
||||
default: // BFI, SBFIZ, UBFIZ
|
||||
immr, imms = (-lsb)%bits, width-1
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
|
||||
}
|
||||
|
||||
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
|
||||
// third field carries Rm or a 5-bit immediate in the same bits, at the
|
||||
// toolchain's choice of register or immediate operand.
|
||||
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
|
||||
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
|
||||
// PRFM (Rn), $imm|<op>
|
||||
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
// Operand-less returns and pointer-authentication hints.
|
||||
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
|
||||
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
|
||||
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
|
||||
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
|
||||
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
|
||||
if len(ops) != 0 {
|
||||
return nil, fmt.Errorf("%s expects no operand", mnem)
|
||||
}
|
||||
return a64wordLE(w), nil
|
||||
}
|
||||
switch mnem {
|
||||
case "BRK", "SVC":
|
||||
base := uint32(0xd4200000)
|
||||
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||
}
|
||||
return a64wordLE(base | uint32(v)<<5), nil
|
||||
case "DMB", "DSB", "ISB":
|
||||
case "DMB", "DSB", "ISB", "CLREX":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if v < 0 || v > 15 {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
|
||||
}
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
|
||||
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<8), nil
|
||||
case "HINT":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 127 {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
|
||||
}
|
||||
return a64wordLE(0xd503201f | uint32(v)<<5), nil
|
||||
case "BTI":
|
||||
op := operandRegName(ops[0])
|
||||
base, ok := map[string]uint32{"C": 0xd503245f}[op]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
|
||||
}
|
||||
return a64wordLE(base), nil
|
||||
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
|
||||
if len(ops) != 1 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("%s expects $immediate", mnem)
|
||||
}
|
||||
v := arm64Imm64(ops[0])
|
||||
if v < 0 || v > 0xFFFF {
|
||||
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
|
||||
}
|
||||
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
|
||||
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
|
||||
"DCPS3": 0xd4a00003}[mnem]
|
||||
return a64wordLE(base | uint32(v)<<5), nil
|
||||
case "DC":
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("DC expects <op>, Rn")
|
||||
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
}
|
||||
return a64wordLE(base | uint32(rd)&31), nil
|
||||
case "MSR":
|
||||
if len(ops) != 2 || !isImmOperand(ops[0]) {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
|
||||
}
|
||||
if !isImmOperand(ops[0]) {
|
||||
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
|
||||
base, ok := a64MSRRegOps[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
|
||||
}
|
||||
rs := arm64RegNum(operandRegName(ops[0]))
|
||||
if rs < 0 {
|
||||
return nil, fmt.Errorf("MSR: invalid source register")
|
||||
}
|
||||
return a64wordLE(base | uint32(rs)&31), nil
|
||||
}
|
||||
base, ok := a64MSROps[operandRegName(ops[1])]
|
||||
if !ok {
|
||||
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
|
||||
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
|
||||
func specBit(i int) uint16 { return 1 << uint(i) }
|
||||
|
||||
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
|
||||
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
|
||||
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
|
||||
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
|
||||
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
|
||||
return true
|
||||
}
|
||||
if !strings.HasPrefix(mnem, "VFCM") {
|
||||
return false
|
||||
}
|
||||
s := strings.Join(strings.Fields(op.Raw), "")
|
||||
s = strings.TrimPrefix(s, "$")
|
||||
s = strings.Trim(s, "()")
|
||||
return s == "0" || s == "0.0"
|
||||
}
|
||||
|
||||
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
|
||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
|
||||
// zero immediate takes its compare-against-zero form instead, and the
|
||||
// polynomial multiplies read the arrangement from their source operands
|
||||
// alone, the result spelling (H8, Q1) riding no encoding bits.
|
||||
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
|
||||
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
|
||||
// their compare-against-zero form instead, and the polynomial multiplies read
|
||||
// the arrangement from their source operands alone, the result spelling
|
||||
// (H8, Q1) riding no encoding bits.
|
||||
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
|
||||
if mnem == "VPMULL" || mnem == "VPMULL2" {
|
||||
if len(ops) != 3 {
|
||||
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
rd, _ := arm64VecOf(ops[2])
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
|
||||
}
|
||||
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
|
||||
if arm64Imm64(ops[0]) != 0 {
|
||||
if len(ops) == 3 && isImmOperand(ops[0]) {
|
||||
base, ok := a64SimdVZero[mnem]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
if !arm64SimdZeroImm(mnem, ops[0]) {
|
||||
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
|
||||
}
|
||||
vn, ok1 := arm64VecOf(ops[1])
|
||||
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
|
||||
return nil, fmt.Errorf("invalid register operand in %s", mnem)
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
|
||||
allowed := uint16(0x7f)
|
||||
if strings.HasPrefix(mnem, "VFCM") {
|
||||
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
|
||||
}
|
||||
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
if len(ops) != 3 {
|
||||
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
|
||||
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
|
||||
if spec.fixed {
|
||||
arrBits = 0
|
||||
}
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
|
||||
}
|
||||
|
||||
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
|
||||
}
|
||||
// VUADDLV spells its arrangement on the source alone; the rest take it
|
||||
// on both.
|
||||
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
arrBits := a64ArrBits[arr]
|
||||
if a64SimdQOnly[mnem] {
|
||||
arrBits &= 1 << 30
|
||||
}
|
||||
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
|
||||
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
|
||||
// index register rides bits 19:16, the first table register bits 9:5, the
|
||||
// destination bits 4:0 and the table length (registers minus one) bits
|
||||
// 14:13. The table registers must be consecutive.
|
||||
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
||||
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
|
||||
if len(ops) < 3 {
|
||||
return nil, fmt.Errorf("VTBL expects index, table list and destination")
|
||||
}
|
||||
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
|
||||
default:
|
||||
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
|
||||
}
|
||||
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
base := uint32(0x0e000000)
|
||||
if mnem == "VTBX" {
|
||||
base |= 1 << 12
|
||||
}
|
||||
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
|
||||
}
|
||||
|
||||
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
|
||||
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
|
||||
}
|
||||
var immval int64
|
||||
switch mnem {
|
||||
case "VSHL":
|
||||
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
|
||||
if sh < 0 || sh >= esize {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
|
||||
}
|
||||
immval = esize + sh
|
||||
default: // VUSHR, VSRI
|
||||
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
|
||||
if sh < 1 || sh > esize {
|
||||
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
|
||||
}
|
||||
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
sym *ast.Symbol // the parsed frame-relative reference (when mem)
|
||||
}
|
||||
aliases := map[string]alias{}
|
||||
raws := map[string]string{}
|
||||
for _, d := range f.Decls {
|
||||
pre, ok := d.(*ast.Preproc)
|
||||
if !ok {
|
||||
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
|
||||
continue
|
||||
}
|
||||
raws[name] = body
|
||||
}
|
||||
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
|
||||
// transitively until nothing changes, bounded against cycles.
|
||||
for range 8 {
|
||||
changed := false
|
||||
for name, body := range raws {
|
||||
if next, ok := raws[body]; ok && next != body {
|
||||
raws[name] = next
|
||||
changed = true
|
||||
}
|
||||
}
|
||||
if !changed {
|
||||
break
|
||||
}
|
||||
}
|
||||
for name, body := range raws {
|
||||
isReg := func(s string) bool {
|
||||
if arm64RegNum(s) >= 0 {
|
||||
return true
|
||||
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
return
|
||||
}
|
||||
|
||||
// replace rewrites whole-word occurrences of the alias names in s.
|
||||
replace := func(s string) string {
|
||||
if s == "" {
|
||||
return s
|
||||
}
|
||||
out := strings.Fields(s)
|
||||
for i, w := range out {
|
||||
if a, ok := aliases[w]; ok {
|
||||
out[i] = a.raw
|
||||
}
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return s
|
||||
}
|
||||
return strings.Join(out, " ")
|
||||
}
|
||||
// replaceToken rewrites an operand whose whole text is one alias use
|
||||
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
|
||||
// ending at a non-identifier character.
|
||||
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
return s, false
|
||||
}
|
||||
|
||||
// replaceScan rewrites alias uses inside a composite operand (a
|
||||
// parenthesised memory operand or a bracketed register list): every
|
||||
// identifier run of word and dot characters is matched against the alias
|
||||
// names, everything else copies verbatim. The whitespace-split replace
|
||||
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
|
||||
// punctuation attached.
|
||||
replaceScan := func(s string) string {
|
||||
var b strings.Builder
|
||||
for i := 0; i < len(s); {
|
||||
if isAliasWordByte(s[i]) || s[i] == '.' {
|
||||
j := i
|
||||
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
|
||||
j++
|
||||
}
|
||||
if nn, ok := replaceToken(s[i:j]); ok {
|
||||
b.WriteString(nn)
|
||||
} else {
|
||||
b.WriteString(s[i:j])
|
||||
}
|
||||
i = j
|
||||
continue
|
||||
}
|
||||
b.WriteByte(s[i])
|
||||
i++
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
for _, d := range f.Decls {
|
||||
t, ok := d.(*ast.Text)
|
||||
if !ok {
|
||||
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
op.Raw = a.raw
|
||||
continue
|
||||
}
|
||||
op.Addr.Sym.Name = nn
|
||||
op.Addr.Sym.Raw = nn
|
||||
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
|
||||
// The span shape depends on what trailed the
|
||||
// name: an element or arrangement selector
|
||||
// (POLY.D[0], POLY.B16) rides in Shift and folds
|
||||
// back onto the rewritten token; a shift
|
||||
// operator stays in Shift while the span carries
|
||||
// the bare register; a split list keeps its
|
||||
// closing bracket, so the rewrite goes through
|
||||
// the scan.
|
||||
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
|
||||
switch {
|
||||
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
|
||||
// Element or arrangement selectors and the
|
||||
// closing bracket of a split list belong to
|
||||
// the token text.
|
||||
op.Raw = nn + sfx
|
||||
op.Addr.Shift = ""
|
||||
case op.Addr.Shift != "":
|
||||
op.Raw = nn
|
||||
default:
|
||||
op.Raw = replaceScan(op.Raw)
|
||||
}
|
||||
continue
|
||||
}
|
||||
}
|
||||
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
|
||||
// [V0.B16, V1.B16] with aliased members.
|
||||
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
|
||||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
|
||||
op.Raw = replace(op.Raw)
|
||||
op.Raw = replaceScan(op.Raw)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+238
-1
@@ -308,6 +308,8 @@ const (
|
||||
a64CondLT = 0xb
|
||||
a64CondGT = 0xc
|
||||
a64CondLE = 0xd
|
||||
a64CondAL = 0xe
|
||||
a64CondNV = 0xf
|
||||
)
|
||||
|
||||
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
|
||||
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
|
||||
"LT": a64CondLT,
|
||||
"GT": a64CondGT,
|
||||
"LE": a64CondLE,
|
||||
"AL": a64CondAL,
|
||||
"NV": a64CondNV,
|
||||
}
|
||||
|
||||
// ---- instruction format tags ----
|
||||
@@ -343,6 +347,7 @@ const (
|
||||
a64FADR // ADR/ADRP
|
||||
a64FEXTR // EXTR
|
||||
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
|
||||
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
|
||||
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
|
||||
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
|
||||
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
|
||||
@@ -487,6 +492,16 @@ func init() {
|
||||
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
|
||||
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
|
||||
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
|
||||
// The widening multiplies: a 64-bit result riding the same layout, the
|
||||
// three-operand forms reading the accumulate register as ZR.
|
||||
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
|
||||
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
|
||||
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
|
||||
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
|
||||
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
|
||||
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
|
||||
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
|
||||
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
|
||||
|
||||
// ---- move wide ----
|
||||
// MOVZ/MOVN/MOVK
|
||||
@@ -536,6 +551,15 @@ func init() {
|
||||
// ---- bitfield ----
|
||||
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
|
||||
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
|
||||
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
|
||||
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
|
||||
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
|
||||
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
|
||||
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
|
||||
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
|
||||
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
|
||||
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
|
||||
@@ -716,6 +740,13 @@ func init() {
|
||||
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
|
||||
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
|
||||
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
|
||||
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
|
||||
// the source width.
|
||||
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
|
||||
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
|
||||
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
|
||||
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
|
||||
"REV16W": 0x5ac00400,
|
||||
}
|
||||
for m, op := range dp1 {
|
||||
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
|
||||
@@ -734,7 +765,7 @@ func init() {
|
||||
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
|
||||
|
||||
// ---- system operations ----
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
|
||||
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
|
||||
a64InstrTable[m] = a64Enc{format: a64FSys}
|
||||
}
|
||||
|
||||
@@ -785,6 +816,73 @@ func init() {
|
||||
for m, op := range lse {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
// The remaining width and ordering spellings of the same shapes, and the
|
||||
// CAS compare-and-swap family, word-verified against go tool asm.
|
||||
lseMore := map[string]uint32{
|
||||
"LDADDAB": 0x38a00000,
|
||||
"LDADDAH": 0x78a00000,
|
||||
"LDADDALB": 0x38e00000,
|
||||
"LDADDALH": 0x78e00000,
|
||||
"LDADDLB": 0x38600000,
|
||||
"LDADDLD": 0xf8600000,
|
||||
"LDADDLH": 0x78600000,
|
||||
"LDADDLW": 0xb8600000,
|
||||
"LDCLRAB": 0x38a01000,
|
||||
"LDCLRAH": 0x78a01000,
|
||||
"LDCLRALH": 0x78e01000,
|
||||
"LDCLRB": 0x38201000,
|
||||
"LDCLRD": 0xf8201000,
|
||||
"LDCLRH": 0x78201000,
|
||||
"LDCLRLB": 0x38601000,
|
||||
"LDCLRLD": 0xf8601000,
|
||||
"LDCLRLH": 0x78601000,
|
||||
"LDCLRLW": 0xb8601000,
|
||||
"LDCLRW": 0xb8201000,
|
||||
"LDEORAB": 0x38a02000,
|
||||
"LDEORAD": 0xf8a02000,
|
||||
"LDEORAH": 0x78a02000,
|
||||
"LDEORALB": 0x38e02000,
|
||||
"LDEORALH": 0x78e02000,
|
||||
"LDEORAW": 0xb8a02000,
|
||||
"LDEORB": 0x38202000,
|
||||
"LDEORD": 0xf8202000,
|
||||
"LDEORH": 0x78202000,
|
||||
"LDEORLB": 0x38602000,
|
||||
"LDEORLD": 0xf8602000,
|
||||
"LDEORLH": 0x78602000,
|
||||
"LDEORLW": 0xb8602000,
|
||||
"LDEORW": 0xb8202000,
|
||||
"LDORAB": 0x38a03000,
|
||||
"LDORAD": 0xf8a03000,
|
||||
"LDORAH": 0x78a03000,
|
||||
"LDORALH": 0x78e03000,
|
||||
"LDORAW": 0xb8a03000,
|
||||
"LDORB": 0x38203000,
|
||||
"LDORD": 0xf8203000,
|
||||
"LDORH": 0x78203000,
|
||||
"LDORLB": 0x38603000,
|
||||
"LDORLD": 0xf8603000,
|
||||
"LDORLH": 0x78603000,
|
||||
"LDORLW": 0xb8603000,
|
||||
"LDORW": 0xb8203000,
|
||||
"SWPAB": 0x38a08000,
|
||||
"SWPAD": 0xf8a08000,
|
||||
"SWPAH": 0x78a08000,
|
||||
"SWPALH": 0x78e08000,
|
||||
"SWPAW": 0xb8a08000,
|
||||
"SWPB": 0x38208000,
|
||||
"SWPH": 0x78208000,
|
||||
"SWPLB": 0x38608000,
|
||||
"SWPLD": 0xf8608000,
|
||||
"SWPLH": 0x78608000,
|
||||
"SWPLW": 0xb8608000,
|
||||
"CASAD": 0xc8e07c00,
|
||||
"CASALB": 0x08e0fc00,
|
||||
"CASLW": 0x88a0fc00,
|
||||
}
|
||||
for m, op := range lseMore {
|
||||
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
|
||||
}
|
||||
|
||||
// ---- carry-setting/carry-using arithmetic and widening multiply ----
|
||||
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
|
||||
@@ -794,6 +892,11 @@ func init() {
|
||||
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
|
||||
"SBC": 0xda000000, "SBCW": 0x5a000000,
|
||||
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
|
||||
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
|
||||
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
|
||||
"NGC": 0xda000000, "NGCW": 0x5a000000,
|
||||
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
|
||||
"NEGSW": 0x6b000000,
|
||||
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
|
||||
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
|
||||
}
|
||||
@@ -835,6 +938,12 @@ func init() {
|
||||
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
|
||||
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
|
||||
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
|
||||
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
|
||||
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
|
||||
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
|
||||
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
|
||||
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
|
||||
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
|
||||
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
|
||||
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
|
||||
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
|
||||
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
|
||||
// accept: H, S and D widths for the pairwise data-processing, H and S for
|
||||
// the across-vector reductions.
|
||||
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
|
||||
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
|
||||
|
||||
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
|
||||
// flag alone, without the size bits: the FP converts, the FP round-to-integral
|
||||
// and pairwise compares among them. Word-verified against go tool asm.
|
||||
var a64SimdQOnly = map[string]bool{
|
||||
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
|
||||
"VFABS": true, "VFNEG": true, "VFSQRT": true,
|
||||
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
|
||||
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
|
||||
"VFMAXV": true, "VFMAXNMV": true,
|
||||
}
|
||||
|
||||
// a64ArrBits carries the fixed bits an arrangement contributes to the
|
||||
// three-same word shape: the element size at bits 23:22 and the 128-bit
|
||||
// flag at bit 30. Bit 29 belongs to the instruction's own base.
|
||||
@@ -928,19 +1054,129 @@ var a64SimdVTable = map[string]a64SimdVSpec{
|
||||
"VZIP1": {0x0e003800, 0x7f, false},
|
||||
"VZIP2": {0x0e007800, 0x7f, false},
|
||||
"VCMEQ": {0x2e208c00, 0x7f, false},
|
||||
"VCMGE": {0x0e203c00, 0x7f, false},
|
||||
"VCMGT": {0x0e203400, 0x7f, false},
|
||||
"VCMHI": {0x2e203400, 0x7f, false},
|
||||
"VCMHS": {0x2e203c00, 0x7f, false},
|
||||
// FP compares take H, S and D arrangements only (the toolchain rejects
|
||||
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
|
||||
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
// FP arithmetic shares the same arrangement restriction.
|
||||
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
// Saturating, halving, polynomial and pairwise arithmetic, the logical
|
||||
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
|
||||
// tool asm.
|
||||
"VBIC": {0x0e601c00, 0x7f, false},
|
||||
"VBIF": {0x2ee01c00, 0x7f, false},
|
||||
"VBIT": {0x6ea01c00, 0x7f, false},
|
||||
"VBSL": {0x6e601c00, 0x7f, false},
|
||||
"VCMTST": {0x0e208c00, 0x7f, false},
|
||||
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
|
||||
"VMLA": {0x4ea09400, 0x7f, false},
|
||||
"VMLS": {0x6ea09400, 0x7f, false},
|
||||
"VORN": {0x4ee01c00, 0x7f, false},
|
||||
"VSHADD": {0x4ea00400, 0x7f, false},
|
||||
"VSRHADD": {0x4ea01400, 0x7f, false},
|
||||
"VUHADD": {0x6ea00400, 0x7f, false},
|
||||
"VURHADD": {0x6ea01400, 0x7f, false},
|
||||
"VSMAX": {0x4ea06400, 0x7f, false},
|
||||
"VSMIN": {0x4ea06c00, 0x7f, false},
|
||||
"VSMAXP": {0x4ea0a400, 0x7f, false},
|
||||
"VSMINP": {0x4ea0ac00, 0x7f, false},
|
||||
"VUMAX": {0x2e206400, 0x7f, false},
|
||||
"VUMIN": {0x2e206c00, 0x7f, false},
|
||||
"VUMAXP": {0x6ea0a400, 0x7f, false},
|
||||
"VUMINP": {0x6ea0ac00, 0x7f, false},
|
||||
"VSQADD": {0x4ea00c00, 0x7f, false},
|
||||
"VUQADD": {0x6ea00c00, 0x7f, false},
|
||||
"VSQSUB": {0x4ea02c00, 0x7f, false},
|
||||
"VUQSUB": {0x6ea02c00, 0x7f, false},
|
||||
"VSSHL": {0x4ee04400, 0x7f, false},
|
||||
"VUSHL": {0x6ee04400, 0x7f, false},
|
||||
"VUZP1": {0x0e001800, 0x7f, false},
|
||||
"VUZP2": {0x4ec05800, 0x7f, false},
|
||||
"VTRN1": {0x4ec02800, 0x7f, false},
|
||||
"VTRN2": {0x4ec06800, 0x7f, false},
|
||||
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
|
||||
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
|
||||
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
|
||||
}
|
||||
|
||||
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
|
||||
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
|
||||
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
|
||||
// combination for them, so they stay out and the encoder rejects the shape.
|
||||
var a64SimdVZero = map[string]uint32{
|
||||
"VCMEQ": 0x0e209800,
|
||||
"VCMGT": 0x0e208800,
|
||||
"VCMGE": 0x2e208800,
|
||||
"VCMLT": 0x0e20a800,
|
||||
"VCMLE": 0x2e209800,
|
||||
// FP compares against (0.0): the register forms above carry the U and op
|
||||
// bits; the zero forms reshape them.
|
||||
"VFCMEQ": 0x0ea0d800,
|
||||
"VFCMGE": 0x2ea0c800,
|
||||
"VFCMGT": 0x0ea0c800,
|
||||
"VFCMLE": 0x2ea0d800,
|
||||
"VFCMLT": 0x0ea0e800,
|
||||
}
|
||||
|
||||
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
|
||||
// (word = base | arrBits | Rn<<5 | Rd). VMOV is served from here too, with
|
||||
// the register pair spelling ORR Vd, Vn, Vm.
|
||||
var a64SimdV2Table = map[string]a64SimdVSpec{
|
||||
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
|
||||
"VREV64": {0x0e200800, 0x3f, false},
|
||||
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
"VUADDLV": {0x2e303800, 0x3f, false},
|
||||
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
|
||||
// Two-register data-processing across one arrangement.
|
||||
"VABS": {0x0e20b800, 0x7f, false},
|
||||
"VNEG": {0x2e20b800, 0x7f, false},
|
||||
"VCLS": {0x0e204800, 0x7f, false},
|
||||
"VCLZ": {0x2e204800, 0x7f, false},
|
||||
"VCNT": {0x0e205800, 0x7f, false},
|
||||
"VNOT": {0x2e205800, 0x7f, false},
|
||||
"VSQABS": {0x0e207800, 0x7f, false},
|
||||
"VSQNEG": {0x2e207800, 0x7f, false},
|
||||
"VRBIT": {0x6e605800, 0x7f, false},
|
||||
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
|
||||
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
|
||||
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
|
||||
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
|
||||
"VFABS": {0x0ea0f800, fpSimdArrs, false},
|
||||
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
|
||||
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
|
||||
"VFRINTN": {0x0e218800, fpSimdArrs, false},
|
||||
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
|
||||
"VFRINTM": {0x0e219800, fpSimdArrs, false},
|
||||
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
|
||||
// Across-vector reductions: the operand arrangement rides as usual and
|
||||
// the destination stays a bare V register.
|
||||
"VADDV": {0x0e31b800, 0x3f, false},
|
||||
"VSMAXV": {0x0e30a800, 0x3f, false},
|
||||
"VSMINV": {0x0e31a800, 0x3f, false},
|
||||
"VUMAXV": {0x2e30a800, 0x3f, false},
|
||||
"VUMINV": {0x2e31a800, 0x3f, false},
|
||||
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
|
||||
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
|
||||
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
|
||||
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
|
||||
}
|
||||
|
||||
// a64CryptoArr is the arrangement each crypto instruction's operands must
|
||||
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
|
||||
// MSR Rn, <sysreg>; the source register rides bits 4:0.
|
||||
var a64MSRRegOps = map[string]uint32{
|
||||
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
|
||||
"ELR_EL1": 0xd5184020,
|
||||
}
|
||||
|
||||
// a64MSROps maps the system register names GOROOT writes to their fixed
|
||||
|
||||
+161
-10
@@ -4,6 +4,7 @@
|
||||
package asm
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
|
||||
@@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
|
||||
// rejected instead of wrapping through int32.
|
||||
func TestArm64AddSubImmRange(t *testing.T) {
|
||||
for _, body := range []string{
|
||||
"\tADD $0x100000000, R0, R1\n",
|
||||
"\tSUB $-0x100000000, R0, R1\n",
|
||||
"\tCMP $0x100000000, R0\n",
|
||||
} {
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
|
||||
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
|
||||
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
|
||||
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
|
||||
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
|
||||
// Comparisons never split, and the W forms classify the 32-bit value. Every
|
||||
// word is go tool asm's own for the same source.
|
||||
func TestArm64AddSubImmWide(t *testing.T) {
|
||||
got := arm64Words(t, strings.Join([]string{
|
||||
"\tADD $0xaaaaaa, R2, R3",
|
||||
"\tSUB $0xaaaaaa, R2",
|
||||
"\tADD $0x186a0, R2, R5",
|
||||
"\tADD $0x1ffe00, R2, R3",
|
||||
"\tADD $0x3fffffffc000, R5",
|
||||
"\tADD $-100000, R2, R3",
|
||||
"\tADD $-2048, R2, R3",
|
||||
"\tCMP $0xaaaaaa, R2",
|
||||
"\tCMP $0xffffffffffa0, R3",
|
||||
"\tCMPW $27745, R2",
|
||||
"\tCMPW $0x60060, R2",
|
||||
"\tADDS $0xaaaaaa, R2, R3",
|
||||
"\tADD $0x12345678, R2, R3",
|
||||
"\tADDW $0x60060, R2",
|
||||
"\tSUB $0xe7791f700, R3, R1",
|
||||
"\tADDW $0x12345678, R2, R3",
|
||||
"\tCMN $0x1000000, R2",
|
||||
}, "\n")+"\n")
|
||||
want := []uint32{
|
||||
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
|
||||
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
|
||||
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
|
||||
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
|
||||
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
|
||||
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
|
||||
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
|
||||
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
|
||||
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
|
||||
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
|
||||
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
|
||||
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
|
||||
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
|
||||
0x11018042, 0x11418042, // ADDW $0x60060: W split
|
||||
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
|
||||
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
|
||||
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
|
||||
// three operands, the ROR shift on the logical group (and its rejection for
|
||||
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
|
||||
// alias with an element selector. Words are go tool asm's own.
|
||||
func TestArm64CarryShiftAlias(t *testing.T) {
|
||||
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
|
||||
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
|
||||
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
|
||||
want := []uint32{
|
||||
0x9a1f0294, // ADC ZR, R20, R20
|
||||
0x9a1f0284, // ADC ZR, R20, R4
|
||||
0xfa1f008c, // SBCS ZR, R4, R12
|
||||
0xfa0f008c, // SBCS R15, R4, R12
|
||||
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
|
||||
0x8ac18443, // AND R1 ROR 33, R2, R3
|
||||
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
|
||||
0xda0203e7, // SBC ZR, R7, R2
|
||||
0x9b0eff77, // MSUB ZR, R27, R14, R23
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
|
||||
// ROR on an arithmetic form is unallocated: the toolchain reports an
|
||||
// unsupported shift operator.
|
||||
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
if _, err := AssembleFileARM64(f); err == nil {
|
||||
t.Errorf("%s: expected an error, got none", body)
|
||||
t.Error("ADD R1@>33: expected an error, got none")
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
|
||||
// operand with an element selector and inside a split register list: the
|
||||
// aliases resolve textually where the parser carries the selector apart from
|
||||
// the name. Words are go tool asm's own.
|
||||
func TestArm64VecAliasElement(t *testing.T) {
|
||||
src := `#include "textflag.h"
|
||||
|
||||
#define POLY V15
|
||||
#define ACC0 V8
|
||||
#define ACC1 V9
|
||||
|
||||
TEXT ·f(SB), NOSPLIT, $0-0
|
||||
VMOV R1, POLY.D[0]
|
||||
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||
VLD1 (R0), [ACC0.B16]
|
||||
VLD1.P (R0), [ACC0.B16, ACC1.B16]
|
||||
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
|
||||
RET
|
||||
`
|
||||
f, errs := parser.Parse("test_arm64.s", src)
|
||||
if len(errs) > 0 {
|
||||
t.Fatalf("parse: %v", errs)
|
||||
}
|
||||
img, err := AssembleFileARM64(f)
|
||||
if err != nil {
|
||||
t.Fatalf("AssembleFileARM64: %v", err)
|
||||
}
|
||||
got := leWords(img.Code)
|
||||
want := []uint32{
|
||||
0x4e081c2f, // INS V15.D[0], R1
|
||||
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
|
||||
0x4c407008, // VLD1 (R0), [V8.B16]
|
||||
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
|
||||
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
|
||||
// once the value leaves every imm12 form: a constant sequence into REGTMP
|
||||
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
|
||||
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
|
||||
// tool asm's own.
|
||||
func TestArm64AddSubImmBeyond32(t *testing.T) {
|
||||
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
|
||||
want := []uint32{
|
||||
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||
0x8b1b0001, // ADD R27, R0, R1
|
||||
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
|
||||
0xcb1b0001, // SUB R27, R0, R1
|
||||
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
|
||||
0xeb1b001f, // CMP R27, R0
|
||||
0xd65f03c0, // RET
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Vendored
+48
@@ -0,0 +1,48 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Carry arithmetic, logical shifts, register aliases with element selectors
|
||||
// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and
|
||||
// gcm_arm64.s exercise. Byte-for-byte against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
#define acc0 V8
|
||||
#define acc1 V9
|
||||
#define const0 R15
|
||||
#define POLY V15
|
||||
|
||||
// carry pins the ADC/SBC family: the $0 spellings in two and three
|
||||
// operands, and the register-carry forms.
|
||||
TEXT ·carry(SB), NOSPLIT, $0-0
|
||||
ADC $0, R20
|
||||
ADC $0, R20, R4
|
||||
SBCS $0, R4
|
||||
SBCS $0, R4, R12
|
||||
SBCS R15, R4, R12
|
||||
SBC $0, R1
|
||||
ADCSW $0, R2, R3
|
||||
RET
|
||||
|
||||
// shift pins the shifted-register forms including ROR, which only the
|
||||
// logical family accepts.
|
||||
TEXT ·shift(SB), NOSPLIT, $0-0
|
||||
ANDW R9@>7, R19, R26
|
||||
AND R1@>33, R2, R3
|
||||
ADD R1<<11, R2, R3
|
||||
SUB R1->33, R2
|
||||
ORR R5<<2, R6, R7
|
||||
RET
|
||||
|
||||
// vecalias pins the vector aliases with element selectors and the
|
||||
// structure loads with aliased members.
|
||||
TEXT ·vecalias(SB), NOSPLIT, $0-0
|
||||
MOVD $0xC2, R1
|
||||
VMOV R1, POLY.D[0]
|
||||
VMOV R0, POLY.D[1]
|
||||
VEOR POLY.B16, POLY.B16, POLY.B16
|
||||
VLD1 (R0), [acc0.B16]
|
||||
VLD1.P (R0), [acc0.B16, acc1.B16]
|
||||
VST1 [acc0.B16, acc1.B16], (R1)
|
||||
VST1.P [acc0.B16, acc1.B16], 32(R1)
|
||||
RET
|
||||
Vendored
+70
@@ -0,0 +1,70 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Wide-immediate arithmetic: every classification band of the ADD/SUB
|
||||
// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/
|
||||
// MOVK materialisations into REGTMP) plus the logical bitmask immediates and
|
||||
// their materialised fallback. Byte-for-byte against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// imm12 covers the plain and shifted-by-12 imm12 forms.
|
||||
TEXT ·imm12(SB), NOSPLIT, $0-0
|
||||
ADD $1, R2, R3
|
||||
ADD $0x000aaa, R2, R3
|
||||
ADD $0xaaa000, R2
|
||||
SUB $0x000aaa, R2, R3
|
||||
SUB $0xaaa000, R2
|
||||
ADDW $40960, R0
|
||||
CMP $40960, R0
|
||||
CMPW $40960, R0
|
||||
RET
|
||||
|
||||
// split pins the ADDCON2 band: two imm12 instructions, low half first.
|
||||
TEXT ·split(SB), NOSPLIT, $0-0
|
||||
ADD $0xaaaaaa, R2, R3
|
||||
SUB $0xaaaaaa, R2
|
||||
ADD $0x186a0, R2, R5
|
||||
SUB $0x186a0, R2, R3
|
||||
ADDW $0x60060, R2
|
||||
RET
|
||||
|
||||
// regtmp covers the single-word materialisations: MOVZ for a movcon value,
|
||||
// MOVN for the complement form, the bitmask ORR otherwise.
|
||||
TEXT ·regtmp(SB), NOSPLIT, $0-0
|
||||
ADD $0x1ffe00, R2, R3
|
||||
ADD $0x3fffffffc000, R5
|
||||
ADD $-2048, R2, R3
|
||||
ADD $-100000, R2, R3
|
||||
CMP $0x1000000, R2
|
||||
CMP $0x100000000, R0
|
||||
SUB $-0x100000000, R0, R1
|
||||
RET
|
||||
|
||||
// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the
|
||||
// compare forms that never split.
|
||||
TEXT ·movseq(SB), NOSPLIT, $0-0
|
||||
ADD $0x12345678, R2, R3
|
||||
SUB $0xe7791f700, R3, R1
|
||||
CMP $0xaaaaaa, R2
|
||||
CMP $0xffffffffffa0, R3
|
||||
CMPW $27745, R2
|
||||
CMPW $0x60060, R2
|
||||
ADDS $0xaaaaaa, R2, R3
|
||||
CMN $0x1000000, R2
|
||||
ADDW $0x12345678, R2, R3
|
||||
RET
|
||||
|
||||
// logical covers the bitmask immediates of the logical family and the
|
||||
// materialised fallback for the values a bitmask cannot carry.
|
||||
TEXT ·logical(SB), NOSPLIT, $0-0
|
||||
AND $0x3ff00000, R2, R3
|
||||
BIC $0x22220000, R3, R4
|
||||
ORR $0x3ff00000, R2
|
||||
EOR $0x3ff00000, R2, R3
|
||||
ANDS $0x3ff00000, R2
|
||||
ORNW $0x3ff00000, R2
|
||||
EONW $0x3ff00000, R2
|
||||
BICSW $0x6006000060060, R5
|
||||
TST $0x4900000049, R0
|
||||
RET
|
||||
Reference in New Issue
Block a user