feat(arm64): wide immediates, SIMD compare and system operand forms

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 14:25:47 +02:00
parent ad82aac663
commit 9b238a525a
5 changed files with 1143 additions and 176 deletions
+569 -108
View File
@@ -5,6 +5,7 @@ package asm
import (
"fmt"
"math/bits"
"strconv"
"strings"
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW":
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
"ADDS", "ADDSW", "SUBS", "SUBSW":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := arm64Imm64(ops[0])
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
if v >= 0 && v <= 0xFFF {
return 4
// Size exactly as the encoder will emit: a single imm12 word, the
// two-word ADDCON2 split, or a materialisation into REGTMP plus
// the register form. Anything else would desynchronise the label
// offsets of pass 1 from the bytes pass 2 lays down.
if v, ok := arm64ImmOperandValue(ops[0]); ok {
rn, rd := 0, 0
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
rd = n
}
if len(ops) == 3 {
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
rn = n
}
}
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
return 4 * len(ws)
}
}
if v >= -2048 && v < 0 {
return 4
}
// Larger immediates need MOV materialisation + op.
return 8
return 4
}
}
return 4
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Bitfield(mnem, enc.op, ops)
}
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
}
// EXTR.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
return encodeARM64Extr(mnem, enc.op, ops)
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below.
if spec, ok := a64SimdVTable[mnem]; ok {
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
// only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
return encodeARM64SimdV(mnem, spec, ops)
}
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
}
// SIMD table lookup.
if mnem == "VTBL" {
return encodeARM64VTBL(ops)
if mnem == "VTBL" || mnem == "VTBX" {
return encodeARM64VTBL(mnem, ops)
}
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
}
op := ops[0]
// Branch to the program counter itself: JMP (PC) spins forever. The
// toolchain encodes it as an unconditional branch with a zero offset.
// Branch to the program counter: JMP (PC) spins forever, and a spelled
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
// units. The toolchain encodes both as a plain branch of that offset.
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
if link {
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
rel := op.Addr.Offset
if rel < -(1<<25) || rel >= (1<<25) {
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
}
return a64wordLE(a64Branch(0, 0)), nil
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, int32(rel))), nil
}
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
// the ADD/SUB-with-flags family an add/sub immediate.
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
mnem == "MVN" || mnem == "MVNW" ||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
var logical bool
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
"TST", "TSTW":
logical = true
}
if logical {
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
}
inverted := false
switch mnem {
case "BIC", "BICW", "BICS", "BICSW":
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
inverted = true
}
if inverted {
@@ -700,7 +727,37 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
}
n, immr, imms, ok := a64LogicalImm(v, width)
if !ok {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
// Beyond the bitmask immediates the toolchain materialises
// the constant into REGTMP (R27) and uses the register form
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
// value, so the materialisation uses v before any inversion.
written := v
if inverted {
written = ^v
}
width := mnem
if strings.HasSuffix(mnem, "W") {
width = "MOVW"
} else {
width = "MOVD"
}
mw, merr := encodeARM64LoadImm(27, written, width)
var rn, rd int
switch len(ops) {
case 3:
rn = arm64RegNum(operandRegName(ops[1]))
rd = arm64RegNum(operandRegName(ops[2]))
default:
rd = arm64RegNum(operandRegName(ops[1]))
rn = rd
}
if isCmp {
rd = 31
}
if merr != nil || rn < 0 || rd < 0 {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
}
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
}
opc := (baseOp >> 29) & 7
sf := (baseOp >> 31) & 1
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
}
if !isExtend {
if amount < 0 || amount > 63 {
// ROR rides the shifted-register field only for the logical
// group; the toolchain reports "unsupported shift operator" for
// the arithmetic forms, whose shift=11 encoding is unallocated.
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
}
// The imm6 field is 5 bits and truncates at the 32-bit width.
limit := 63
if strings.HasSuffix(mnem, "W") {
limit = 31
}
if amount < 0 || amount > limit {
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
}
// SP-based ADD/SUB have no shifted-register encoding: the
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
switch len(ops) {
case 3:
// The carry family carries an immediate spelling in three operands
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
// operand, the same shape the two-operand form takes.
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
if v := arm64Imm64(ops[0]); v != 0 {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// OP Rm, Rn, Rd
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
// shifted-register group, the only forms whose register operand accepts the
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
func arm64LogicalShifted(mnem string) bool {
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
"TST", "TSTW", "MVN", "MVNW":
return true
}
return false
}
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
// data-processing instructions the toolchain accepts an immediate $0
// operand spelling for.
func arm64CarryOp(mnem string) bool {
switch mnem {
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
return true
}
return false
}
// arm64RegMod reports whether a register operand carries the shifted-register
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
// option (R0.UXTW, R3.SXTW<<2).
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
// arm64RegModifier resolves a modified register operand: the register number,
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
// option (UXTB=0..SXTX=7) with its shift amount.
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
// before the operator match.
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
name := operandRegName(op)
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
if before, after, ok0 := strings.Cut(name, "."); ok0 {
switch strings.ToUpper(strings.TrimSpace(after)) {
case "UXTB":
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
if rm < 0 {
return 0, 0, 0, false, 0, false
}
amount, ok = arm64ShiftAmount(op.Addr.Shift)
amount, ok = arm64ShiftAmount(shift)
if !ok || amount < 0 || amount > 4 {
return 0, 0, 0, false, 0, false
}
return rm, 0, extendOpt, true, amount, true
}
shiftKind = 0 // LSL
shift := strings.TrimSpace(op.Addr.Shift)
switch {
case strings.HasPrefix(shift, "<<"):
shiftKind = 0
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
default:
return 0, 0, 0, false, 0, false
}
amount, ok = arm64ShiftAmount(op.Addr.Shift)
amount, ok = arm64ShiftAmount(shift)
if !ok {
return 0, 0, 0, false, 0, false
}
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
// MUL mnemonic, not to these.
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// The widening three-operand forms (SMULL, UMNEGL, …) read the
// accumulate register as ZR, already preset in the table's base word.
if len(ops) == 3 {
switch mnem {
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
default:
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
}
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
}
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// ---- ADD/SUB immediate ----
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
// following the toolchain's immediate classification (asm7.go conclass and
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
// otherwise a constant materialisation into REGTMP (R27) followed by the
// register form.
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := arm64Imm64(ops[0])
v, ok := arm64ImmOperandValue(ops[0])
if !ok {
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
}
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd
if len(ops) == 3 {
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
sf := uint32(1) // 64-bit
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
sf = 0 // 32-bit
}
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
// CMP/CMN discard the destination.
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
rd = 31 // ZR
}
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
return a64WordsLE(ws...), nil
}
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
// their W forms. rn and rd are resolved register numbers (a comparison
// discards rd, so the caller passes 31).
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
w := strings.HasSuffix(mnem, "W")
sf := uint32(1) // 64-bit
d := v
if w {
sf = 0 // 32-bit
// The W forms classify the 32-bit value (asm7.go con32class).
d = int64(uint32(v))
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
op := uint32(0) // ADD
S := uint32(0)
if isSub {
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if isS {
S = 1
}
single := func(sh, imm12 uint32) []uint32 {
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
}
if v >= 0 && v <= 0xFFF {
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
// imm12: plain, then the one-shifted-by-12 form.
if d >= 0 && d <= 0xFFF {
return single(0, uint32(d)), nil
}
if v >= -2048 && v < 0 {
// Encode as the opposite operation with positive immediate.
opp := op ^ 1
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
return single(1, uint32(d>>12)), nil
}
// Try with shift by 12.
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
// split into two imm12 instructions, low half first (asm7.go case 48).
// The encoding is complete in itself: no REGTMP, no register form. The S
// forms must not break addition/subtraction, so the toolchain
// reclassifies them and falls through to the materialisation below.
dm := ^d
if w {
dm = ^d & 0xFFFFFFFF
}
// The imm12 field cannot carry the value; rejecting (rather than
// truncating) matches the toolchain, which reports the same shape.
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
return []uint32{
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
}, nil
}
// Constant into REGTMP (R27), then the register form. The first word
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
// the complement form, the bitmask ORR otherwise, and the full
// omovlconst sequence when no single word carries the value.
var seq []uint32
switch s := arm64Movcon(d); {
case s >= 0:
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
case arm64Movcon(dm) >= 0:
s := arm64Movcon(dm)
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
case isBitcon:
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
default:
seq = arm64MovLConst(d, sf)
}
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
for i := range seq[:len(seq)-1] {
seq[i] |= 27 // REGTMP
}
return seq, nil
}
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
// returned with the destination field clear so the caller can OR its own
// register in. movcon and movcon-of-complement must fail for d before this
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
func arm64MovLConst(d int64, sf uint32) []uint32 {
if sf == 0 {
// omovlconst AMOVW: both 16-bit halves, low first.
return []uint32{
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
}
}
dn := ^d
var immh [4]uint64
zero, neg := 0, 0
for i := range immh {
immh[i] = uint64(d>>(i*16)) & 0xFFFF
switch immh[i] {
case 0:
zero++
case 0xFFFF:
neg++
}
}
mw := func(opc uint32, val int64, chunk int) uint32 {
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
}
var os []uint32
switch {
case zero == 2:
// one MOVZ and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg == 2:
// one MOVN and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
// a single MOVK (constants from strength-reduced division).
if zero == 0 && neg == 0 {
for i := range 4 {
mask := uint64(0xFFFF) << (i * 16)
for period := 2; period <= 32; period *= 2 {
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
os = append(os, mw(3, d, i))
return os
}
}
}
}
switch {
case zero >= 1:
// one MOVZ and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg >= 1:
// one MOVN and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// one MOVZ and three MOVKs
os = append(os, mw(2, d, 0))
for i := 1; i < 4; i++ {
os = append(os, mw(3, d, i))
}
}
}
return os
}
// ---- MOV pseudo-instruction ----
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
}
}
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
var ws []uint32
first := true
for i := range 4 {
chunk := (d >> uint(i*16)) & 0xFFFF
if chunk == 0 {
continue
}
if first {
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
first = false
} else {
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
}
}
if len(ws) == 0 {
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
// for the first special 16-bit chunk, then MOVK per remaining one, with
// the bitmask-plus-fixup shortcut for strength-reduced constants).
ws := arm64MovLConst(d, sf)
for i := range ws {
ws[i] |= uint32(rd)
}
return a64WordsLE(ws...), nil
}
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
// with imms = lsb+width-1.
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
}
lsb := arm64Imm64(ops[0])
width := arm64Imm64(ops[2])
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
bits := int64(32) << (baseOp >> 31 & 1)
if lsb < 0 || lsb >= bits {
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
}
if width < 1 || width > bits || lsb+width > bits {
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
}
var immr, imms int64
switch mnem {
case "BFXIL", "BFXILW":
immr, imms = lsb, lsb+width-1
default: // BFI, SBFIZ, UBFIZ
immr, imms = (-lsb)%bits, width-1
}
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
// third field carries Rm or a 5-bit immediate in the same bits, at the
// toolchain's choice of register or immediate operand.
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op>
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(w), nil
}
switch mnem {
case "BRK", "SVC":
base := uint32(0xd4200000)
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
return a64wordLE(base | uint32(v)<<5), nil
case "DMB", "DSB", "ISB":
case "DMB", "DSB", "ISB", "CLREX":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if v < 0 || v > 15 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
}
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 127 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
}
return a64wordLE(0xd503201f | uint32(v)<<5), nil
case "BTI":
op := operandRegName(ops[0])
base, ok := map[string]uint32{"C": 0xd503245f}[op]
if !ok {
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
}
return a64wordLE(base), nil
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 0xFFFF {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
"DCPS3": 0xd4a00003}[mnem]
return a64wordLE(base | uint32(v)<<5), nil
case "DC":
if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn")
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
return a64wordLE(base | uint32(rd)&31), nil
case "MSR":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(base | uint32(rs)&31), nil
}
base, ok := a64MSROps[operandRegName(ops[1])]
if !ok {
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
func specBit(i int) uint16 { return 1 << uint(i) }
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
return true
}
if !strings.HasPrefix(mnem, "VFCM") {
return false
}
s := strings.Join(strings.Fields(op.Raw), "")
s = strings.TrimPrefix(s, "$")
s = strings.Trim(s, "()")
return s == "0" || s == "0.0"
}
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
// zero immediate takes its compare-against-zero form instead, and the
// polynomial multiplies read the arrangement from their source operands
// alone, the result spelling (H8, Q1) riding no encoding bits.
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
// their compare-against-zero form instead, and the polynomial multiplies read
// the arrangement from their source operands alone, the result spelling
// (H8, Q1) riding no encoding bits.
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
if mnem == "VPMULL" || mnem == "VPMULL2" {
if len(ops) != 3 {
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
rd, _ := arm64VecOf(ops[2])
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
}
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
if arm64Imm64(ops[0]) != 0 {
if len(ops) == 3 && isImmOperand(ops[0]) {
base, ok := a64SimdVZero[mnem]
if !ok {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
if !arm64SimdZeroImm(mnem, ops[0]) {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
vn, ok1 := arm64VecOf(ops[1])
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil {
return nil, err
}
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if spec.fixed {
arrBits = 0
}
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
}
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil {
return nil, err
}
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
}
// VUADDLV spells its arrangement on the source alone; the rest take it
// on both.
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil {
return nil, err
}
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
// index register rides bits 19:16, the first table register bits 9:5, the
// destination bits 4:0 and the table length (registers minus one) bits
// 14:13. The table registers must be consecutive.
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) < 3 {
return nil, fmt.Errorf("VTBL expects index, table list and destination")
}
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
default:
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
}
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
base := uint32(0x0e000000)
if mnem == "VTBX" {
base |= 1 << 12
}
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
}
var immval int64
switch mnem {
case "VSHL":
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
if sh < 0 || sh >= esize {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
}
immval = esize + sh
default: // VUSHR, VSRI
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
if sh < 1 || sh > esize {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
}
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
sym *ast.Symbol // the parsed frame-relative reference (when mem)
}
aliases := map[string]alias{}
raws := map[string]string{}
for _, d := range f.Decls {
pre, ok := d.(*ast.Preproc)
if !ok {
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
continue
}
raws[name] = body
}
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
// transitively until nothing changes, bounded against cycles.
for range 8 {
changed := false
for name, body := range raws {
if next, ok := raws[body]; ok && next != body {
raws[name] = next
changed = true
}
}
if !changed {
break
}
}
for name, body := range raws {
isReg := func(s string) bool {
if arm64RegNum(s) >= 0 {
return true
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
return
}
// replace rewrites whole-word occurrences of the alias names in s.
replace := func(s string) string {
if s == "" {
return s
}
out := strings.Fields(s)
for i, w := range out {
if a, ok := aliases[w]; ok {
out[i] = a.raw
}
}
if len(out) == 0 {
return s
}
return strings.Join(out, " ")
}
// replaceToken rewrites an operand whose whole text is one alias use
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
// ending at a non-identifier character.
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
return s, false
}
// replaceScan rewrites alias uses inside a composite operand (a
// parenthesised memory operand or a bracketed register list): every
// identifier run of word and dot characters is matched against the alias
// names, everything else copies verbatim. The whitespace-split replace
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
// punctuation attached.
replaceScan := func(s string) string {
var b strings.Builder
for i := 0; i < len(s); {
if isAliasWordByte(s[i]) || s[i] == '.' {
j := i
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
j++
}
if nn, ok := replaceToken(s[i:j]); ok {
b.WriteString(nn)
} else {
b.WriteString(s[i:j])
}
i = j
continue
}
b.WriteByte(s[i])
i++
}
return b.String()
}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
op.Raw = a.raw
continue
}
op.Addr.Sym.Name = nn
op.Addr.Sym.Raw = nn
op.Raw = nn
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
// The span shape depends on what trailed the
// name: an element or arrangement selector
// (POLY.D[0], POLY.B16) rides in Shift and folds
// back onto the rewritten token; a shift
// operator stays in Shift while the span carries
// the bare register; a split list keeps its
// closing bracket, so the rewrite goes through
// the scan.
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
switch {
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
// Element or arrangement selectors and the
// closing bracket of a split list belong to
// the token text.
op.Raw = nn + sfx
op.Addr.Shift = ""
case op.Addr.Shift != "":
op.Raw = nn
default:
op.Raw = replaceScan(op.Raw)
}
continue
}
}
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
// [V0.B16, V1.B16] with aliased members.
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
op.Raw = replace(op.Raw)
op.Raw = replaceScan(op.Raw)
}
}
}
+292 -55
View File
@@ -308,6 +308,8 @@ const (
a64CondLT = 0xb
a64CondGT = 0xc
a64CondLE = 0xd
a64CondAL = 0xe
a64CondNV = 0xf
)
// arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes.
@@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{
"LT": a64CondLT,
"GT": a64CondGT,
"LE": a64CondLE,
"AL": a64CondAL,
"NV": a64CondNV,
}
// ---- instruction format tags ----
@@ -335,46 +339,47 @@ var arm64CondMap = map[string]uint32{
type a64Format uint8
const (
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc.
a64FMovWide // move wide: MOVZ, MOVN, MOVK
a64FBranch // unconditional branch (B/BL)
a64FBranchCond // conditional branch (B.cond)
a64FUncondBranch // unconditional branch register (BR/BLR/RET)
a64FADR // ADR/ADRP
a64FEXTR // EXTR
a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM
a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd)
a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source
a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10
a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc.
a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT*
a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc.
a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE
a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE
a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc.
a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL
a64FCRC32 // CRC32
a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG
a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP
a64FLSE // LSE atomics: LDADD, CAS, SWP
a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS
a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms
a64FCondCmp // conditional compare: CCMP, CCMN
a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms
a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms
a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD
a64FAcqRel // acquire/release: LDAR family, STLR family
a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM
a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ...
a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ...
a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ...
a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd
a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV
a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT
a64FVTBL // SIMD table lookup: VTBL
a64FDUP // SIMD element moves: VDUP, VMOV with element indices
a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R
a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI
a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool)
)
// a64Enc is one instruction's encoding: its bit layout (format) and the
@@ -487,6 +492,16 @@ func init() {
a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24}
a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15}
a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15}
// The widening multiplies: a 64-bit result riding the same layout, the
// three-operand forms reading the accumulate register as ZR.
a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21}
a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23}
a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15}
a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15}
a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10}
a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10}
a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10}
a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10}
// ---- move wide ----
// MOVZ/MOVN/MOVK
@@ -536,6 +551,15 @@ func init() {
// ---- bitfield ----
a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22}
// The four-operand bitfield aliases: ($lsb, Rn, $width, Rd).
a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22}
a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23}
a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000}
a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000}
a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000}
a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000}
a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22}
a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22}
a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22}
@@ -716,6 +740,13 @@ func init() {
"RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800,
"REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400,
"RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400,
// Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing
// the source width.
"SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00,
"SXTHW": 0x13003c00, "SXTW": 0x93407c00,
"UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00,
"UXTHW": 0x53003c00, "UXTW": 0x53407c00,
"REV16W": 0x5ac00400,
}
for m, op := range dp1 {
a64InstrTable[m] = a64Enc{format: a64FDP1, op: op}
@@ -734,7 +765,7 @@ func init() {
a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000}
// ---- system operations ----
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} {
for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} {
a64InstrTable[m] = a64Enc{format: a64FSys}
}
@@ -785,6 +816,73 @@ func init() {
for m, op := range lse {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// The remaining width and ordering spellings of the same shapes, and the
// CAS compare-and-swap family, word-verified against go tool asm.
lseMore := map[string]uint32{
"LDADDAB": 0x38a00000,
"LDADDAH": 0x78a00000,
"LDADDALB": 0x38e00000,
"LDADDALH": 0x78e00000,
"LDADDLB": 0x38600000,
"LDADDLD": 0xf8600000,
"LDADDLH": 0x78600000,
"LDADDLW": 0xb8600000,
"LDCLRAB": 0x38a01000,
"LDCLRAH": 0x78a01000,
"LDCLRALH": 0x78e01000,
"LDCLRB": 0x38201000,
"LDCLRD": 0xf8201000,
"LDCLRH": 0x78201000,
"LDCLRLB": 0x38601000,
"LDCLRLD": 0xf8601000,
"LDCLRLH": 0x78601000,
"LDCLRLW": 0xb8601000,
"LDCLRW": 0xb8201000,
"LDEORAB": 0x38a02000,
"LDEORAD": 0xf8a02000,
"LDEORAH": 0x78a02000,
"LDEORALB": 0x38e02000,
"LDEORALH": 0x78e02000,
"LDEORAW": 0xb8a02000,
"LDEORB": 0x38202000,
"LDEORD": 0xf8202000,
"LDEORH": 0x78202000,
"LDEORLB": 0x38602000,
"LDEORLD": 0xf8602000,
"LDEORLH": 0x78602000,
"LDEORLW": 0xb8602000,
"LDEORW": 0xb8202000,
"LDORAB": 0x38a03000,
"LDORAD": 0xf8a03000,
"LDORAH": 0x78a03000,
"LDORALH": 0x78e03000,
"LDORAW": 0xb8a03000,
"LDORB": 0x38203000,
"LDORD": 0xf8203000,
"LDORH": 0x78203000,
"LDORLB": 0x38603000,
"LDORLD": 0xf8603000,
"LDORLH": 0x78603000,
"LDORLW": 0xb8603000,
"LDORW": 0xb8203000,
"SWPAB": 0x38a08000,
"SWPAD": 0xf8a08000,
"SWPAH": 0x78a08000,
"SWPALH": 0x78e08000,
"SWPAW": 0xb8a08000,
"SWPB": 0x38208000,
"SWPH": 0x78208000,
"SWPLB": 0x38608000,
"SWPLD": 0xf8608000,
"SWPLH": 0x78608000,
"SWPLW": 0xb8608000,
"CASAD": 0xc8e07c00,
"CASALB": 0x08e0fc00,
"CASLW": 0x88a0fc00,
}
for m, op := range lseMore {
a64InstrTable[m] = a64Enc{format: a64FLSE, op: op}
}
// ---- carry-setting/carry-using arithmetic and widening multiply ----
// MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate
@@ -794,7 +892,12 @@ func init() {
"ADCS": 0xba000000, "ADCSW": 0x3a000000,
"SBC": 0xda000000, "SBCW": 0x5a000000,
"SBCS": 0xfa000000, "SBCSW": 0x7a000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
// MNEG/MSUB and NGC/SBC with the complementing register preset to ZR.
"MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00,
"NGC": 0xda000000, "NGCW": 0x5a000000,
"NGCS": 0xfa000000, "NGCSW": 0x7a000000,
"NEGSW": 0x6b000000,
"MUL": 0x9b007c00, "MULW": 0x1b007c00,
"SMULH": 0x9b407c00, "UMULH": 0x9bc07c00,
}
for m, op := range dpsrExtra {
@@ -835,6 +938,12 @@ func init() {
a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10}
a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10}
a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10}
a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10}
a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10}
a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10}
a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10}
a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10}
a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10}
a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST}
a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1}
a64InstrTable["VST1"] = a64Enc{format: a64FVLDST}
@@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool {
return false
}
// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms
// accept: H, S and D widths for the pairwise data-processing, H and S for
// the across-vector reductions.
var fpSimdArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D)
var fpAcrossArrs = uint16(1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S)
// a64SimdQOnly names the forms whose arrangement contributes the 128-bit
// flag alone, without the size bits: the FP converts, the FP round-to-integral
// and pairwise compares among them. Word-verified against go tool asm.
var a64SimdQOnly = map[string]bool{
"VSCVTF": true, "VUCVTF": true, "VFCVTZS": true, "VFCVTZU": true,
"VFABS": true, "VFNEG": true, "VFSQRT": true,
"VFRINTN": true, "VFRINTP": true, "VFRINTM": true, "VFRINTZ": true,
"VFADDP": true, "VFMAXP": true, "VFMAXNMP": true,
"VFMAXV": true, "VFMAXNMV": true,
}
// a64ArrBits carries the fixed bits an arrangement contributes to the
// three-same word shape: the element size at bits 23:22 and the 128-bit
// flag at bit 30. Bit 29 belongs to the instruction's own base.
@@ -918,19 +1044,96 @@ var a64ArrBits = [a64ArrCount]uint32{
// instructions (word = base | arrBits | Rm<<16 | Rn<<5 | Rd). Every base
// word and arrangement bit was read off go tool asm.
var a64SimdVTable = map[string]a64SimdVSpec{
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
"VADD": {0x0e208400, 0x7f, false},
"VSUB": {0x2e208400, 0x7f, false},
"VMUL": {0x0e209c00, 0x3f, false}, // no 2D: integer multiply stops at 4S
"VAND": {0x0e201c00, 0x03, false}, // logical ops accept 8B and 16B only
"VEOR": {0x2e201c00, 0x03, false},
"VORR": {0x0ea01c00, 0x03, false},
"VADDP": {0x0e20bc00, 0x7f, false},
"VZIP1": {0x0e003800, 0x7f, false},
"VZIP2": {0x0e007800, 0x7f, false},
"VCMEQ": {0x2e208c00, 0x7f, false},
"VCMGE": {0x0e203c00, 0x7f, false},
"VCMGT": {0x0e203400, 0x7f, false},
"VCMHI": {0x2e203400, 0x7f, false},
"VCMHS": {0x2e203c00, 0x7f, false},
// FP compares take H, S and D arrangements only (the toolchain rejects
// the byte forms), and VFCMLE/VFCMLT have no register form at all.
"VFCMEQ": {0x0e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGE": {0x2e20e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFCMGT": {0x2ea0e400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// FP arithmetic shares the same arrangement restriction.
"VFADD": {0x0e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFSUB": {0x0ea0d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMUL": {0x2e20dc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFDIV": {0x2e20fc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAX": {0x0e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMIN": {0x0ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNM": {0x0e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNM": {0x0ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLA": {0x0e20cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMLS": {0x0ea0cc00, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
// Saturating, halving, polynomial and pairwise arithmetic, the logical
// VBIT/VBSL family and the FP pairwise forms: word-verified against go
// tool asm.
"VBIC": {0x0e601c00, 0x7f, false},
"VBIF": {0x2ee01c00, 0x7f, false},
"VBIT": {0x6ea01c00, 0x7f, false},
"VBSL": {0x6e601c00, 0x7f, false},
"VCMTST": {0x0e208c00, 0x7f, false},
"VFADDP": {0x2e20d400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXP": {0x2e20f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINP": {0x6ea0f400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMAXNMP": {0x2e20c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VFMINNMP": {0x6ea0c400, 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D, false},
"VMLA": {0x4ea09400, 0x7f, false},
"VMLS": {0x6ea09400, 0x7f, false},
"VORN": {0x4ee01c00, 0x7f, false},
"VSHADD": {0x4ea00400, 0x7f, false},
"VSRHADD": {0x4ea01400, 0x7f, false},
"VUHADD": {0x6ea00400, 0x7f, false},
"VURHADD": {0x6ea01400, 0x7f, false},
"VSMAX": {0x4ea06400, 0x7f, false},
"VSMIN": {0x4ea06c00, 0x7f, false},
"VSMAXP": {0x4ea0a400, 0x7f, false},
"VSMINP": {0x4ea0ac00, 0x7f, false},
"VUMAX": {0x2e206400, 0x7f, false},
"VUMIN": {0x2e206c00, 0x7f, false},
"VUMAXP": {0x6ea0a400, 0x7f, false},
"VUMINP": {0x6ea0ac00, 0x7f, false},
"VSQADD": {0x4ea00c00, 0x7f, false},
"VUQADD": {0x6ea00c00, 0x7f, false},
"VSQSUB": {0x4ea02c00, 0x7f, false},
"VUQSUB": {0x6ea02c00, 0x7f, false},
"VSSHL": {0x4ee04400, 0x7f, false},
"VUSHL": {0x6ee04400, 0x7f, false},
"VUZP1": {0x0e001800, 0x7f, false},
"VUZP2": {0x4ec05800, 0x7f, false},
"VTRN1": {0x4ec02800, 0x7f, false},
"VTRN2": {0x4ec06800, 0x7f, false},
"VRAX1": {0xce608c00, 1 << a64Arr2D, true}, // SHA3 group, D2 only
"VPMULL": {0x0e20e000, 1<<a64Arr8B | 1<<a64ArrD1, false},
"VPMULL2": {0x0e20e000, 1<<a64Arr16B | 1<<a64Arr2D, false},
}
// a64SimdVZero holds the compare-against-zero words of the SIMD compares
// spelled with a $0 first operand (word = base | arrBits | Rn<<5 | Rd).
// VCMHI and VCMHS have no zero form: the toolchain reports an illegal
// combination for them, so they stay out and the encoder rejects the shape.
var a64SimdVZero = map[string]uint32{
"VCMEQ": 0x0e209800,
"VCMGT": 0x0e208800,
"VCMGE": 0x2e208800,
"VCMLT": 0x0e20a800,
"VCMLE": 0x2e209800,
// FP compares against (0.0): the register forms above carry the U and op
// bits; the zero forms reshape them.
"VFCMEQ": 0x0ea0d800,
"VFCMGE": 0x2ea0c800,
"VFCMGT": 0x0ea0c800,
"VFCMLE": 0x2ea0d800,
"VFCMLT": 0x0ea0e800,
}
// a64SimdV2Table holds the arrangement-aware two-register SIMD instructions
@@ -939,8 +1142,41 @@ var a64SimdVTable = map[string]a64SimdVSpec{
var a64SimdV2Table = map[string]a64SimdVSpec{
"VREV32": {0x2e200800, 1<<a64Arr8B | 1<<a64Arr16B | 1<<a64Arr4H | 1<<a64Arr8H, false},
"VREV64": {0x0e200800, 0x3f, false},
"VREV16": {0x0e201800, 1<<a64Arr8B | 1<<a64Arr16B, false},
"VUADDLV": {0x2e303800, 0x3f, false},
"VMOV": {0x0ea01c00, 1<<a64Arr8B | 1<<a64Arr16B, false},
// Two-register data-processing across one arrangement.
"VABS": {0x0e20b800, 0x7f, false},
"VNEG": {0x2e20b800, 0x7f, false},
"VCLS": {0x0e204800, 0x7f, false},
"VCLZ": {0x2e204800, 0x7f, false},
"VCNT": {0x0e205800, 0x7f, false},
"VNOT": {0x2e205800, 0x7f, false},
"VSQABS": {0x0e207800, 0x7f, false},
"VSQNEG": {0x2e207800, 0x7f, false},
"VRBIT": {0x6e605800, 0x7f, false},
"VSCVTF": {0x4e21d800, fpSimdArrs, false},
"VUCVTF": {0x6e21d800, fpSimdArrs, false},
"VFCVTZS": {0x4ea1b800, fpSimdArrs, false},
"VFCVTZU": {0x6ea1b800, fpSimdArrs, false},
"VFABS": {0x0ea0f800, fpSimdArrs, false},
"VFNEG": {0x2ea0f800, fpSimdArrs, false},
"VFSQRT": {0x2ea1f800, fpSimdArrs, false},
"VFRINTN": {0x0e218800, fpSimdArrs, false},
"VFRINTP": {0x0ea18800, fpSimdArrs, false},
"VFRINTM": {0x0e219800, fpSimdArrs, false},
"VFRINTZ": {0x0ea19800, fpSimdArrs, false},
// Across-vector reductions: the operand arrangement rides as usual and
// the destination stays a bare V register.
"VADDV": {0x0e31b800, 0x3f, false},
"VSMAXV": {0x0e30a800, 0x3f, false},
"VSMINV": {0x0e31a800, 0x3f, false},
"VUMAXV": {0x2e30a800, 0x3f, false},
"VUMINV": {0x2e31a800, 0x3f, false},
"VFMAXV": {0x2e30f800, fpAcrossArrs, false},
"VFMINV": {0x2eb0f800, fpAcrossArrs, false},
"VFMAXNMV": {0x2e30c800, fpAcrossArrs, false},
"VFMINNMV": {0x2eb0c800, fpAcrossArrs, false},
}
// a64CryptoArr is the arrangement each crypto instruction's operands must
@@ -976,6 +1212,7 @@ var a64MRSOps = map[string]uint32{
// MSR Rn, <sysreg>; the source register rides bits 4:0.
var a64MSRRegOps = map[string]uint32{
"NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420,
"ELR_EL1": 0xd5184020,
}
// a64MSROps maps the system register names GOROOT writes to their fixed
+164 -13
View File
@@ -4,6 +4,7 @@
package asm
import (
"strings"
"testing"
"sourcedock.dev/petrbalvin/gasm-devkit/ast"
@@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) {
}
}
// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are
// rejected instead of wrapping through int32.
func TestArm64AddSubImmRange(t *testing.T) {
for _, body := range []string{
"\tADD $0x100000000, R0, R1\n",
"\tSUB $-0x100000000, R0, R1\n",
"\tCMP $0x100000000, R0\n",
} {
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain
// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split
// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP,
// and the MOVZ/MOVN/MOVK materialisations followed by the register form.
// Comparisons never split, and the W forms classify the 32-bit value. Every
// word is go tool asm's own for the same source.
func TestArm64AddSubImmWide(t *testing.T) {
got := arm64Words(t, strings.Join([]string{
"\tADD $0xaaaaaa, R2, R3",
"\tSUB $0xaaaaaa, R2",
"\tADD $0x186a0, R2, R5",
"\tADD $0x1ffe00, R2, R3",
"\tADD $0x3fffffffc000, R5",
"\tADD $-100000, R2, R3",
"\tADD $-2048, R2, R3",
"\tCMP $0xaaaaaa, R2",
"\tCMP $0xffffffffffa0, R3",
"\tCMPW $27745, R2",
"\tCMPW $0x60060, R2",
"\tADDS $0xaaaaaa, R2, R3",
"\tADD $0x12345678, R2, R3",
"\tADDW $0x60060, R2",
"\tSUB $0xe7791f700, R3, R1",
"\tADDW $0x12345678, R2, R3",
"\tCMN $0x1000000, R2",
}, "\n")+"\n")
want := []uint32{
0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split
0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn
0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split
0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split
0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP
0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK
0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD
0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK
0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup
0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW
0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split
0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS
0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK
0x11018042, 0x11418042, // ADDW $0x60060: W split
0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700
0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW
0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i])
}
if _, err := AssembleFileARM64(f); err == nil {
t.Errorf("%s: expected an error, got none", body)
}
}
// TestArm64CarryImmWide pins the carry family's $0 spellings in two and
// three operands, the ROR shift on the logical group (and its rejection for
// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector
// alias with an element selector. Words are go tool asm's own.
func TestArm64CarryShiftAlias(t *testing.T) {
got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+
"\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+
"\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n")
want := []uint32{
0x9a1f0294, // ADC ZR, R20, R20
0x9a1f0284, // ADC ZR, R20, R4
0xfa1f008c, // SBCS ZR, R4, R12
0xfa0f008c, // SBCS R15, R4, R12
0x0ac91e7a, // ANDW R9 ROR 7, R19, R26
0x8ac18443, // AND R1 ROR 33, R2, R3
0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1
0xda0203e7, // SBC ZR, R7, R2
0x9b0eff77, // MSUB ZR, R27, R14, R23
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i])
}
}
// ROR on an arithmetic form is unallocated: the toolchain reports an
// unsupported shift operator.
f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n")
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
if _, err := AssembleFileARM64(f); err == nil {
t.Error("ADD R1@>33: expected an error, got none")
}
}
// TestArm64VecAliasElement pins the register-alias rewrite inside a vector
// operand with an element selector and inside a split register list: the
// aliases resolve textually where the parser carries the selector apart from
// the name. Words are go tool asm's own.
func TestArm64VecAliasElement(t *testing.T) {
src := `#include "textflag.h"
#define POLY V15
#define ACC0 V8
#define ACC1 V9
TEXT ·f(SB), NOSPLIT, $0-0
VMOV R1, POLY.D[0]
VEOR POLY.B16, POLY.B16, POLY.B16
VLD1 (R0), [ACC0.B16]
VLD1.P (R0), [ACC0.B16, ACC1.B16]
VST1.P [ACC0.B16, ACC1.B16], 32(R1)
RET
`
f, errs := parser.Parse("test_arm64.s", src)
if len(errs) > 0 {
t.Fatalf("parse: %v", errs)
}
img, err := AssembleFileARM64(f)
if err != nil {
t.Fatalf("AssembleFileARM64: %v", err)
}
got := leWords(img.Code)
want := []uint32{
0x4e081c2f, // INS V15.D[0], R1
0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16
0x4c407008, // VLD1 (R0), [V8.B16]
0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16]
0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1)
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies
// once the value leaves every imm12 form: a constant sequence into REGTMP
// (R27) followed by the register form. SUB $-0x100000000 is a bitmask
// immediate, so it rides the ORR form; the others take MOVZ. Words are go
// tool asm's own.
func TestArm64AddSubImmBeyond32(t *testing.T) {
got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n")
want := []uint32{
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0x8b1b0001, // ADD R27, R0, R1
0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask)
0xcb1b0001, // SUB R27, R0, R1
0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2)
0xeb1b001f, // CMP R27, R0
0xd65f03c0, // RET
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}