feat(arm64): wide immediates, SIMD compare and system operand forms

Assisted-by: GLM 5.3 Flash
This commit is contained in:
2026-09-20 14:25:47 +02:00
parent ad82aac663
commit 9b238a525a
5 changed files with 1143 additions and 176 deletions
+569 -108
View File
@@ -5,6 +5,7 @@ package asm
import (
"fmt"
"math/bits"
"strconv"
"strings"
@@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU",
"FMOVS", "FMOVD":
return arm64MovSize(mnem, ops, fi)
case "ADD", "ADDW", "SUB", "SUBW":
case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW",
"ADDS", "ADDSW", "SUBS", "SUBSW":
if len(ops) >= 2 && isImmOperand(ops[0]) {
v := arm64Imm64(ops[0])
// Small immediate (0..4095 or -2048..-1) fits in one instruction.
if v >= 0 && v <= 0xFFF {
return 4
// Size exactly as the encoder will emit: a single imm12 word, the
// two-word ADDCON2 split, or a materialisation into REGTMP plus
// the register form. Anything else would desynchronise the label
// offsets of pass 1 from the bytes pass 2 lays down.
if v, ok := arm64ImmOperandValue(ops[0]); ok {
rn, rd := 0, 0
if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 {
rd = n
}
if len(ops) == 3 {
if n := arm64RegNum(operandRegName(ops[1])); n >= 0 {
rn = n
}
}
if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil {
return 4 * len(ws)
}
}
if v >= -2048 && v < 0 {
return 4
}
// Larger immediates need MOV materialisation + op.
return 8
return 4
}
}
return 4
@@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return encodeARM64Bitfield(mnem, enc.op, ops)
}
// Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias {
return encodeARM64BitfieldAlias(mnem, enc.op, ops)
}
// EXTR.
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR {
return encodeARM64Extr(mnem, enc.op, ops)
@@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
// Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1,
// VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so
// this check precedes the plain SIMD3 path below.
if spec, ok := a64SimdVTable[mnem]; ok {
// this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist
// only in the zero-immediate form (a64SimdVZero), so they route here with
// an empty register-form spec.
if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 {
return encodeARM64SimdV(mnem, spec, ops)
}
@@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
}
// SIMD table lookup.
if mnem == "VTBL" {
return encodeARM64VTBL(ops)
if mnem == "VTBL" || mnem == "VTBX" {
return encodeARM64VTBL(mnem, ops)
}
// SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R,
@@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
}
op := ops[0]
// Branch to the program counter itself: JMP (PC) spins forever. The
// toolchain encodes it as an unconditional branch with a zero offset.
// Branch to the program counter: JMP (PC) spins forever, and a spelled
// offset (CALL -1(PC), the return stub) rides the imm26 field in word
// units. The toolchain encodes both as a plain branch of that offset.
if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") {
if link {
return nil, fmt.Errorf("%s: branch to PC is not a call", mnem)
rel := op.Addr.Offset
if rel < -(1<<25) || rel >= (1<<25) {
return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel)
}
return a64wordLE(a64Branch(0, 0)), nil
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, int32(rel))), nil
}
// Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The
@@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in
// the ADD/SUB-with-flags family an add/sub immediate.
func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW"
isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" ||
mnem == "MVN" || mnem == "MVNW" ||
mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW"
// Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating
// bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the
@@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
var logical bool
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW",
"BIC", "BICW", "BICS", "BICSW", "TST", "TSTW":
"BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW",
"TST", "TSTW":
logical = true
}
if logical {
@@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
}
inverted := false
switch mnem {
case "BIC", "BICW", "BICS", "BICSW":
case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW":
inverted = true
}
if inverted {
@@ -700,7 +727,37 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
}
n, immr, imms, ok := a64LogicalImm(v, width)
if !ok {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
// Beyond the bitmask immediates the toolchain materialises
// the constant into REGTMP (R27) and uses the register form
// (asm7.go cases 62 and 13). BIC/ORN/EON read the written
// value, so the materialisation uses v before any inversion.
written := v
if inverted {
written = ^v
}
width := mnem
if strings.HasSuffix(mnem, "W") {
width = "MOVW"
} else {
width = "MOVD"
}
mw, merr := encodeARM64LoadImm(27, written, width)
var rn, rd int
switch len(ops) {
case 3:
rn = arm64RegNum(operandRegName(ops[1]))
rd = arm64RegNum(operandRegName(ops[2]))
default:
rd = arm64RegNum(operandRegName(ops[1]))
rn = rd
}
if isCmp {
rd = 31
}
if merr != nil || rn < 0 || rd < 0 {
return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " "))
}
return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil
}
opc := (baseOp >> 29) & 7
sf := (baseOp >> 31) & 1
@@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem)
}
if !isExtend {
if amount < 0 || amount > 63 {
// ROR rides the shifted-register field only for the logical
// group; the toolchain reports "unsupported shift operator" for
// the arithmetic forms, whose shift=11 encoding is unallocated.
if shiftBits == 3 && !arm64LogicalShifted(mnem) {
return nil, fmt.Errorf("%s: unsupported shift operator", mnem)
}
// The imm6 field is 5 bits and truncates at the 32-bit width.
limit := 63
if strings.HasSuffix(mnem, "W") {
limit = 31
}
if amount < 0 || amount > limit {
return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount)
}
// SP-based ADD/SUB have no shifted-register encoding: the
@@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
switch len(ops) {
case 3:
// The carry family carries an immediate spelling in three operands
// too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register
// operand, the same shape the two-operand form takes.
if isImmOperand(ops[0]) && arm64CarryOp(mnem) {
if v := arm64Imm64(ops[0]); v != 0 {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
// OP Rm, Rn, Rd
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
@@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
// arm64LogicalShifted reports whether a mnemonic belongs to the logical
// shifted-register group, the only forms whose register operand accepts the
// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms).
func arm64LogicalShifted(mnem string) bool {
switch mnem {
case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW",
"ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW",
"TST", "TSTW", "MVN", "MVNW":
return true
}
return false
}
// arm64CarryOp reports whether a mnemonic belongs to the carry-using
// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only
// data-processing instructions the toolchain accepts an immediate $0
// operand spelling for.
func arm64CarryOp(mnem string) bool {
switch mnem {
case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW":
return true
}
return false
}
// arm64RegMod reports whether a register operand carries the shifted-register
// or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend
// option (R0.UXTW, R3.SXTW<<2).
@@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool {
// arm64RegModifier resolves a modified register operand: the register number,
// the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend
// option (UXTB=0..SXTX=7) with its shift amount.
// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives
// from the parser with the raw token spacing ("@ > 7"), so it is compacted
// before the operator match.
func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) {
name := operandRegName(op)
shift := strings.Join(strings.Fields(op.Addr.Shift), "")
if before, after, ok0 := strings.Cut(name, "."); ok0 {
switch strings.ToUpper(strings.TrimSpace(after)) {
case "UXTB":
@@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
if rm < 0 {
return 0, 0, 0, false, 0, false
}
amount, ok = arm64ShiftAmount(op.Addr.Shift)
amount, ok = arm64ShiftAmount(shift)
if !ok || amount < 0 || amount > 4 {
return 0, 0, 0, false, 0, false
}
return rm, 0, extendOpt, true, amount, true
}
shiftKind = 0 // LSL
shift := strings.TrimSpace(op.Addr.Shift)
switch {
case strings.HasPrefix(shift, "<<"):
shiftKind = 0
@@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext
default:
return 0, 0, 0, false, 0, false
}
amount, ok = arm64ShiftAmount(op.Addr.Shift)
amount, ok = arm64ShiftAmount(shift)
if !ok {
return 0, 0, 0, false, 0, false
}
@@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e
// operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the
// MUL mnemonic, not to these.
func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
// The widening three-operand forms (SMULL, UMNEGL, …) read the
// accumulate register as ZR, already preset in the table's base word.
if len(ops) == 3 {
switch mnem {
case "SMULL", "UMULL", "SMNEGL", "UMNEGL":
default:
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem)
}
rm := arm64RegNum(operandRegName(ops[0]))
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[2]))
if rm < 0 || rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil
}
if len(ops) != 4 {
return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops))
}
@@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// ---- ADD/SUB immediate ----
// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction.
// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction,
// following the toolchain's immediate classification (asm7.go conclass and
// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an
// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and
// otherwise a constant materialisation into REGTMP (R27) followed by the
// register form.
func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 2 && len(ops) != 3 {
return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops))
}
v := arm64Imm64(ops[0])
v, ok := arm64ImmOperandValue(ops[0])
if !ok {
return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw)
}
rd := arm64RegNum(operandRegName(ops[len(ops)-1]))
rn := rd
if len(ops) == 3 {
@@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
sf := uint32(1) // 64-bit
if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" {
sf = 0 // 32-bit
}
// CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings
// keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n).
// CMP/CMN discard the destination.
if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" {
rd = 31 // ZR
}
ws, err := arm64AddSubImmWords(mnem, v, rn, rd)
if err != nil {
return nil, fmt.Errorf("%s: %w", mnem, err)
}
return a64WordsLE(ws...), nil
}
// arm64AddSubImmWords returns the word sequence the toolchain emits for an
// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and
// their W forms. rn and rd are resolved register numbers (a comparison
// discards rd, so the caller passes 31).
func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) {
w := strings.HasSuffix(mnem, "W")
sf := uint32(1) // 64-bit
d := v
if w {
sf = 0 // 32-bit
// The W forms classify the 32-bit value (asm7.go con32class).
d = int64(uint32(v))
}
isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW"
isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" ||
mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW"
op := uint32(0) // ADD
S := uint32(0)
if isSub {
@@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) {
if isS {
S = 1
}
single := func(sh, imm12 uint32) []uint32 {
return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))}
}
if v >= 0 && v <= 0xFFF {
return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil
// imm12: plain, then the one-shifted-by-12 form.
if d >= 0 && d <= 0xFFF {
return single(0, uint32(d)), nil
}
if v >= -2048 && v < 0 {
// Encode as the opposite operation with positive immediate.
opp := op ^ 1
return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil
if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF {
return single(1, uint32(d>>12)), nil
}
// Try with shift by 12.
if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 {
return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil
// ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB
// split into two imm12 instructions, low half first (asm7.go case 48).
// The encoding is complete in itself: no REGTMP, no register form. The S
// forms must not break addition/subtraction, so the toolchain
// reclassifies them and falls through to the materialisation below.
dm := ^d
if w {
dm = ^d & 0xFFFFFFFF
}
// The imm12 field cannot carry the value; rejecting (rather than
// truncating) matches the toolchain, which reports the same shape.
return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v)
_, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf))
if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon {
return []uint32{
a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)),
a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)),
}, nil
}
// Constant into REGTMP (R27), then the register form. The first word
// mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for
// the complement form, the bitmask ORR otherwise, and the full
// omovlconst sequence when no single word carries the value.
var seq []uint32
switch s := arm64Movcon(d); {
case s >= 0:
seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)}
case arm64Movcon(dm) >= 0:
s := arm64Movcon(dm)
seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)}
case isBitcon:
n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf))
seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5}
default:
seq = arm64MovLConst(d, sf)
}
// The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr).
seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd))
for i := range seq[:len(seq)-1] {
seq[i] |= 27 // REGTMP
}
return seq, nil
}
// arm64MovLConst returns the toolchain's multi-word constant sequence for a
// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go
// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is
// returned with the destination field clear so the caller can OR its own
// register in. movcon and movcon-of-complement must fail for d before this
// is reached, so no branch sees all-zero or all-0xFFFF chunks.
func arm64MovLConst(d int64, sf uint32) []uint32 {
if sf == 0 {
// omovlconst AMOVW: both 16-bit halves, low first.
return []uint32{
a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0),
a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0),
}
}
dn := ^d
var immh [4]uint64
zero, neg := 0, 0
for i := range immh {
immh[i] = uint64(d>>(i*16)) & 0xFFFF
switch immh[i] {
case 0:
zero++
case 0xFFFF:
neg++
}
}
mw := func(opc uint32, val int64, chunk int) uint32 {
return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0)
}
var os []uint32
switch {
case zero == 2:
// one MOVZ and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg == 2:
// one MOVN and one MOVK
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// A two-word shortcut: a bitmask in every chunk but one, fixed up by
// a single MOVK (constants from strength-reduced division).
if zero == 0 && neg == 0 {
for i := range 4 {
mask := uint64(0xFFFF) << (i * 16)
for period := 2; period <= 32; period *= 2 {
x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask
if n, immr, imms, ok := arm64Bitmask(x, 1); ok {
os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5)
os = append(os, mw(3, d, i))
return os
}
}
}
}
switch {
case zero >= 1:
// one MOVZ and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(2, d, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0 {
os = append(os, mw(3, d, i))
}
}
case neg >= 1:
// one MOVN and up to three MOVKs
i := 0
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(0, dn, i))
i++
break
}
}
for ; i < 4; i++ {
if immh[i] != 0xFFFF {
os = append(os, mw(3, d, i))
}
}
default:
// one MOVZ and three MOVKs
os = append(os, mw(2, d, 0))
for i := 1; i < 4; i++ {
os = append(os, mw(3, d, i))
}
}
}
return os
}
// ---- MOV pseudo-instruction ----
@@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) {
}
}
// Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk.
var ws []uint32
first := true
for i := range 4 {
chunk := (d >> uint(i*16)) & 0xFFFF
if chunk == 0 {
continue
}
if first {
ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ
first = false
} else {
ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK
}
}
if len(ws) == 0 {
op := uint32(1<<31 | 1<<29 | 0x0a<<24)
return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil
// Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN
// for the first special 16-bit chunk, then MOVK per remaining one, with
// the bitmask-plus-fixup shortcut for strength-reduced constants).
ws := arm64MovLConst(d, sf)
for i := range ws {
ws[i] |= uint32(rd)
}
return a64WordsLE(ws...), nil
}
@@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt
return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases
// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the
// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb
// with imms = lsb+width-1.
func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) {
if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) {
return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem)
}
lsb := arm64Imm64(ops[0])
width := arm64Imm64(ops[2])
rn := arm64RegNum(operandRegName(ops[1]))
rd := arm64RegNum(operandRegName(ops[3]))
if rn < 0 || rd < 0 {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
bits := int64(32) << (baseOp >> 31 & 1)
if lsb < 0 || lsb >= bits {
return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1)
}
if width < 1 || width > bits || lsb+width > bits {
return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits)
}
var immr, imms int64
switch mnem {
case "BFXIL", "BFXILW":
immr, imms = lsb, lsb+width-1
default: // BFI, SBFIZ, UBFIZ
immr, imms = (-lsb)%bits, width-1
}
return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil
}
// encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The
// third field carries Rm or a 5-bit immediate in the same bits, at the
// toolchain's choice of register or immediate operand.
@@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte,
// MRS <sysreg>, Rd MSR $imm4, <sysreg>
// PRFM (Rn), $imm|<op>
func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
// Operand-less returns and pointer-authentication hints.
if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0,
"AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff,
"AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f,
"YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f,
"SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok {
if len(ops) != 0 {
return nil, fmt.Errorf("%s expects no operand", mnem)
}
return a64wordLE(w), nil
}
switch mnem {
case "BRK", "SVC":
base := uint32(0xd4200000)
@@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
return a64wordLE(base | uint32(v)<<5), nil
case "DMB", "DSB", "ISB":
case "DMB", "DSB", "ISB", "CLREX":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
@@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
if v < 0 || v > 15 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v)
}
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem]
base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem]
return a64wordLE(base | uint32(v)<<8), nil
case "HINT":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 127 {
return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v)
}
return a64wordLE(0xd503201f | uint32(v)<<5), nil
case "BTI":
op := operandRegName(ops[0])
base, ok := map[string]uint32{"C": 0xd503245f}[op]
if !ok {
return nil, fmt.Errorf("%s: unknown kind %q", mnem, op)
}
return a64wordLE(base), nil
case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3":
if len(ops) != 1 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("%s expects $immediate", mnem)
}
v := arm64Imm64(ops[0])
if v < 0 || v > 0xFFFF {
return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v)
}
base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003,
"HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002,
"DCPS3": 0xd4a00003}[mnem]
return a64wordLE(base | uint32(v)<<5), nil
case "DC":
if len(ops) != 2 {
return nil, fmt.Errorf("DC expects <op>, Rn")
@@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) {
}
return a64wordLE(base | uint32(rd)&31), nil
case "MSR":
if len(ops) != 2 || !isImmOperand(ops[0]) {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg>")
if len(ops) != 2 {
return nil, fmt.Errorf("MSR expects $immediate, <sysreg> or Rn, <sysreg>")
}
if !isImmOperand(ops[0]) {
// Register form: MSR Rn, <sysreg> (the a64MSRRegOps words).
base, ok := a64MSRRegOps[operandRegName(ops[1])]
if !ok {
return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1]))
}
rs := arm64RegNum(operandRegName(ops[0]))
if rs < 0 {
return nil, fmt.Errorf("MSR: invalid source register")
}
return a64wordLE(base | uint32(rs)&31), nil
}
base, ok := a64MSROps[operandRegName(ops[1])]
if !ok {
@@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) {
// specBit returns the a64SimdVSpec bitmask bit for an arrangement index.
func specBit(i int) uint16 { return 1 << uint(i) }
// arm64SimdZeroImm reports whether the first operand of a SIMD compare is
// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones
// (the toolchain accepts the FP zero only as a spelled float or integer 0).
func arm64SimdZeroImm(mnem string, op *ast.Operand) bool {
if v, ok := arm64ImmOperandValue(op); ok && v == 0 {
return true
}
if !strings.HasPrefix(mnem, "VFCM") {
return false
}
s := strings.Join(strings.Fields(op.Raw), "")
s = strings.TrimPrefix(s, "$")
s = strings.Trim(s, "()")
return s == "0" || s == "0.0"
}
// encodeARM64SimdV encodes an arrangement-aware three-register SIMD
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a
// zero immediate takes its compare-against-zero form instead, and the
// polynomial multiplies read the arrangement from their source operands
// alone, the result spelling (H8, Q1) riding no encoding bits.
// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD
// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take
// their compare-against-zero form instead, and the polynomial multiplies read
// the arrangement from their source operands alone, the result spelling
// (H8, Q1) riding no encoding bits.
func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) {
if mnem == "VPMULL" || mnem == "VPMULL2" {
if len(ops) != 3 {
@@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
rd, _ := arm64VecOf(ops[2])
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil
}
if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) {
if arm64Imm64(ops[0]) != 0 {
if len(ops) == 3 && isImmOperand(ops[0]) {
base, ok := a64SimdVZero[mnem]
if !ok {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
if !arm64SimdZeroImm(mnem, ops[0]) {
return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem)
}
vn, ok1 := arm64VecOf(ops[1])
@@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx {
return nil, fmt.Errorf("invalid register operand in %s", mnem)
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f)
allowed := uint16(0x7f)
if strings.HasPrefix(mnem, "VFCM") {
allowed = 1<<a64Arr4H | 1<<a64Arr8H | 1<<a64Arr2S | 1<<a64Arr4S | 1<<a64Arr2D
}
arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, allowed)
if err != nil {
return nil, err
}
return a64wordLE(0x0e209800 | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
return a64wordLE(base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
if len(ops) != 3 {
return nil, fmt.Errorf("%s expects 3 operands, got %d", mnem, len(ops))
@@ -2802,6 +3199,9 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt
if spec.fixed {
arrBits = 0
}
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(vs[2].reg)), nil
}
@@ -2830,7 +3230,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil {
return nil, err
}
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vs[0].reg)<<16 | uint32(vs[0].reg)<<5 | uint32(vs[1].reg)), nil
}
// VUADDLV spells its arrangement on the source alone; the rest take it
// on both.
@@ -2843,7 +3247,11 @@ func encodeARM64SimdV2(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]by
if err != nil {
return nil, err
}
return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
arrBits := a64ArrBits[arr]
if a64SimdQOnly[mnem] {
arrBits &= 1 << 30
}
return a64wordLE(spec.base | arrBits | uint32(vn.reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64SimdV4 encodes the four-register crypto group (VEOR3, VBCAX:
@@ -2928,7 +3336,7 @@ func encodeARM64SimdV4(mnem string, base uint32, ops []*ast.Operand) ([]byte, er
// index register rides bits 19:16, the first table register bits 9:5, the
// destination bits 4:0 and the table length (registers minus one) bits
// 14:13. The table registers must be consecutive.
func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
func encodeARM64VTBL(mnem string, ops []*ast.Operand) ([]byte, error) {
if len(ops) < 3 {
return nil, fmt.Errorf("VTBL expects index, table list and destination")
}
@@ -2957,7 +3365,11 @@ func encodeARM64VTBL(ops []*ast.Operand) ([]byte, error) {
default:
return nil, fmt.Errorf("VTBL: invalid arrangement %q", vi.arr)
}
return a64wordLE(0x0e000000 | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
base := uint32(0x0e000000)
if mnem == "VTBX" {
base |= 1 << 12
}
return a64wordLE(base | q | uint32(len(ts)-1)<<13 | uint32(vi.reg)<<16 | uint32(ts[0].reg)<<5 | uint32(vd.reg)), nil
}
// encodeARM64Dup encodes the SIMD element moves VDUP and VMOV spell with
@@ -3263,12 +3675,12 @@ func encodeARM64ShiftImm(mnem string, base uint32, ops []*ast.Operand) ([]byte,
}
var immval int64
switch mnem {
case "VSHL":
case "VSHL", "VSLI", "VSQSHL", "VUQSHL", "VSQSHLU":
if sh < 0 || sh >= esize {
return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1)
}
immval = esize + sh
default: // VUSHR, VSRI
default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR
if sh < 1 || sh > esize {
return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize)
}
@@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) {
sym *ast.Symbol // the parsed frame-relative reference (when mem)
}
aliases := map[string]alias{}
raws := map[string]string{}
for _, d := range f.Decls {
pre, ok := d.(*ast.Preproc)
if !ok {
@@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) {
strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") {
continue
}
raws[name] = body
}
// Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute
// transitively until nothing changes, bounded against cycles.
for range 8 {
changed := false
for name, body := range raws {
if next, ok := raws[body]; ok && next != body {
raws[name] = next
changed = true
}
}
if !changed {
break
}
}
for name, body := range raws {
isReg := func(s string) bool {
if arm64RegNum(s) >= 0 {
return true
@@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) {
return
}
// replace rewrites whole-word occurrences of the alias names in s.
replace := func(s string) string {
if s == "" {
return s
}
out := strings.Fields(s)
for i, w := range out {
if a, ok := aliases[w]; ok {
out[i] = a.raw
}
}
if len(out) == 0 {
return s
}
return strings.Join(out, " ")
}
// replaceToken rewrites an operand whose whole text is one alias use
// possibly followed by syntax (POLY.D[0]): the alias must be a prefix
// ending at a non-identifier character.
@@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) {
return s, false
}
// replaceScan rewrites alias uses inside a composite operand (a
// parenthesised memory operand or a bracketed register list): every
// identifier run of word and dot characters is matched against the alias
// names, everything else copies verbatim. The whitespace-split replace
// above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their
// punctuation attached.
replaceScan := func(s string) string {
var b strings.Builder
for i := 0; i < len(s); {
if isAliasWordByte(s[i]) || s[i] == '.' {
j := i
for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') {
j++
}
if nn, ok := replaceToken(s[i:j]); ok {
b.WriteString(nn)
} else {
b.WriteString(s[i:j])
}
i = j
continue
}
b.WriteByte(s[i])
i++
}
return b.String()
}
for _, d := range f.Decls {
t, ok := d.(*ast.Text)
if !ok {
@@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) {
op.Raw = a.raw
continue
}
op.Addr.Sym.Name = nn
op.Addr.Sym.Raw = nn
op.Raw = nn
op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn
// The span shape depends on what trailed the
// name: an element or arrangement selector
// (POLY.D[0], POLY.B16) rides in Shift and folds
// back onto the rewritten token; a shift
// operator stays in Shift while the span carries
// the bare register; a split list keeps its
// closing bracket, so the rewrite goes through
// the scan.
sfx := strings.Join(strings.Fields(op.Addr.Shift), "")
switch {
case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]":
// Element or arrangement selectors and the
// closing bracket of a split list belong to
// the token text.
op.Raw = nn + sfx
op.Addr.Shift = ""
case op.Addr.Shift != "":
op.Raw = nn
default:
op.Raw = replaceScan(op.Raw)
}
continue
}
}
@@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) {
// [V0.B16, V1.B16] with aliased members.
if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") ||
strings.HasPrefix(strings.TrimSpace(op.Raw), "[") {
op.Raw = replace(op.Raw)
op.Raw = replaceScan(op.Raw)
}
}
}