From 9b238a525aad43df13bc25ad044bcb3ff98ffb18 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Sun, 20 Sep 2026 14:25:47 +0200 Subject: [PATCH] feat(arm64): wide immediates, SIMD compare and system operand forms Assisted-by: GLM 5.3 Flash --- asm/arm64_assemble.go | 677 ++++++++++++++++++++++++----- asm/arm64_encode.go | 347 ++++++++++++--- asm/arm64_encode_test.go | 177 +++++++- testdata/verify/carryshift_arm64.s | 48 ++ testdata/verify/widenimm_arm64.s | 70 +++ 5 files changed, 1143 insertions(+), 176 deletions(-) create mode 100644 testdata/verify/carryshift_arm64.s create mode 100644 testdata/verify/widenimm_arm64.s diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index 5686b5a..b9b178b 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -5,6 +5,7 @@ package asm import ( "fmt" + "math/bits" "strconv" "strings" @@ -271,18 +272,28 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { case "MOV", "MOVD", "MOVW", "MOVWU", "MOVH", "MOVHU", "MOVB", "MOVBU", "FMOVS", "FMOVD": return arm64MovSize(mnem, ops, fi) - case "ADD", "ADDW", "SUB", "SUBW": + case "ADD", "ADDW", "SUB", "SUBW", "CMP", "CMPW", "CMN", "CMNW", + "ADDS", "ADDSW", "SUBS", "SUBSW": if len(ops) >= 2 && isImmOperand(ops[0]) { - v := arm64Imm64(ops[0]) - // Small immediate (0..4095 or -2048..-1) fits in one instruction. - if v >= 0 && v <= 0xFFF { - return 4 + // Size exactly as the encoder will emit: a single imm12 word, the + // two-word ADDCON2 split, or a materialisation into REGTMP plus + // the register form. Anything else would desynchronise the label + // offsets of pass 1 from the bytes pass 2 lays down. + if v, ok := arm64ImmOperandValue(ops[0]); ok { + rn, rd := 0, 0 + if n := arm64RegNum(operandRegName(ops[len(ops)-1])); n >= 0 { + rd = n + } + if len(ops) == 3 { + if n := arm64RegNum(operandRegName(ops[1])); n >= 0 { + rn = n + } + } + if ws, err := arm64AddSubImmWords(mnem, v, rn, rd); err == nil { + return 4 * len(ws) + } } - if v >= -2048 && v < 0 { - return 4 - } - // Larger immediates need MOV materialisation + op. - return 8 + return 4 } } return 4 @@ -439,6 +450,11 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 return encodeARM64Bitfield(mnem, enc.op, ops) } + // Bitfield aliases: BFI, BFXIL, SBFIZ, UBFIZ and their W forms. + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FBitfieldAlias { + return encodeARM64BitfieldAlias(mnem, enc.op, ops) + } + // EXTR. if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FEXTR { return encodeARM64Extr(mnem, enc.op, ops) @@ -510,8 +526,10 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 // Arrangement-aware SIMD three-register (VADD, VAND, VCMEQ, VZIP1, // VPMULL, VRAX1 and friends). VADD, VSUB and VMUL appear here too, so - // this check precedes the plain SIMD3 path below. - if spec, ok := a64SimdVTable[mnem]; ok { + // this check precedes the plain SIMD3 path below. VCMLE and VCMLT exist + // only in the zero-immediate form (a64SimdVZero), so they route here with + // an empty register-form spec. + if spec, ok := a64SimdVTable[mnem]; ok || a64SimdVZero[mnem] != 0 { return encodeARM64SimdV(mnem, spec, ops) } @@ -527,8 +545,8 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64 } // SIMD table lookup. - if mnem == "VTBL" { - return encodeARM64VTBL(ops) + if mnem == "VTBL" || mnem == "VTBX" { + return encodeARM64VTBL(mnem, ops) } // SIMD structure loads and stores (VLD1, VST1, VLD1.P, VST1.P, VLD1R, @@ -559,13 +577,19 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri } op := ops[0] - // Branch to the program counter itself: JMP (PC) spins forever. The - // toolchain encodes it as an unconditional branch with a zero offset. + // Branch to the program counter: JMP (PC) spins forever, and a spelled + // offset (CALL -1(PC), the return stub) rides the imm26 field in word + // units. The toolchain encodes both as a plain branch of that offset. if op.Addr.Base == "PC" || (op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "PC") { - if link { - return nil, fmt.Errorf("%s: branch to PC is not a call", mnem) + rel := op.Addr.Offset + if rel < -(1<<25) || rel >= (1<<25) { + return nil, fmt.Errorf("%s: branch offset %d out of 26-bit range", mnem, rel) } - return a64wordLE(a64Branch(0, 0)), nil + bop := uint32(0) // B + if link { + bop = 1 // BL + } + return a64wordLE(a64Branch(bop, int32(rel))), nil } // Register-indirect: JMP (R0) is BR R0, CALL (R0) is BLR R0. The @@ -669,7 +693,9 @@ func encodeARM64BranchCond(mnem string, baseOp uint32, ops []*ast.Operand, pc in // the ADD/SUB-with-flags family an add/sub immediate. func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { isCmp := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || mnem == "TST" || mnem == "TSTW" - isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "MVN" || mnem == "MVNW" + isNeg := mnem == "NEG" || mnem == "NEGW" || mnem == "NEGS" || mnem == "NEGSW" || + mnem == "MVN" || mnem == "MVNW" || + mnem == "NGC" || mnem == "NGCW" || mnem == "NGCS" || mnem == "NGCSW" // Bitmask immediate: AND/ORR/EOR/ANDS/BIC and friends take the repeating // bit-pattern immediate. The inverted mnemonics (BIC, BICS) encode the @@ -678,7 +704,8 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er var logical bool switch mnem { case "AND", "ANDW", "ANDS", "ANDSW", "ORR", "ORRW", "EOR", "EORW", - "BIC", "BICW", "BICS", "BICSW", "TST", "TSTW": + "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW", + "TST", "TSTW": logical = true } if logical { @@ -688,7 +715,7 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er } inverted := false switch mnem { - case "BIC", "BICW", "BICS", "BICSW": + case "BIC", "BICW", "BICS", "BICSW", "ORN", "ORNW", "EON", "EONW": inverted = true } if inverted { @@ -700,7 +727,37 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er } n, immr, imms, ok := a64LogicalImm(v, width) if !ok { - return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " ")) + // Beyond the bitmask immediates the toolchain materialises + // the constant into REGTMP (R27) and uses the register form + // (asm7.go cases 62 and 13). BIC/ORN/EON read the written + // value, so the materialisation uses v before any inversion. + written := v + if inverted { + written = ^v + } + width := mnem + if strings.HasSuffix(mnem, "W") { + width = "MOVW" + } else { + width = "MOVD" + } + mw, merr := encodeARM64LoadImm(27, written, width) + var rn, rd int + switch len(ops) { + case 3: + rn = arm64RegNum(operandRegName(ops[1])) + rd = arm64RegNum(operandRegName(ops[2])) + default: + rd = arm64RegNum(operandRegName(ops[1])) + rn = rd + } + if isCmp { + rd = 31 + } + if merr != nil || rn < 0 || rd < 0 { + return nil, fmt.Errorf("%s: immediate %q is not a logical (bitmask) immediate", mnem, strings.Join(strings.Fields(ops[0].Raw), " ")) + } + return append(mw, a64wordLE(baseOp|27<<16|uint32(rn)<<5|uint32(rd))...), nil } opc := (baseOp >> 29) & 7 sf := (baseOp >> 31) & 1 @@ -735,7 +792,18 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er return nil, fmt.Errorf("%s: extend modifier only applies to ADD/SUB and comparisons", mnem) } if !isExtend { - if amount < 0 || amount > 63 { + // ROR rides the shifted-register field only for the logical + // group; the toolchain reports "unsupported shift operator" for + // the arithmetic forms, whose shift=11 encoding is unallocated. + if shiftBits == 3 && !arm64LogicalShifted(mnem) { + return nil, fmt.Errorf("%s: unsupported shift operator", mnem) + } + // The imm6 field is 5 bits and truncates at the 32-bit width. + limit := 63 + if strings.HasSuffix(mnem, "W") { + limit = 31 + } + if amount < 0 || amount > limit { return nil, fmt.Errorf("%s: shift amount %d out of range", mnem, amount) } // SP-based ADD/SUB have no shifted-register encoding: the @@ -781,6 +849,20 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er switch len(ops) { case 3: + // The carry family carries an immediate spelling in three operands + // too: ADC $0, Rn, Rd reads the carry into Rd with ZR as the register + // operand, the same shape the two-operand form takes. + if isImmOperand(ops[0]) && arm64CarryOp(mnem) { + if v := arm64Imm64(ops[0]); v != 0 { + return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) + } + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[2])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | 31<<16 | uint32(rn)<<5 | uint32(rd)), nil + } // OP Rm, Rn, Rd rm := arm64RegNum(operandRegName(ops[0])) rn := arm64RegNum(operandRegName(ops[1])) @@ -833,6 +915,31 @@ func encodeARM64DPSR(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, er return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } +// arm64LogicalShifted reports whether a mnemonic belongs to the logical +// shifted-register group, the only forms whose register operand accepts the +// ROR shift kind (AND/ORR/EOR/BIC and their complements, flags and W forms). +func arm64LogicalShifted(mnem string) bool { + switch mnem { + case "AND", "ANDW", "ANDS", "ANDSW", "BIC", "BICW", "BICS", "BICSW", + "ORR", "ORRW", "ORN", "ORNW", "EOR", "EORW", "EON", "EONW", + "TST", "TSTW", "MVN", "MVNW": + return true + } + return false +} + +// arm64CarryOp reports whether a mnemonic belongs to the carry-using +// arithmetic family (ADC/ADCS/SBC/SBCS and the W forms), the only +// data-processing instructions the toolchain accepts an immediate $0 +// operand spelling for. +func arm64CarryOp(mnem string) bool { + switch mnem { + case "ADC", "ADCW", "ADCS", "ADCSW", "SBC", "SBCW", "SBCS", "SBCSW": + return true + } + return false +} + // arm64RegMod reports whether a register operand carries the shifted-register // or extend-modifier syntax: a shift suffix (R0<<2) or a spelled extend // option (R0.UXTW, R3.SXTW<<2). @@ -849,9 +956,12 @@ func arm64RegMod(op *ast.Operand) bool { // arm64RegModifier resolves a modified register operand: the register number, // the shifted-register kind (0 LSL, 1 LSR) with its amount, or the extend -// option (UXTB=0..SXTX=7) with its shift amount. +// option (UXTB=0..SXTX=7) with its shift amount. The shift suffix arrives +// from the parser with the raw token spacing ("@ > 7"), so it is compacted +// before the operator match. func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, extend bool, amount int, ok bool) { name := operandRegName(op) + shift := strings.Join(strings.Fields(op.Addr.Shift), "") if before, after, ok0 := strings.Cut(name, "."); ok0 { switch strings.ToUpper(strings.TrimSpace(after)) { case "UXTB": @@ -878,14 +988,13 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext if rm < 0 { return 0, 0, 0, false, 0, false } - amount, ok = arm64ShiftAmount(op.Addr.Shift) + amount, ok = arm64ShiftAmount(shift) if !ok || amount < 0 || amount > 4 { return 0, 0, 0, false, 0, false } return rm, 0, extendOpt, true, amount, true } shiftKind = 0 // LSL - shift := strings.TrimSpace(op.Addr.Shift) switch { case strings.HasPrefix(shift, "<<"): shiftKind = 0 @@ -898,7 +1007,7 @@ func arm64RegModifier(op *ast.Operand) (rm int, shiftKind, extendOpt uint32, ext default: return 0, 0, 0, false, 0, false } - amount, ok = arm64ShiftAmount(op.Addr.Shift) + amount, ok = arm64ShiftAmount(shift) if !ok { return 0, 0, 0, false, 0, false } @@ -1000,6 +1109,22 @@ func encodeARM64Shift(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, e // operands are mandatory; MUL's two-operand spelling (Ra = ZR) belongs to the // MUL mnemonic, not to these. func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + // The widening three-operand forms (SMULL, UMNEGL, …) read the + // accumulate register as ZR, already preset in the table's base word. + if len(ops) == 3 { + switch mnem { + case "SMULL", "UMULL", "SMNEGL", "UMNEGL": + default: + return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got 3", mnem) + } + rm := arm64RegNum(operandRegName(ops[0])) + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[2])) + if rm < 0 || rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + return a64wordLE(baseOp | uint32(rm)<<16 | uint32(rn)<<5 | uint32(rd)), nil + } if len(ops) != 4 { return nil, fmt.Errorf("%s expects 4 operands (Rm, Ra, Rn, Rd), got %d", mnem, len(ops)) } @@ -1015,12 +1140,20 @@ func encodeARM64MAddSub(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, // ---- ADD/SUB immediate ---- -// encodeARM64AddSubImm encodes an ADD/SUB immediate instruction. +// encodeARM64AddSubImm encodes an ADD/SUB-family immediate instruction, +// following the toolchain's immediate classification (asm7.go conclass and +// optab cases 2, 48, 62 and 13): a single imm12 form when the value fits, an +// ADDCON2 split into two imm12 instructions for the plain ADD/SUB band, and +// otherwise a constant materialisation into REGTMP (R27) followed by the +// register form. func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if len(ops) != 2 && len(ops) != 3 { return nil, fmt.Errorf("%s expects 2 or 3 operands, got %d", mnem, len(ops)) } - v := arm64Imm64(ops[0]) + v, ok := arm64ImmOperandValue(ops[0]) + if !ok { + return nil, fmt.Errorf("%s: unsupported immediate %q", mnem, ops[0].Raw) + } rd := arm64RegNum(operandRegName(ops[len(ops)-1])) rn := rd if len(ops) == 3 { @@ -1029,20 +1162,33 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if rn < 0 || rd < 0 { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - - isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW" - isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || - mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" - sf := uint32(1) // 64-bit - if mnem == "ADDW" || mnem == "SUBW" || mnem == "CMPW" || mnem == "CMNW" || mnem == "ADDSW" || mnem == "SUBSW" { - sf = 0 // 32-bit - } - // CMP/CMN discard the destination. The two-operand ADDS/SUBS spellings - // keep Rd = Rn (the toolchain encodes SUBS $n, R3 as SUBS R3, R3, #n). + // CMP/CMN discard the destination. if mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" { rd = 31 // ZR } + ws, err := arm64AddSubImmWords(mnem, v, rn, rd) + if err != nil { + return nil, fmt.Errorf("%s: %w", mnem, err) + } + return a64WordsLE(ws...), nil +} +// arm64AddSubImmWords returns the word sequence the toolchain emits for an +// ADD/SUB-family immediate: the mnemonics ADD, ADDS, SUB, SUBS, CMP, CMN and +// their W forms. rn and rd are resolved register numbers (a comparison +// discards rd, so the caller passes 31). +func arm64AddSubImmWords(mnem string, v int64, rn, rd int) ([]uint32, error) { + w := strings.HasSuffix(mnem, "W") + sf := uint32(1) // 64-bit + d := v + if w { + sf = 0 // 32-bit + // The W forms classify the 32-bit value (asm7.go con32class). + d = int64(uint32(v)) + } + isSub := mnem == "SUB" || mnem == "SUBW" || mnem == "CMP" || mnem == "CMPW" || mnem == "SUBS" || mnem == "SUBSW" + isS := mnem == "CMP" || mnem == "CMPW" || mnem == "CMN" || mnem == "CMNW" || + mnem == "ADDS" || mnem == "ADDSW" || mnem == "SUBS" || mnem == "SUBSW" op := uint32(0) // ADD S := uint32(0) if isSub { @@ -1051,22 +1197,177 @@ func encodeARM64AddSubImm(mnem string, ops []*ast.Operand) ([]byte, error) { if isS { S = 1 } + single := func(sh, imm12 uint32) []uint32 { + return []uint32{a64AddSub(sf, op, S, sh, imm12, uint32(rn), uint32(rd))} + } - if v >= 0 && v <= 0xFFF { - return a64wordLE(a64AddSub(sf, op, S, 0, uint32(v), uint32(rn), uint32(rd))), nil + // imm12: plain, then the one-shifted-by-12 form. + if d >= 0 && d <= 0xFFF { + return single(0, uint32(d)), nil } - if v >= -2048 && v < 0 { - // Encode as the opposite operation with positive immediate. - opp := op ^ 1 - return a64wordLE(a64AddSub(sf, opp, S, 0, uint32(-v), uint32(rn), uint32(rd))), nil + if d >= 0 && d&0xFFF == 0 && d>>12 <= 0xFFF { + return single(1, uint32(d>>12)), nil } - // Try with shift by 12. - if v >= 0 && v <= 0xFFF000 && v&0xFFF == 0 { - return a64wordLE(a64AddSub(sf, op, S, 1, uint32(v>>12), uint32(rn), uint32(rd))), nil + + // ADDCON2 band (0..0xFFFFFF, neither bitmask nor movcon): plain ADD/SUB + // split into two imm12 instructions, low half first (asm7.go case 48). + // The encoding is complete in itself: no REGTMP, no register form. The S + // forms must not break addition/subtraction, so the toolchain + // reclassifies them and falls through to the materialisation below. + dm := ^d + if w { + dm = ^d & 0xFFFFFFFF } - // The imm12 field cannot carry the value; rejecting (rather than - // truncating) matches the toolchain, which reports the same shape. - return nil, fmt.Errorf("%s: immediate %d out of range for single instruction", mnem, v) + _, _, _, isBitcon := arm64Bitmask(uint64(d), int(sf)) + if !isS && d >= 0 && d <= 0xFFFFFF && arm64Movcon(d) < 0 && arm64Movcon(dm) < 0 && !isBitcon { + return []uint32{ + a64AddSub(sf, op, 0, 0, uint32(d)&0xFFF, uint32(rn), uint32(rd)), + a64AddSub(sf, op, 0, 1, uint32(d>>12)&0xFFF, uint32(rd), uint32(rd)), + }, nil + } + + // Constant into REGTMP (R27), then the register form. The first word + // mirrors omovconst (asm7.go case 62): MOVZ for a movcon value, MOVN for + // the complement form, the bitmask ORR otherwise, and the full + // omovlconst sequence when no single word carries the value. + var seq []uint32 + switch s := arm64Movcon(d); { + case s >= 0: + seq = []uint32{a64MoveWide(sf, 2, uint32(s>>4), uint32(d>>uint(s))&0xFFFF, 0)} + case arm64Movcon(dm) >= 0: + s := arm64Movcon(dm) + seq = []uint32{a64MoveWide(sf, 0, uint32(s>>4), uint32(dm>>uint(s))&0xFFFF, 0)} + case isBitcon: + n, immr, imms, _ := arm64Bitmask(uint64(d), int(sf)) + seq = []uint32{sf<<31 | 1<<29 | 0x24<<23 | n<<22 | immr<<16 | imms<<10 | 31<<5} + default: + seq = arm64MovLConst(d, sf) + } + // The register form reads REGTMP: Rd = Rn op R27 (opxrrr/oprrr). + seq = append(seq, a64InstrTable[mnem].op|27<<16|uint32(rn)<<5|uint32(rd)) + for i := range seq[:len(seq)-1] { + seq[i] |= 27 // REGTMP + } + return seq, nil +} + +// arm64MovLConst returns the toolchain's multi-word constant sequence for a +// value neither MOVZ, MOVN nor a bitmask immediate carries (asm7.go +// omovlconst, AMOVD case; the W form is always MOVZW+MOVKW). Every word is +// returned with the destination field clear so the caller can OR its own +// register in. movcon and movcon-of-complement must fail for d before this +// is reached, so no branch sees all-zero or all-0xFFFF chunks. +func arm64MovLConst(d int64, sf uint32) []uint32 { + if sf == 0 { + // omovlconst AMOVW: both 16-bit halves, low first. + return []uint32{ + a64MoveWide(0, 2, 0, uint32(d)&0xFFFF, 0), + a64MoveWide(0, 3, 1, uint32(d>>16)&0xFFFF, 0), + } + } + dn := ^d + var immh [4]uint64 + zero, neg := 0, 0 + for i := range immh { + immh[i] = uint64(d>>(i*16)) & 0xFFFF + switch immh[i] { + case 0: + zero++ + case 0xFFFF: + neg++ + } + } + mw := func(opc uint32, val int64, chunk int) uint32 { + return a64MoveWide(1, opc, uint32(chunk), uint32(val>>(16*chunk))&0xFFFF, 0) + } + var os []uint32 + switch { + case zero == 2: + // one MOVZ and one MOVK + i := 0 + for ; i < 4; i++ { + if immh[i] != 0 { + os = append(os, mw(2, d, i)) + i++ + break + } + } + for ; i < 4; i++ { + if immh[i] != 0 { + os = append(os, mw(3, d, i)) + } + } + case neg == 2: + // one MOVN and one MOVK + i := 0 + for ; i < 4; i++ { + if immh[i] != 0xFFFF { + os = append(os, mw(0, dn, i)) + i++ + break + } + } + for ; i < 4; i++ { + if immh[i] != 0xFFFF { + os = append(os, mw(3, d, i)) + } + } + default: + // A two-word shortcut: a bitmask in every chunk but one, fixed up by + // a single MOVK (constants from strength-reduced division). + if zero == 0 && neg == 0 { + for i := range 4 { + mask := uint64(0xFFFF) << (i * 16) + for period := 2; period <= 32; period *= 2 { + x := uint64(d)&^mask | bits.RotateLeft64(uint64(d), max(period, 16))&mask + if n, immr, imms, ok := arm64Bitmask(x, 1); ok { + os = append(os, 1<<31|1<<29|0x24<<23|n<<22|immr<<16|imms<<10|31<<5) + os = append(os, mw(3, d, i)) + return os + } + } + } + } + switch { + case zero >= 1: + // one MOVZ and up to three MOVKs + i := 0 + for ; i < 4; i++ { + if immh[i] != 0 { + os = append(os, mw(2, d, i)) + i++ + break + } + } + for ; i < 4; i++ { + if immh[i] != 0 { + os = append(os, mw(3, d, i)) + } + } + case neg >= 1: + // one MOVN and up to three MOVKs + i := 0 + for ; i < 4; i++ { + if immh[i] != 0xFFFF { + os = append(os, mw(0, dn, i)) + i++ + break + } + } + for ; i < 4; i++ { + if immh[i] != 0xFFFF { + os = append(os, mw(3, d, i)) + } + } + default: + // one MOVZ and three MOVKs + os = append(os, mw(2, d, 0)) + for i := 1; i < 4; i++ { + os = append(os, mw(3, d, i)) + } + } + } + return os } // ---- MOV pseudo-instruction ---- @@ -1310,24 +1611,12 @@ func encodeARM64LoadImm(rd int, v int64, mnem string) ([]byte, error) { } } - // Multi-instruction: MOVZ + MOVK for each non-zero 16-bit chunk. - var ws []uint32 - first := true - for i := range 4 { - chunk := (d >> uint(i*16)) & 0xFFFF - if chunk == 0 { - continue - } - if first { - ws = append(ws, a64MoveWide(sf, 2, uint32(i), uint32(chunk), uint32(rd))) // MOVZ - first = false - } else { - ws = append(ws, a64MoveWide(sf, 3, uint32(i), uint32(chunk), uint32(rd))) // MOVK - } - } - if len(ws) == 0 { - op := uint32(1<<31 | 1<<29 | 0x0a<<24) - return a64wordLE(op | 31<<16 | 31<<5 | uint32(rd)), nil + // Multi-instruction: the toolchain's omovlconst sequence (MOVZ or MOVN + // for the first special 16-bit chunk, then MOVK per remaining one, with + // the bitmask-plus-fixup shortcut for strength-reduced constants). + ws := arm64MovLConst(d, sf) + for i := range ws { + ws[i] |= uint32(rd) } return a64WordsLE(ws...), nil } @@ -2274,6 +2563,38 @@ func encodeARM64Bitfield2(mnem string, baseOp uint32, ops []*ast.Operand) ([]byt return a64wordLE(baseOp | n | immr<<16 | uint32(lsb+width-1)<<10 | uint32(rn)<<5 | uint32(rd)), nil } +// encodeARM64BitfieldAlias encodes the four-operand bitfield aliases +// ($lsb, Rn, $width, Rd): BFI and the signed/unsigned FIZ forms insert the +// field at (-lsb mod W) with imms = width-1, and BFXIL extracts from lsb +// with imms = lsb+width-1. +func encodeARM64BitfieldAlias(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, error) { + if len(ops) != 4 || !isImmOperand(ops[0]) || !isImmOperand(ops[2]) { + return nil, fmt.Errorf("%s expects 4 operands ($lsb, Rn, $width, Rd)", mnem) + } + lsb := arm64Imm64(ops[0]) + width := arm64Imm64(ops[2]) + rn := arm64RegNum(operandRegName(ops[1])) + rd := arm64RegNum(operandRegName(ops[3])) + if rn < 0 || rd < 0 { + return nil, fmt.Errorf("invalid register operand in %s", mnem) + } + bits := int64(32) << (baseOp >> 31 & 1) + if lsb < 0 || lsb >= bits { + return nil, fmt.Errorf("%s: lsb %d out of range (0..%d)", mnem, lsb, bits-1) + } + if width < 1 || width > bits || lsb+width > bits { + return nil, fmt.Errorf("%s: illegal bit number (lsb %d width %d, register %d bits)", mnem, lsb, width, bits) + } + var immr, imms int64 + switch mnem { + case "BFXIL", "BFXILW": + immr, imms = lsb, lsb+width-1 + default: // BFI, SBFIZ, UBFIZ + immr, imms = (-lsb)%bits, width-1 + } + return a64wordLE(baseOp | uint32(immr)<<16 | uint32(imms)<<10 | uint32(rn)<<5 | uint32(rd)), nil +} + // encodeARM64CondCmp encodes CCMP/CCMN: (cond, Rn, Rm|$imm, $nzcv). The // third field carries Rm or a 5-bit immediate in the same bits, at the // toolchain's choice of register or immediate operand. @@ -2487,6 +2808,17 @@ func encodeARM64AcqRel(mnem string, baseOp uint32, ops []*ast.Operand) ([]byte, // MRS , Rd MSR $imm4, // PRFM (Rn), $imm| func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { + // Operand-less returns and pointer-authentication hints. + if w, ok := map[string]uint32{"DRPS": 0xd6bf03e0, "ERET": 0xd69f03e0, + "AUTIASP": 0xd50323bf, "AUTIBSP": 0xd50323ff, + "AUTIA1716": 0xd503211f, "AUTIB1716": 0xd503213f, + "YIELD": 0xd503203d, "WFE": 0xd503205f, "WFI": 0xd503207f, + "SEVL": 0xd50320bf, "SEV": 0xd503209f}[mnem]; ok { + if len(ops) != 0 { + return nil, fmt.Errorf("%s expects no operand", mnem) + } + return a64wordLE(w), nil + } switch mnem { case "BRK", "SVC": base := uint32(0xd4200000) @@ -2504,7 +2836,7 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v) } return a64wordLE(base | uint32(v)<<5), nil - case "DMB", "DSB", "ISB": + case "DMB", "DSB", "ISB", "CLREX": if len(ops) != 1 || !isImmOperand(ops[0]) { return nil, fmt.Errorf("%s expects $immediate", mnem) } @@ -2512,8 +2844,36 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { if v < 0 || v > 15 { return nil, fmt.Errorf("%s: immediate %d out of range (0..15)", mnem, v) } - base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df}[mnem] + base := map[string]uint32{"DMB": 0xd50330bf, "DSB": 0xd503309f, "ISB": 0xd50330df, "CLREX": 0xd503305f}[mnem] return a64wordLE(base | uint32(v)<<8), nil + case "HINT": + if len(ops) != 1 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects $immediate", mnem) + } + v := arm64Imm64(ops[0]) + if v < 0 || v > 127 { + return nil, fmt.Errorf("%s: immediate %d out of range (0..127)", mnem, v) + } + return a64wordLE(0xd503201f | uint32(v)<<5), nil + case "BTI": + op := operandRegName(ops[0]) + base, ok := map[string]uint32{"C": 0xd503245f}[op] + if !ok { + return nil, fmt.Errorf("%s: unknown kind %q", mnem, op) + } + return a64wordLE(base), nil + case "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3": + if len(ops) != 1 || !isImmOperand(ops[0]) { + return nil, fmt.Errorf("%s expects $immediate", mnem) + } + v := arm64Imm64(ops[0]) + if v < 0 || v > 0xFFFF { + return nil, fmt.Errorf("%s: immediate %d out of range (0..65535)", mnem, v) + } + base := map[string]uint32{"HLT": 0xd4400000, "SMC": 0xd4000003, + "HVC": 0xd4000002, "DCPS1": 0xd4a00001, "DCPS2": 0xd4a00002, + "DCPS3": 0xd4a00003}[mnem] + return a64wordLE(base | uint32(v)<<5), nil case "DC": if len(ops) != 2 { return nil, fmt.Errorf("DC expects , Rn") @@ -2541,8 +2901,20 @@ func encodeARM64Sys(mnem string, ops []*ast.Operand) ([]byte, error) { } return a64wordLE(base | uint32(rd)&31), nil case "MSR": - if len(ops) != 2 || !isImmOperand(ops[0]) { - return nil, fmt.Errorf("MSR expects $immediate, ") + if len(ops) != 2 { + return nil, fmt.Errorf("MSR expects $immediate, or Rn, ") + } + if !isImmOperand(ops[0]) { + // Register form: MSR Rn, (the a64MSRRegOps words). + base, ok := a64MSRRegOps[operandRegName(ops[1])] + if !ok { + return nil, fmt.Errorf("MSR: unknown system register %q", operandRegName(ops[1])) + } + rs := arm64RegNum(operandRegName(ops[0])) + if rs < 0 { + return nil, fmt.Errorf("MSR: invalid source register") + } + return a64wordLE(base | uint32(rs)&31), nil } base, ok := a64MSROps[operandRegName(ops[1])] if !ok { @@ -2738,11 +3110,28 @@ func arm64SimdArrs(mnem string, arrs []string, allowed uint16) (int, error) { // specBit returns the a64SimdVSpec bitmask bit for an arrangement index. func specBit(i int) uint16 { return 1 << uint(i) } +// arm64SimdZeroImm reports whether the first operand of a SIMD compare is +// the zero immediate: $0 for the integer compares, $(0.0) for the FP ones +// (the toolchain accepts the FP zero only as a spelled float or integer 0). +func arm64SimdZeroImm(mnem string, op *ast.Operand) bool { + if v, ok := arm64ImmOperandValue(op); ok && v == 0 { + return true + } + if !strings.HasPrefix(mnem, "VFCM") { + return false + } + s := strings.Join(strings.Fields(op.Raw), "") + s = strings.TrimPrefix(s, "$") + s = strings.Trim(s, "()") + return s == "0" || s == "0.0" +} + // encodeARM64SimdV encodes an arrangement-aware three-register SIMD -// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. VCMEQ with a -// zero immediate takes its compare-against-zero form instead, and the -// polynomial multiplies read the arrangement from their source operands -// alone, the result spelling (H8, Q1) riding no encoding bits. +// instruction: word = base | arrBits | Rm<<16 | Rn<<5 | Rd. The SIMD +// compares with a zero immediate (VCMEQ $0 and friends, a64SimdVZero) take +// their compare-against-zero form instead, and the polynomial multiplies read +// the arrangement from their source operands alone, the result spelling +// (H8, Q1) riding no encoding bits. func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byte, error) { if mnem == "VPMULL" || mnem == "VPMULL2" { if len(ops) != 3 { @@ -2767,8 +3156,12 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt rd, _ := arm64VecOf(ops[2]) return a64wordLE(spec.base | a64ArrBits[arr] | uint32(vs[0].reg)<<16 | uint32(vs[1].reg)<<5 | uint32(rd.reg)), nil } - if mnem == "VCMEQ" && len(ops) == 3 && isImmOperand(ops[0]) { - if arm64Imm64(ops[0]) != 0 { + if len(ops) == 3 && isImmOperand(ops[0]) { + base, ok := a64SimdVZero[mnem] + if !ok { + return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) + } + if !arm64SimdZeroImm(mnem, ops[0]) { return nil, fmt.Errorf("%s: only $0 is supported as immediate", mnem) } vn, ok1 := arm64VecOf(ops[1]) @@ -2776,11 +3169,15 @@ func encodeARM64SimdV(mnem string, spec a64SimdVSpec, ops []*ast.Operand) ([]byt if !ok1 || !ok2 || vn.hasIdx || vd.hasIdx { return nil, fmt.Errorf("invalid register operand in %s", mnem) } - arr, err := arm64SimdArrs(mnem, []string{vn.arr, vd.arr}, 0x7f) + allowed := uint16(0x7f) + if strings.HasPrefix(mnem, "VFCM") { + allowed = 1<= esize { return nil, fmt.Errorf("%s: shift %d out of range (0..%d)", mnem, sh, esize-1) } immval = esize + sh - default: // VUSHR, VSRI + default: // VUSHR, VSRI, VSSHR, VSRA, VSRSHR if sh < 1 || sh > esize { return nil, fmt.Errorf("%s: shift %d out of range (1..%d)", mnem, sh, esize) } @@ -3504,6 +3916,7 @@ func arm64ResolveAliases(f *ast.File) { sym *ast.Symbol // the parsed frame-relative reference (when mem) } aliases := map[string]alias{} + raws := map[string]string{} for _, d := range f.Decls { pre, ok := d.(*ast.Preproc) if !ok { @@ -3520,6 +3933,23 @@ func arm64ResolveAliases(f *ast.File) { strings.HasPrefix(body, "(") || strings.ContainsAny(body, "();") { continue } + raws[name] = body + } + // Alias bodies may name other aliases (hlp1 → res_ptr → R0): substitute + // transitively until nothing changes, bounded against cycles. + for range 8 { + changed := false + for name, body := range raws { + if next, ok := raws[body]; ok && next != body { + raws[name] = next + changed = true + } + } + if !changed { + break + } + } + for name, body := range raws { isReg := func(s string) bool { if arm64RegNum(s) >= 0 { return true @@ -3547,22 +3977,6 @@ func arm64ResolveAliases(f *ast.File) { return } - // replace rewrites whole-word occurrences of the alias names in s. - replace := func(s string) string { - if s == "" { - return s - } - out := strings.Fields(s) - for i, w := range out { - if a, ok := aliases[w]; ok { - out[i] = a.raw - } - } - if len(out) == 0 { - return s - } - return strings.Join(out, " ") - } // replaceToken rewrites an operand whose whole text is one alias use // possibly followed by syntax (POLY.D[0]): the alias must be a prefix // ending at a non-identifier character. @@ -3581,6 +3995,34 @@ func arm64ResolveAliases(f *ast.File) { return s, false } + // replaceScan rewrites alias uses inside a composite operand (a + // parenthesised memory operand or a bracketed register list): every + // identifier run of word and dot characters is matched against the alias + // names, everything else copies verbatim. The whitespace-split replace + // above cannot see "[ACC0.B16" or "(tPtr)", whose members carry their + // punctuation attached. + replaceScan := func(s string) string { + var b strings.Builder + for i := 0; i < len(s); { + if isAliasWordByte(s[i]) || s[i] == '.' { + j := i + for j < len(s) && (isAliasWordByte(s[j]) || s[j] == '.') { + j++ + } + if nn, ok := replaceToken(s[i:j]); ok { + b.WriteString(nn) + } else { + b.WriteString(s[i:j]) + } + i = j + continue + } + b.WriteByte(s[i]) + i++ + } + return b.String() + } + for _, d := range f.Decls { t, ok := d.(*ast.Text) if !ok { @@ -3637,9 +4079,28 @@ func arm64ResolveAliases(f *ast.File) { op.Raw = a.raw continue } - op.Addr.Sym.Name = nn - op.Addr.Sym.Raw = nn - op.Raw = nn + op.Addr.Sym.Name, op.Addr.Sym.Raw = nn, nn + // The span shape depends on what trailed the + // name: an element or arrangement selector + // (POLY.D[0], POLY.B16) rides in Shift and folds + // back onto the rewritten token; a shift + // operator stays in Shift while the span carries + // the bare register; a split list keeps its + // closing bracket, so the rewrite goes through + // the scan. + sfx := strings.Join(strings.Fields(op.Addr.Shift), "") + switch { + case strings.HasPrefix(sfx, ".") || strings.HasPrefix(sfx, "[") || sfx == "]": + // Element or arrangement selectors and the + // closing bracket of a split list belong to + // the token text. + op.Raw = nn + sfx + op.Addr.Shift = "" + case op.Addr.Shift != "": + op.Raw = nn + default: + op.Raw = replaceScan(op.Raw) + } continue } } @@ -3647,7 +4108,7 @@ func arm64ResolveAliases(f *ast.File) { // [V0.B16, V1.B16] with aliased members. if strings.HasPrefix(strings.TrimSpace(op.Raw), "(") || strings.HasPrefix(strings.TrimSpace(op.Raw), "[") { - op.Raw = replace(op.Raw) + op.Raw = replaceScan(op.Raw) } } } diff --git a/asm/arm64_encode.go b/asm/arm64_encode.go index 762cd13..b8b6843 100644 --- a/asm/arm64_encode.go +++ b/asm/arm64_encode.go @@ -308,6 +308,8 @@ const ( a64CondLT = 0xb a64CondGT = 0xc a64CondLE = 0xd + a64CondAL = 0xe + a64CondNV = 0xf ) // arm64CondMap maps Go assembler condition mnemonics to AArch64 condition codes. @@ -328,6 +330,8 @@ var arm64CondMap = map[string]uint32{ "LT": a64CondLT, "GT": a64CondGT, "LE": a64CondLE, + "AL": a64CondAL, + "NV": a64CondNV, } // ---- instruction format tags ---- @@ -335,46 +339,47 @@ var arm64CondMap = map[string]uint32{ type a64Format uint8 const ( - a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc. - a64FMovWide // move wide: MOVZ, MOVN, MOVK - a64FBranch // unconditional branch (B/BL) - a64FBranchCond // conditional branch (B.cond) - a64FUncondBranch // unconditional branch register (BR/BLR/RET) - a64FADR // ADR/ADRP - a64FEXTR // EXTR - a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM - a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source - a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10 - a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. - a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT* - a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc. - a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE - a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE - a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc. - a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL - a64FCRC32 // CRC32 - a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG - a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP - a64FLSE // LSE atomics: LDADD, CAS, SWP - a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS - a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms - a64FCondCmp // conditional compare: CCMP, CCMN - a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms - a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms - a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD - a64FAcqRel // acquire/release: LDAR family, STLR family - a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM - a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ... - a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ... - a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ... - a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd - a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV - a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT - a64FVTBL // SIMD table lookup: VTBL - a64FDUP // SIMD element moves: VDUP, VMOV with element indices - a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R - a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI - a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool) + a64FDPSR a64Format = iota // data-processing (shifted register): ADD, SUB, AND, ORR, EOR, etc. + a64FMovWide // move wide: MOVZ, MOVN, MOVK + a64FBranch // unconditional branch (B/BL) + a64FBranchCond // conditional branch (B.cond) + a64FUncondBranch // unconditional branch register (BR/BLR/RET) + a64FADR // ADR/ADRP + a64FEXTR // EXTR + a64FBitfield // bitfield: BFI/BFXIL/SBFM/UBFM/BFM + a64FBitfieldAlias // bitfield alias: BFI/BFXIL/SBFIZ/UBFIZ, ($lsb, Rn, $width, Rd) + a64FShift // shifts: LSL/LSR/ASR alias SBFM/UBFM, ROR aliases EXTR; register forms are two-source + a64FDPR4 // data-processing 4-register: MADD/MSUB, Ra in bits 14:10 + a64FFP3 // FP 3-operand (Rm, Rn, Rd): FADD, FSUB, FMUL, FDIV, etc. + a64FFPUnary // FP unary (Rn, Rd): FMOV, FABS, FNEG, FSQRT, FCVT, FRINT* + a64FFP4 // FP 4-operand FMA (Ra, Rm, Rn, Rd): FMADD, FMSUB, etc. + a64FFPCmp // FP compare (Rm, Rn): FCMP, FCMPE + a64FFPCCmp // FP conditional compare (Rm, Rn, nzcv, cond): FCCMP, FCCMPE + a64FFPCvt // FP↔integer conversion: FCVTZS, SCVTF, etc. + a64FFPSel // FP conditional select (Rm, Rn, Rd, cond): FCSEL + a64FCRC32 // CRC32 + a64FCSEL // conditional select: CSEL, CSINC, CSINV, CSNEG + a64FExcl // exclusive load/store: LDXR, STXR, LDAXR, STLXR and pair forms LDXP, STXP + a64FLSE // LSE atomics: LDADD, CAS, SWP + a64FDP1 // data-processing (1 source): RBIT, REV, CLZ, CLS + a64FBitfield2 // bitfield extract: UBFX, SBFX and the W forms + a64FCondCmp // conditional compare: CCMP, CCMN + a64FBranch19 // compare-and-branch: CBZ, CBNZ and the W forms + a64FTestBranch // test-and-branch: TBZ, TBNZ and the W forms + a64FPair // load/store pair: LDP, STP, LDPW, STPW, FLDPD, FSTPD + a64FAcqRel // acquire/release: LDAR family, STLR family + a64FSys // system: BRK, SVC, DMB, DSB, ISB, DC, MRS, MSR, PRFM + a64FCrypto2 // crypto 2-register: AESD, AESE, AESIMC, AESMC, SHA1H, ... + a64FCrypto3 // crypto 3-register: SHA1C, SHA256H, SHA512SU1, ... + a64FSIMDV // SIMD 3-register with arrangement: VADD, VAND, VCMEQ, VZIP1, ... + a64FSIMDVZero // SIMD compare against zero: VCMEQ $0, Vn, Vd + a64FSIMDV2 // SIMD 2-register with arrangement: VREV32, VREV64, VUADDLV, VMOV + a64FSIMDV4 // SIMD 4-register / imm 3-register: VEOR3, VBCAX, VXAR, VEXT + a64FVTBL // SIMD table lookup: VTBL + a64FDUP // SIMD element moves: VDUP, VMOV with element indices + a64FVLDST // SIMD structure loads/stores: VLD1, VST1, VLD1R, VLD4R + a64FShiftImm // SIMD shift by immediate: VSHL, VUSHR, VSRI + a64FMoviLit // VMOVS/VMOVD/VMOVQ with a large constant (literal pool) ) // a64Enc is one instruction's encoding: its bit layout (format) and the @@ -487,6 +492,16 @@ func init() { a64InstrTable["MADDW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24} a64InstrTable["MSUB"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<15} a64InstrTable["MSUBW"] = a64Enc{format: a64FDPR4, op: 0<<31 | 0x1b<<24 | 1<<15} + // The widening multiplies: a 64-bit result riding the same layout, the + // three-operand forms reading the accumulate register as ZR. + a64InstrTable["SMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21} + a64InstrTable["UMADDL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23} + a64InstrTable["SMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15} + a64InstrTable["UMSUBL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15} + a64InstrTable["SMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 31<<10} + a64InstrTable["UMULL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 31<<10} + a64InstrTable["SMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<15 | 31<<10} + a64InstrTable["UMNEGL"] = a64Enc{format: a64FDPR4, op: 1<<31 | 0x1b<<24 | 1<<21 | 1<<23 | 1<<15 | 31<<10} // ---- move wide ---- // MOVZ/MOVN/MOVK @@ -536,6 +551,15 @@ func init() { // ---- bitfield ---- a64InstrTable["BFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} a64InstrTable["BFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 1<<29 | 0x26<<23 | 0<<22} + // The four-operand bitfield aliases: ($lsb, Rn, $width, Rd). + a64InstrTable["BFI"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} + a64InstrTable["BFIW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23} + a64InstrTable["BFXIL"] = a64Enc{format: a64FBitfieldAlias, op: 1<<31 | 1<<29 | 0x26<<23 | 1<<22} + a64InstrTable["BFXILW"] = a64Enc{format: a64FBitfieldAlias, op: 0<<31 | 1<<29 | 0x26<<23} + a64InstrTable["SBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x93400000} + a64InstrTable["SBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x13000000} + a64InstrTable["UBFIZ"] = a64Enc{format: a64FBitfieldAlias, op: 0x53000000} + a64InstrTable["UBFIZW"] = a64Enc{format: a64FBitfieldAlias, op: 0x33000000} a64InstrTable["SBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 0<<29 | 0x26<<23 | 1<<22} a64InstrTable["SBFMW"] = a64Enc{format: a64FBitfield, op: 0<<31 | 0<<29 | 0x26<<23 | 0<<22} a64InstrTable["UBFM"] = a64Enc{format: a64FBitfield, op: 1<<31 | 2<<29 | 0x26<<23 | 1<<22} @@ -716,6 +740,13 @@ func init() { "RBIT": 0xdac00000, "REV16": 0xdac00400, "REV32": 0xdac00800, "REV": 0xdac00c00, "CLZ": 0xdac01000, "CLS": 0xdac01400, "RBITW": 0x5ac00000, "REVW": 0x5ac00800, "CLZW": 0x5ac01000, "CLSW": 0x5ac01400, + // Extend and byte-reverse: the UBFM/SBFM aliases with imms fixing + // the source width. + "SXTB": 0x93401c00, "SXTBW": 0x13001c00, "SXTH": 0x93403c00, + "SXTHW": 0x13003c00, "SXTW": 0x93407c00, + "UXTB": 0x53001c00, "UXTBW": 0x53001c00, "UXTH": 0x53403c00, + "UXTHW": 0x53003c00, "UXTW": 0x53407c00, + "REV16W": 0x5ac00400, } for m, op := range dp1 { a64InstrTable[m] = a64Enc{format: a64FDP1, op: op} @@ -734,7 +765,7 @@ func init() { a64InstrTable["CCMNW"] = a64Enc{format: a64FCondCmp, op: 0x3a400000} // ---- system operations ---- - for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "DC", "MRS", "MSR", "PRFM"} { + for _, m := range []string{"BRK", "SVC", "DMB", "DSB", "ISB", "CLREX", "HINT", "BTI", "HLT", "SMC", "HVC", "DCPS1", "DCPS2", "DCPS3", "DRPS", "ERET", "AUTIASP", "AUTIBSP", "AUTIA1716", "AUTIB1716", "SEVL", "SEV", "WFE", "WFI", "YIELD", "DC", "MRS", "MSR", "PRFM"} { a64InstrTable[m] = a64Enc{format: a64FSys} } @@ -785,6 +816,73 @@ func init() { for m, op := range lse { a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} } + // The remaining width and ordering spellings of the same shapes, and the + // CAS compare-and-swap family, word-verified against go tool asm. + lseMore := map[string]uint32{ + "LDADDAB": 0x38a00000, + "LDADDAH": 0x78a00000, + "LDADDALB": 0x38e00000, + "LDADDALH": 0x78e00000, + "LDADDLB": 0x38600000, + "LDADDLD": 0xf8600000, + "LDADDLH": 0x78600000, + "LDADDLW": 0xb8600000, + "LDCLRAB": 0x38a01000, + "LDCLRAH": 0x78a01000, + "LDCLRALH": 0x78e01000, + "LDCLRB": 0x38201000, + "LDCLRD": 0xf8201000, + "LDCLRH": 0x78201000, + "LDCLRLB": 0x38601000, + "LDCLRLD": 0xf8601000, + "LDCLRLH": 0x78601000, + "LDCLRLW": 0xb8601000, + "LDCLRW": 0xb8201000, + "LDEORAB": 0x38a02000, + "LDEORAD": 0xf8a02000, + "LDEORAH": 0x78a02000, + "LDEORALB": 0x38e02000, + "LDEORALH": 0x78e02000, + "LDEORAW": 0xb8a02000, + "LDEORB": 0x38202000, + "LDEORD": 0xf8202000, + "LDEORH": 0x78202000, + "LDEORLB": 0x38602000, + "LDEORLD": 0xf8602000, + "LDEORLH": 0x78602000, + "LDEORLW": 0xb8602000, + "LDEORW": 0xb8202000, + "LDORAB": 0x38a03000, + "LDORAD": 0xf8a03000, + "LDORAH": 0x78a03000, + "LDORALH": 0x78e03000, + "LDORAW": 0xb8a03000, + "LDORB": 0x38203000, + "LDORD": 0xf8203000, + "LDORH": 0x78203000, + "LDORLB": 0x38603000, + "LDORLD": 0xf8603000, + "LDORLH": 0x78603000, + "LDORLW": 0xb8603000, + "LDORW": 0xb8203000, + "SWPAB": 0x38a08000, + "SWPAD": 0xf8a08000, + "SWPAH": 0x78a08000, + "SWPALH": 0x78e08000, + "SWPAW": 0xb8a08000, + "SWPB": 0x38208000, + "SWPH": 0x78208000, + "SWPLB": 0x38608000, + "SWPLD": 0xf8608000, + "SWPLH": 0x78608000, + "SWPLW": 0xb8608000, + "CASAD": 0xc8e07c00, + "CASALB": 0x08e0fc00, + "CASLW": 0x88a0fc00, + } + for m, op := range lseMore { + a64InstrTable[m] = a64Enc{format: a64FLSE, op: op} + } // ---- carry-setting/carry-using arithmetic and widening multiply ---- // MUL and SMULH/UMULH are the MADD/MSUB layout with the accumulate @@ -794,7 +892,12 @@ func init() { "ADCS": 0xba000000, "ADCSW": 0x3a000000, "SBC": 0xda000000, "SBCW": 0x5a000000, "SBCS": 0xfa000000, "SBCSW": 0x7a000000, - "MUL": 0x9b007c00, "MULW": 0x1b007c00, + // MNEG/MSUB and NGC/SBC with the complementing register preset to ZR. + "MNEG": 0x9b00fc00, "MNEGW": 0x1b00fc00, + "NGC": 0xda000000, "NGCW": 0x5a000000, + "NGCS": 0xfa000000, "NGCSW": 0x7a000000, + "NEGSW": 0x6b000000, + "MUL": 0x9b007c00, "MULW": 0x1b007c00, "SMULH": 0x9b407c00, "UMULH": 0x9bc07c00, } for m, op := range dpsrExtra { @@ -835,6 +938,12 @@ func init() { a64InstrTable["VSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 21<<10} a64InstrTable["VUSHR"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 1<<10} a64InstrTable["VSRI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 17<<10} + a64InstrTable["VSSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 1<<10} + a64InstrTable["VSRA"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 17<<10} + a64InstrTable["VSRSHR"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 9<<10} + a64InstrTable["VSLI"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 21<<10} + a64InstrTable["VSQSHL"] = a64Enc{format: a64FShiftImm, op: 0x0f000000 | 29<<10} + a64InstrTable["VUQSHL"] = a64Enc{format: a64FShiftImm, op: 0x2f000000 | 29<<10} a64InstrTable["VLD1"] = a64Enc{format: a64FVLDST} a64InstrTable["VLD1.P"] = a64Enc{format: a64FVLDST, op: 1} a64InstrTable["VST1"] = a64Enc{format: a64FVLDST} @@ -899,6 +1008,23 @@ func a64ElemLetter(s string) bool { return false } +// fpSimdArrs and fpAcrossArrs bound the arrangements the FP SIMD forms +// accept: H, S and D widths for the pairwise data-processing, H and S for +// the across-vector reductions. +var fpSimdArrs = uint16(1<; the source register rides bits 4:0. var a64MSRRegOps = map[string]uint32{ "NZCV": 0xd51b4200, "FPCR": 0xd51b4400, "FPSR": 0xd51b4420, + "ELR_EL1": 0xd5184020, } // a64MSROps maps the system register names GOROOT writes to their fixed diff --git a/asm/arm64_encode_test.go b/asm/arm64_encode_test.go index 3d0fc86..8b25270 100644 --- a/asm/arm64_encode_test.go +++ b/asm/arm64_encode_test.go @@ -4,6 +4,7 @@ package asm import ( + "strings" "testing" "sourcedock.dev/petrbalvin/gasm-devkit/ast" @@ -1295,20 +1296,170 @@ func TestArm64ExclNoOffset(t *testing.T) { } } -// TestArm64AddSubImmRange: immediates that cannot ride the imm12 field are -// rejected instead of wrapping through int32. -func TestArm64AddSubImmRange(t *testing.T) { - for _, body := range []string{ - "\tADD $0x100000000, R0, R1\n", - "\tSUB $-0x100000000, R0, R1\n", - "\tCMP $0x100000000, R0\n", - } { - f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n"+body+"\tRET\n") - if len(errs) > 0 { - t.Fatalf("parse: %v", errs) +// TestArm64AddSubImmWide pins the wide-immediate classification the toolchain +// applies to the ADD/SUB family (asm7.go cases 48, 62, 13): the ADDCON2 split +// into two imm12 instructions for plain ADD/SUB, the bitmask ORR into REGTMP, +// and the MOVZ/MOVN/MOVK materialisations followed by the register form. +// Comparisons never split, and the W forms classify the 32-bit value. Every +// word is go tool asm's own for the same source. +func TestArm64AddSubImmWide(t *testing.T) { + got := arm64Words(t, strings.Join([]string{ + "\tADD $0xaaaaaa, R2, R3", + "\tSUB $0xaaaaaa, R2", + "\tADD $0x186a0, R2, R5", + "\tADD $0x1ffe00, R2, R3", + "\tADD $0x3fffffffc000, R5", + "\tADD $-100000, R2, R3", + "\tADD $-2048, R2, R3", + "\tCMP $0xaaaaaa, R2", + "\tCMP $0xffffffffffa0, R3", + "\tCMPW $27745, R2", + "\tCMPW $0x60060, R2", + "\tADDS $0xaaaaaa, R2, R3", + "\tADD $0x12345678, R2, R3", + "\tADDW $0x60060, R2", + "\tSUB $0xe7791f700, R3, R1", + "\tADDW $0x12345678, R2, R3", + "\tCMN $0x1000000, R2", + }, "\n")+"\n") + want := []uint32{ + 0x912aa843, 0x916aa863, // ADD $0xaaaaaa, R2, R3: ADDCON2 split + 0xd12aa842, 0xd16aa842, // SUB $0xaaaaaa, R2: split with Rd = Rn + 0x911a8045, 0x914060a5, // ADD $0x186a0, R2, R5: split + 0xb2772ffb, 0x8b1b0043, // ADD $0x1ffe00: bitmask beats the split + 0xb2727ffb, 0x8b1b00a5, // ADD $0x3fffffffc000: bitmask into REGTMP + 0x9290d3fb, 0xf2bfffdb, 0x8b1b0043, // ADD $-100000: MOVN + MOVK + 0x9280fffb, 0x8b1b0043, // ADD $-2048: single MOVN + ADD + 0xd295555b, 0xf2a0155b, 0xeb1b005f, // CMP: never split, MOVZ + MOVK + 0x92800bfb, 0xf2e0001b, 0xeb1b007f, // CMP $0xffffffffffa0: MOVN + fixup + 0x528d8c3b, 0x6b1b005f, // CMPW $27745: W movcon, single MOVZW + 0x52800c1b, 0x72a000db, 0x6b1b005f, // CMPW $0x60060: S form skips the split + 0xd295555b, 0xf2a0155b, 0xab1b0043, // ADDS $0xaaaaaa: MOVZ + MOVK + ADDS + 0xd28acf1b, 0xf2a2469b, 0x8b1b0043, // ADD $0x12345678: MOVZ + MOVK + 0x11018042, 0x11418042, // ADDW $0x60060: W split + 0xd29ee01b, 0xf2aef23b, 0xf2c001db, 0xcb1b0061, // SUB $0xe7791f700 + 0x528acf1b, 0x72a2469b, 0x0b1b0043, // ADDW $0x12345678: MOVZW + MOVKW + 0xd2a0201b, 0xab1b005f, // CMN $0x1000000: single MOVZ + CMN + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("wide word %d = %08x, want %08x", i, got[i], want[i]) } - if _, err := AssembleFileARM64(f); err == nil { - t.Errorf("%s: expected an error, got none", body) + } +} + +// TestArm64CarryImmWide pins the carry family's $0 spellings in two and +// three operands, the ROR shift on the logical group (and its rejection for +// the arithmetic forms), the NGC/MNEG zero-register aliases and the vector +// alias with an element selector. Words are go tool asm's own. +func TestArm64CarryShiftAlias(t *testing.T) { + got := arm64Words(t, "\tADC $0, R20\n\tADC $0, R20, R4\n\tSBCS $0, R4, R12\n"+ + "\tSBCS R15, R4, R12\n\tANDW R9@>7, R19, R26\n\tAND R1@>33, R2, R3\n"+ + "\tNEGSW R23<<1, R30\n\tNGC R2, R7\n\tMNEG R14, R27, R23\n") + want := []uint32{ + 0x9a1f0294, // ADC ZR, R20, R20 + 0x9a1f0284, // ADC ZR, R20, R4 + 0xfa1f008c, // SBCS ZR, R4, R12 + 0xfa0f008c, // SBCS R15, R4, R12 + 0x0ac91e7a, // ANDW R9 ROR 7, R19, R26 + 0x8ac18443, // AND R1 ROR 33, R2, R3 + 0x6b1707fe, // SUBSW ZR, R30, R23 LSL 1 + 0xda0203e7, // SBC ZR, R7, R2 + 0x9b0eff77, // MSUB ZR, R27, R14, R23 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("carry word %d = %08x, want %08x", i, got[i], want[i]) + } + } + + // ROR on an arithmetic form is unallocated: the toolchain reports an + // unsupported shift operator. + f, errs := parser.Parse("test_arm64.s", "#include \"textflag.h\"\n\nTEXT ·f(SB), NOSPLIT, $0-0\n\tADD R1@>33, R2, R3\n\tRET\n") + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + if _, err := AssembleFileARM64(f); err == nil { + t.Error("ADD R1@>33: expected an error, got none") + } +} + +// TestArm64VecAliasElement pins the register-alias rewrite inside a vector +// operand with an element selector and inside a split register list: the +// aliases resolve textually where the parser carries the selector apart from +// the name. Words are go tool asm's own. +func TestArm64VecAliasElement(t *testing.T) { + src := `#include "textflag.h" + +#define POLY V15 +#define ACC0 V8 +#define ACC1 V9 + +TEXT ·f(SB), NOSPLIT, $0-0 + VMOV R1, POLY.D[0] + VEOR POLY.B16, POLY.B16, POLY.B16 + VLD1 (R0), [ACC0.B16] + VLD1.P (R0), [ACC0.B16, ACC1.B16] + VST1.P [ACC0.B16, ACC1.B16], 32(R1) + RET +` + f, errs := parser.Parse("test_arm64.s", src) + if len(errs) > 0 { + t.Fatalf("parse: %v", errs) + } + img, err := AssembleFileARM64(f) + if err != nil { + t.Fatalf("AssembleFileARM64: %v", err) + } + got := leWords(img.Code) + want := []uint32{ + 0x4e081c2f, // INS V15.D[0], R1 + 0x6e2f1def, // VEOR V15.B16, V15.B16, V15.B16 + 0x4c407008, // VLD1 (R0), [V8.B16] + 0x4cdfa008, // VLD1.P (R0), [V8.B16, V9.B16] + 0x4c9fa028, // VST1.P [V8.B16, V9.B16], 32(R1) + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("vecalias word %d = %08x, want %08x", i, got[i], want[i]) + } + } +} + +// TestArm64AddSubImmBeyond32 pins the materialisation the toolchain applies +// once the value leaves every imm12 form: a constant sequence into REGTMP +// (R27) followed by the register form. SUB $-0x100000000 is a bitmask +// immediate, so it rides the ORR form; the others take MOVZ. Words are go +// tool asm's own. +func TestArm64AddSubImmBeyond32(t *testing.T) { + got := arm64Words(t, "\tADD $0x100000000, R0, R1\n\tSUB $-0x100000000, R0, R1\n\tCMP $0x100000000, R0\n") + want := []uint32{ + 0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2) + 0x8b1b0001, // ADD R27, R0, R1 + 0xb2607ffb, // ORR $-4294967296, ZR, R27 (bitmask) + 0xcb1b0001, // SUB R27, R0, R1 + 0xd2c0003b, // MOVZ $(1<<32>>16), R27 (hw=2) + 0xeb1b001f, // CMP R27, R0 + 0xd65f03c0, // RET + } + if len(got) != len(want) { + t.Fatalf("word count = %d, want %d", len(got), len(want)) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("word %d = %08x, want %08x", i, got[i], want[i]) } } } diff --git a/testdata/verify/carryshift_arm64.s b/testdata/verify/carryshift_arm64.s new file mode 100644 index 0000000..32f1973 --- /dev/null +++ b/testdata/verify/carryshift_arm64.s @@ -0,0 +1,48 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Carry arithmetic, logical shifts, register aliases with element selectors +// and the ADC/SBC immediate spellings: the shapes nat_arm64.s, p256 and +// gcm_arm64.s exercise. Byte-for-byte against go tool asm. + +#include "textflag.h" + +#define acc0 V8 +#define acc1 V9 +#define const0 R15 +#define POLY V15 + +// carry pins the ADC/SBC family: the $0 spellings in two and three +// operands, and the register-carry forms. +TEXT ·carry(SB), NOSPLIT, $0-0 + ADC $0, R20 + ADC $0, R20, R4 + SBCS $0, R4 + SBCS $0, R4, R12 + SBCS R15, R4, R12 + SBC $0, R1 + ADCSW $0, R2, R3 + RET + +// shift pins the shifted-register forms including ROR, which only the +// logical family accepts. +TEXT ·shift(SB), NOSPLIT, $0-0 + ANDW R9@>7, R19, R26 + AND R1@>33, R2, R3 + ADD R1<<11, R2, R3 + SUB R1->33, R2 + ORR R5<<2, R6, R7 + RET + +// vecalias pins the vector aliases with element selectors and the +// structure loads with aliased members. +TEXT ·vecalias(SB), NOSPLIT, $0-0 + MOVD $0xC2, R1 + VMOV R1, POLY.D[0] + VMOV R0, POLY.D[1] + VEOR POLY.B16, POLY.B16, POLY.B16 + VLD1 (R0), [acc0.B16] + VLD1.P (R0), [acc0.B16, acc1.B16] + VST1 [acc0.B16, acc1.B16], (R1) + VST1.P [acc0.B16, acc1.B16], 32(R1) + RET diff --git a/testdata/verify/widenimm_arm64.s b/testdata/verify/widenimm_arm64.s new file mode 100644 index 0000000..0ac9a36 --- /dev/null +++ b/testdata/verify/widenimm_arm64.s @@ -0,0 +1,70 @@ +// Copyright (c) 2026 Petr Balvín (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Wide-immediate arithmetic: every classification band of the ADD/SUB +// immediate family (single imm12, the ADDCON2 split, bitmask and MOVZ/MOVN/ +// MOVK materialisations into REGTMP) plus the logical bitmask immediates and +// their materialised fallback. Byte-for-byte against go tool asm. + +#include "textflag.h" + +// imm12 covers the plain and shifted-by-12 imm12 forms. +TEXT ·imm12(SB), NOSPLIT, $0-0 + ADD $1, R2, R3 + ADD $0x000aaa, R2, R3 + ADD $0xaaa000, R2 + SUB $0x000aaa, R2, R3 + SUB $0xaaa000, R2 + ADDW $40960, R0 + CMP $40960, R0 + CMPW $40960, R0 + RET + +// split pins the ADDCON2 band: two imm12 instructions, low half first. +TEXT ·split(SB), NOSPLIT, $0-0 + ADD $0xaaaaaa, R2, R3 + SUB $0xaaaaaa, R2 + ADD $0x186a0, R2, R5 + SUB $0x186a0, R2, R3 + ADDW $0x60060, R2 + RET + +// regtmp covers the single-word materialisations: MOVZ for a movcon value, +// MOVN for the complement form, the bitmask ORR otherwise. +TEXT ·regtmp(SB), NOSPLIT, $0-0 + ADD $0x1ffe00, R2, R3 + ADD $0x3fffffffc000, R5 + ADD $-2048, R2, R3 + ADD $-100000, R2, R3 + CMP $0x1000000, R2 + CMP $0x100000000, R0 + SUB $-0x100000000, R0, R1 + RET + +// movseq covers the omovlconst sequences: MOVZ/MOVN ladders and the +// compare forms that never split. +TEXT ·movseq(SB), NOSPLIT, $0-0 + ADD $0x12345678, R2, R3 + SUB $0xe7791f700, R3, R1 + CMP $0xaaaaaa, R2 + CMP $0xffffffffffa0, R3 + CMPW $27745, R2 + CMPW $0x60060, R2 + ADDS $0xaaaaaa, R2, R3 + CMN $0x1000000, R2 + ADDW $0x12345678, R2, R3 + RET + +// logical covers the bitmask immediates of the logical family and the +// materialised fallback for the values a bitmask cannot carry. +TEXT ·logical(SB), NOSPLIT, $0-0 + AND $0x3ff00000, R2, R3 + BIC $0x22220000, R3, R4 + ORR $0x3ff00000, R2 + EOR $0x3ff00000, R2, R3 + ANDS $0x3ff00000, R2 + ORNW $0x3ff00000, R2 + EONW $0x3ff00000, R2 + BICSW $0x6006000060060, R5 + TST $0x4900000049, R0 + RET