From 71e8dd550dfdf3cec5cd5132df7801d6a87a02af Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Petr=20Balv=C3=ADn?= Date: Tue, 6 Oct 2026 20:41:36 +0200 Subject: [PATCH] feat(asm): split wide arm64 load and store offsets into REGTMP MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Offsets the single-instruction forms cannot carry lower the way the toolchain lowers them: ADD or SUB moves the whole distance into REGTMP within the ±4095 band, and the 24-bit band above it splits into an ADD of the high half and an access of the low half, with the pair family taking the two-ADD sequence. The split band follows loadStoreClass per width, byte accesses taking the full 24 bits and the Q width the widest, so an offset the toolchain pools is never split instead. Assisted-by: GLM 5.3 Flash --- asm/arm64_assemble.go | 233 +++++++++++++++++++++++++++------ asm/arm64_encode_test.go | 34 +++++ testdata/verify/splits_arm64.s | 57 ++++++++ 3 files changed, 282 insertions(+), 42 deletions(-) create mode 100644 testdata/verify/splits_arm64.s diff --git a/asm/arm64_assemble.go b/asm/arm64_assemble.go index c2ae0b0..bec6dd2 100644 --- a/asm/arm64_assemble.go +++ b/asm/arm64_assemble.go @@ -265,6 +265,12 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int { return 4 } } + // The load/store pair family: one word in range, otherwise the REGTMP + // expansion the toolchain lowers to (two words within ±4095, three + // beyond, the pool words riding the function end). + if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair { + return arm64PairSize(mnem, ops, fi) + } // The funcdata pseudo-statements contribute no bytes, the expanded // FUNCDATA/PCDATA forms included. switch mnem { @@ -1599,6 +1605,38 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo, return encodeARM64RegMove(mnem, src, dst) } +// arm64PairSize returns the encoded size of a load/store pair instruction, +// mirroring the encoder's branch order exactly: the label offsets of pass 1 +// must match the bytes pass 2 lays down. +func arm64PairSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { + if len(ops) != 2 { + return 4 + } + load := strings.Contains(mnem, "LDP") + memOp := ops[0] + if !load { + memOp = ops[1] + } + if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" { + return 12 // ADRP + ADD + pair + } + _, off := arm64MemWithFrame(memOp, fi) + if off < 0 { + return 4 // the encoder rejects it in pass 2 + } + scale := arm64PairScale(mnem) + if off%scale == 0 && off >= -64*scale && off <= 63*scale { + return 4 + } + if off >= -4095 && off <= 4095 { + return 8 // ADD/SUB the whole offset into REGTMP + pair + } + if off >= 0 && off <= 0xffffff { + return 12 // the two-ADD split + } + return 12 // the pool sequence: LDR literal + ADD + pair +} + // arm64MovSize returns the encoded size of a MOV instruction. func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if len(ops) != 2 { @@ -1651,10 +1689,15 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int { if off >= -256 && off <= 255 { return 4 // unscaled } - if _, _, _, ok := arm64SplitOffset(off, scale); ok { - return 8 // ADD base, REGTMP + access + if off >= -4095 && off <= 4095 { + return 8 // ADD/SUB the whole offset into REGTMP } - return 12 // literal pool range: encoding reports it as unsupported + if arm64OffsetSplitReach(off, lt) { + if _, _, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale))); ok { + return 8 // ADD base, REGTMP + access + } + } + return 12 // the pool sequence: LDR literal + register-offset access default: return 4 // register move } @@ -1929,40 +1972,72 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6 if off >= -256 && off <= 255 { return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil } + // Offsets within ±4095 that no single-instruction form carries ride the + // toolchain's ADD/SUB fallback: the whole offset moves into REGTMP and + // the access reads it back at zero (asm7.go cases 30/31). + if off >= -4095 && off <= 4095 { + op, v := uint32(0), off // ADD + if v < 0 { + op, v = 1, -v // SUB + } + return a64WordsLE( + a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27 + a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), 0, 27, uint32(reg)), + ), nil + } // Large offset: materialise the base in REGTMP (R27) the way the - // toolchain does and access what remains. The ADD offsets from the - // operand's own base register, [SP] and [Rn] alike. - addImm, addShift, access, ok := arm64SplitOffset(off, scale) + // toolchain does and access what remains (asm7.go cases 30/31). The + // ADD offsets from the operand's own base register, [SP] and [Rn] + // alike; beyond the split band the offset reaches the literal pool. + if !arm64OffsetSplitReach(off, lt) { + return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) + } + hi, lo, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale))) if !ok { return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) } return a64WordsLE( - a64AddSub(1, 0, 0, addShift, addImm, uint32(rn), 27), // ADD $addImm<= -4095 && off <= 4095 { + return true + } if off < 0 { - return 0, 0, 0, false + return false } - // Plain ADD: bring the base to within the largest scaled access. - l := min(off, 4095*scale) - l -= l % scale - if a := off - l; a <= 4095 { - return uint32(a), 0, l, true + if lt.size == 0 && lt.V == 1 { // the Q width + return off <= 0xfff000+0xfff<<4 && off&15 == 0 || off <= 0xfff+0xfff<<4 } - // Shifted ADD: cover everything but the bits the access imm12 carries. - rest := off &^ (0xFFF * scale) - if rest >= 0 && rest>>12 <= 4095 { - return uint32(rest >> 12), 1, off - rest, true + switch lt.size { + case 0: // byte: the full 24-bit band, alignment trivial + return off <= 0xffffff + case 1: + return off <= 0xfff000+0xfff<<1 && off&1 == 0 || off <= 0xfff+0xfff<<1 + case 2: + return off <= 0xfff000+0xfff<<2 && off&3 == 0 || off <= 0xfff+0xfff<<2 + default: + return off <= 0xfff000+0xfff<<3 && off&7 == 0 || off <= 0xfff+0xfff<<3 } - return 0, 0, 0, false +} + +// arm64AddImmWord encodes ADD $v, Rn, Rd the way the toolchain's oaddi +// does: a non-zero multiple of 0x1000 encodes shifted left by twelve. +func arm64AddImmWord(v int64, rn, rd uint32) uint32 { + sh := arm64AddShift(v) + if sh == 1 { + v >>= 12 + } + return a64AddSub(1, 0, 0, sh, uint32(v), rn, rd) } // ---- static symbol references (ADRP + offset) ---- @@ -3146,13 +3221,7 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra if len(ops) != 2 { return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops)) } - scale := int64(8) - switch { - case strings.HasSuffix(mnem, "W"): - scale = 4 // the 32-bit pairs, signed and unsigned - case strings.HasSuffix(mnem, "Q"): - scale = 16 // the 128-bit FP pairs - } + scale := arm64PairScale(mnem) load := strings.Contains(mnem, "LDP") memOp, pairOp := ops[0], ops[1] if !load { @@ -3193,19 +3262,99 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" { return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem) } - if off%scale != 0 || off < -64*scale || off > 63*scale { + if off%scale == 0 && off >= -64*scale && off <= 63*scale { + imm7 := off / scale + // The signed-offset form carries bits 24:23 = 10; post-index drops + // bit 24 and pre-index sets both, with the base carrying the opc, V + // and L halves. + switch wb { + case "P": + baseOp = baseOp&^(1<<24) | 1<<23 + case "W": + baseOp |= 1 << 23 + } + return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil + } + if wb != "" { return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale) } - imm7 := off / scale - // The signed-offset form carries bits 24:23 = 10; post-index drops bit 24 - // and pre-index sets both, with the base carrying the opc, V and L halves. - switch wb { - case "P": - baseOp = baseOp&^(1<<24) | 1<<23 - case "W": - baseOp |= 1 << 23 + // Offsets within ±4095 the imm7 field cannot carry move the whole + // distance into REGTMP first (asm7.go cases 74/76: add/sub + ldp/stp). + if off >= -4095 && off <= 4095 { + op, v := uint32(0), off // ADD + if v < 0 { + op, v = 1, -v // SUB + } + return a64WordsLE( + a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27 + baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…) + ), nil } - return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil + // Positive offsets up to 16 MiB split into two ADDs: the low imm12 bits + // from the base register into REGTMP, the high multiple of 0x1000 on top + // (asm7.go cases 75/77). Beyond the band the offset reaches the pool. + if off < 0 || off > 0xffffff { + return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) + } + hi, lo, ok := arm64SplitImm24(off, 0) + if !ok { + return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off) + } + return a64WordsLE( + arm64AddImmWord(lo, uint32(rn), 27), // ADD $lo, Rn, R27 + arm64AddImmWord(hi, 27, 27), // ADD $hi[<<12], R27, R27 + baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…) + ), nil +} + +// arm64PairScale returns the byte width a pair access's imm7 offset divides +// by: four for the 32-bit pairs, sixteen for the 128-bit FP pairs, eight for +// everything else. +func arm64PairScale(mnem string) int64 { + switch { + case strings.HasSuffix(mnem, "W"): + return 4 // the 32-bit pairs, signed and unsigned + case strings.HasSuffix(mnem, "Q"): + return 16 // the 128-bit FP pairs + } + return 8 +} + +// arm64AddShift returns the ADD-immediate shift bit: a non-zero value whose +// low twelve bits are zero encodes shifted left by twelve. +func arm64AddShift(v int64) uint32 { + if v != 0 && v&0xfff == 0 { + return 1 + } + return 0 +} + +// arm64SplitImm24 decomposes an offset the way the toolchain's +// splitImm24uScaled does for the two-ADD sequences: hi either fits imm12 +// alone or is a multiple of 0x1000 up to 0xfff000, and the low half always +// fits imm12 after division by the access scale. ok is false when the value +// is negative or beyond the 24-bit reach, where the toolchain pools. +func arm64SplitImm24(v int64, shift int) (hi, lo int64, ok bool) { + if v < 0 || v > 0xfff000+0xfff<> uint(shift) + if l <= 0xfff { + return h, l, true + } + } + l := (v >> uint(shift)) & 0xfff + h = v - l< 0xfff000 { + h = 0xfff000 + l = (v - h) >> uint(shift) + } + if h&^0xfff000 == 0 && h+l< (https://petrbalvin.org) +// SPDX-License-Identifier: BSD-3-Clause + +// Differential kernel for the arm64 out-of-band offsets: the ADD/SUB into +// REGTMP for offsets within ±4095 no single instruction carries, and the +// 24-bit hi/lo split for the wide bands. Every function is byte-compared +// against go tool asm. + +#include "textflag.h" + +// func bytes() +TEXT ·bytes(SB), NOSPLIT, $0-0 + MOVB R1, 4096(R2) + MOVB R1, 8190(R2) + MOVB R1, 65520(R2) + MOVB R1, -4095(R2) + MOVB 4096(R2), R1 + RET + +// func halves() +TEXT ·halves(SB), NOSPLIT, $0-0 + MOVH R1, 16384(R2) + MOVH R1, 16781310(R2) + MOVH 0x2002(R1), R2 + MOVH R1, -4095(R2) + RET + +// func words() +TEXT ·words(SB), NOSPLIT, $0-0 + MOVW R1, 16388(R2) + MOVW 0x4004(R1), R2 + MOVD R1, 4094(R2) + MOVD R1, -300(R2) + MOVD R1, 0x1006ff8(R2) + MOVD 0x1006ff8(R1), R2 + FMOVS F1, 0x4004(R2) + FMOVD F1, 0x1006ff8(R1) + FMOVQ F1, 0x1003000(R2) + RET + +// func pairs() +TEXT ·pairs(SB), NOSPLIT, $0-0 + STP (R3, R4), 11(R0) + STP (R3, R4), 1024(R0) + STP (R3, R4), -4(R5) + STP (R3, R4), 65536(R2) + STPW (R3, R4), 1024(R0) + LDP 11(R0), (R1, R2) + LDP 1024(R0), (R1, R2) + LDP -31(R0), (R1, R2) + LDP 65536(R2), (R1, R2) + LDPW -5(R0), (R1, R2) + LDPSW 1024(R0), (R1, R2) + FLDPD 1024(R0), (F1, F2) + FLDPQ 1024(R0), (F1, F2) + FLDPS 1024(R0), (F1, F2) + RET