feat(asm): split wide arm64 load and store offsets into REGTMP

Offsets the single-instruction forms cannot carry lower the way the
toolchain lowers them: ADD or SUB moves the whole distance into REGTMP
within the ±4095 band, and the 24-bit band above it splits into an ADD of
the high half and an access of the low half, with the pair family taking
the two-ADD sequence.  The split band follows loadStoreClass per width,
byte accesses taking the full 24 bits and the Q width the widest, so an
offset the toolchain pools is never split instead.

Assisted-by: GLM 5.3 Flash
This commit is contained in:
petrbalvin committed 2026-10-07 02:36:24 +02:00
1 parent 2bd52eb7ad
commit 71e8dd550d
3 files changed
+282 -42

No files matched your search

+191 -42
View File
@@ -265,6 +265,12 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
return 4
}
}
// The load/store pair family: one word in range, otherwise the REGTMP
// expansion the toolchain lowers to (two words within ±4095, three
// beyond, the pool words riding the function end).
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair {
return arm64PairSize(mnem, ops, fi)
}
// The funcdata pseudo-statements contribute no bytes, the expanded
// FUNCDATA/PCDATA forms included.
switch mnem {
@@ -1599,6 +1605,38 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
return encodeARM64RegMove(mnem, src, dst)
}
// arm64PairSize returns the encoded size of a load/store pair instruction,
// mirroring the encoder's branch order exactly: the label offsets of pass 1
// must match the bytes pass 2 lays down.
func arm64PairSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if len(ops) != 2 {
return 4
}
load := strings.Contains(mnem, "LDP")
memOp := ops[0]
if !load {
memOp = ops[1]
}
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" {
return 12 // ADRP + ADD + pair
}
_, off := arm64MemWithFrame(memOp, fi)
if off < 0 {
return 4 // the encoder rejects it in pass 2
}
scale := arm64PairScale(mnem)
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
return 4
}
if off >= -4095 && off <= 4095 {
return 8 // ADD/SUB the whole offset into REGTMP + pair
}
if off >= 0 && off <= 0xffffff {
return 12 // the two-ADD split
}
return 12 // the pool sequence: LDR literal + ADD + pair
}
// arm64MovSize returns the encoded size of a MOV instruction.
func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if len(ops) != 2 {
@@ -1651,10 +1689,15 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if off >= -256 && off <= 255 {
return 4 // unscaled
}
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
return 8 // ADD base, REGTMP + access
if off >= -4095 && off <= 4095 {
return 8 // ADD/SUB the whole offset into REGTMP
}
return 12 // literal pool range: encoding reports it as unsupported
if arm64OffsetSplitReach(off, lt) {
if _, _, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale))); ok {
return 8 // ADD base, REGTMP + access
}
}
return 12 // the pool sequence: LDR literal + register-offset access
default:
return 4 // register move
}
@@ -1929,40 +1972,72 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil
}
// Offsets within ±4095 that no single-instruction form carries ride the
// toolchain's ADD/SUB fallback: the whole offset moves into REGTMP and
// the access reads it back at zero (asm7.go cases 30/31).
if off >= -4095 && off <= 4095 {
op, v := uint32(0), off // ADD
if v < 0 {
op, v = 1, -v // SUB
}
return a64WordsLE(
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), 0, 27, uint32(reg)),
), nil
}
// Large offset: materialise the base in REGTMP (R27) the way the
// toolchain does and access what remains. The ADD offsets from the
// operand's own base register, [SP] and [Rn] alike.
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
// toolchain does and access what remains (asm7.go cases 30/31). The
// ADD offsets from the operand's own base register, [SP] and [Rn]
// alike; beyond the split band the offset reaches the literal pool.
if !arm64OffsetSplitReach(off, lt) {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
hi, lo, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale)))
if !ok {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
return a64WordsLE(
a64AddSub(1, 0, 0, addShift, addImm, uint32(rn), 27), // ADD $addImm<<shift, Rn, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
arm64AddImmWord(hi, uint32(rn), 27), // ADD $hi[<<12], Rn, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(lo), 27, uint32(reg)),
), nil
}
// arm64SplitOffset decomposes an out-of-range offset for a REGTMP base: an
// ADD (plain, or shifted left by 12) brings the base near the target and the
// access covers what remains. ok is false when no decomposition exists
// (negative offsets, or beyond 16 MiB, where the toolchain falls back to a
// literal pool).
func arm64SplitOffset(off int64, scale int64) (addImm, addShift uint32, access int64, ok bool) {
// arm64OffsetSplitReach reports whether an offset stays within the band the
// toolchain lowers to ADD plus access instead of pooling (loadStoreClass):
// ±4095 for every width, then the width's own 24-bit band, aligned to the
// access size or under its small unaligned ceiling. Byte accesses take the
// full 24 bits with no alignment clause; the Q width, whose size lives in
// opc, carries the widest band.
func arm64OffsetSplitReach(off int64, lt a64LSType) bool {
if off >= -4095 && off <= 4095 {
return true
}
if off < 0 {
return 0, 0, 0, false
return false
}
// Plain ADD: bring the base to within the largest scaled access.
l := min(off, 4095*scale)
l -= l % scale
if a := off - l; a <= 4095 {
return uint32(a), 0, l, true
if lt.size == 0 && lt.V == 1 { // the Q width
return off <= 0xfff000+0xfff<<4 && off&15 == 0 || off <= 0xfff+0xfff<<4
}
// Shifted ADD: cover everything but the bits the access imm12 carries.
rest := off &^ (0xFFF * scale)
if rest >= 0 && rest>>12 <= 4095 {
return uint32(rest >> 12), 1, off - rest, true
switch lt.size {
case 0: // byte: the full 24-bit band, alignment trivial
return off <= 0xffffff
case 1:
return off <= 0xfff000+0xfff<<1 && off&1 == 0 || off <= 0xfff+0xfff<<1
case 2:
return off <= 0xfff000+0xfff<<2 && off&3 == 0 || off <= 0xfff+0xfff<<2
default:
return off <= 0xfff000+0xfff<<3 && off&7 == 0 || off <= 0xfff+0xfff<<3
}
return 0, 0, 0, false
}
// arm64AddImmWord encodes ADD $v, Rn, Rd the way the toolchain's oaddi
// does: a non-zero multiple of 0x1000 encodes shifted left by twelve.
func arm64AddImmWord(v int64, rn, rd uint32) uint32 {
sh := arm64AddShift(v)
if sh == 1 {
v >>= 12
}
return a64AddSub(1, 0, 0, sh, uint32(v), rn, rd)
}
// ---- static symbol references (ADRP + offset) ----
@@ -3146,13 +3221,7 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra
if len(ops) != 2 {
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
}
scale := int64(8)
switch {
case strings.HasSuffix(mnem, "W"):
scale = 4 // the 32-bit pairs, signed and unsigned
case strings.HasSuffix(mnem, "Q"):
scale = 16 // the 128-bit FP pairs
}
scale := arm64PairScale(mnem)
load := strings.Contains(mnem, "LDP")
memOp, pairOp := ops[0], ops[1]
if !load {
@@ -3193,19 +3262,99 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" {
return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem)
}
if off%scale != 0 || off < -64*scale || off > 63*scale {
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
imm7 := off / scale
// The signed-offset form carries bits 24:23 = 10; post-index drops
// bit 24 and pre-index sets both, with the base carrying the opc, V
// and L halves.
switch wb {
case "P":
baseOp = baseOp&^(1<<24) | 1<<23
case "W":
baseOp |= 1 << 23
}
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
}
if wb != "" {
return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale)
}
imm7 := off / scale
// The signed-offset form carries bits 24:23 = 10; post-index drops bit 24
// and pre-index sets both, with the base carrying the opc, V and L halves.
switch wb {
case "P":
baseOp = baseOp&^(1<<24) | 1<<23
case "W":
baseOp |= 1 << 23
// Offsets within ±4095 the imm7 field cannot carry move the whole
// distance into REGTMP first (asm7.go cases 74/76: add/sub + ldp/stp).
if off >= -4095 && off <= 4095 {
op, v := uint32(0), off // ADD
if v < 0 {
op, v = 1, -v // SUB
}
return a64WordsLE(
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
), nil
}
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
// Positive offsets up to 16 MiB split into two ADDs: the low imm12 bits
// from the base register into REGTMP, the high multiple of 0x1000 on top
// (asm7.go cases 75/77). Beyond the band the offset reaches the pool.
if off < 0 || off > 0xffffff {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
hi, lo, ok := arm64SplitImm24(off, 0)
if !ok {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
return a64WordsLE(
arm64AddImmWord(lo, uint32(rn), 27), // ADD $lo, Rn, R27
arm64AddImmWord(hi, 27, 27), // ADD $hi[<<12], R27, R27
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
), nil
}
// arm64PairScale returns the byte width a pair access's imm7 offset divides
// by: four for the 32-bit pairs, sixteen for the 128-bit FP pairs, eight for
// everything else.
func arm64PairScale(mnem string) int64 {
switch {
case strings.HasSuffix(mnem, "W"):
return 4 // the 32-bit pairs, signed and unsigned
case strings.HasSuffix(mnem, "Q"):
return 16 // the 128-bit FP pairs
}
return 8
}
// arm64AddShift returns the ADD-immediate shift bit: a non-zero value whose
// low twelve bits are zero encodes shifted left by twelve.
func arm64AddShift(v int64) uint32 {
if v != 0 && v&0xfff == 0 {
return 1
}
return 0
}
// arm64SplitImm24 decomposes an offset the way the toolchain's
// splitImm24uScaled does for the two-ADD sequences: hi either fits imm12
// alone or is a multiple of 0x1000 up to 0xfff000, and the low half always
// fits imm12 after division by the access scale. ok is false when the value
// is negative or beyond the 24-bit reach, where the toolchain pools.
func arm64SplitImm24(v int64, shift int) (hi, lo int64, ok bool) {
if v < 0 || v > 0xfff000+0xfff<<uint(shift) {
return 0, 0, false
}
h := max(v-(0xfff<<uint(shift)), v&((1<<uint(shift))-1))
if h <= 0xfff {
l := (v - h) >> uint(shift)
if l <= 0xfff {
return h, l, true
}
}
l := (v >> uint(shift)) & 0xfff
h = v - l<<uint(shift)
if h > 0xfff000 {
h = 0xfff000
l = (v - h) >> uint(shift)
}
if h&^0xfff000 == 0 && h+l<<uint(shift) == v {
return h, l, true
}
return 0, 0, false
}
// encodeARM64AcqRel encodes the acquire/release loads and stores. Loads
+34
View File
@@ -719,6 +719,40 @@ func TestArm64QMove(t *testing.T) {
}
}
// TestArm64OffsetSplit pins the ADD/SUB-into-REGTMP fallbacks against the
// toolchain words: the whole-offset form for ±4095 and the 24-bit hi/lo
// split for the wide bands, on the MOV and pair families alike.
func TestArm64OffsetSplit(t *testing.T) {
got := arm64Words(t, "\tMOVD R1, 4094(R2)\n\tMOVD R1, -300(R2)\n\tMOVD R1, 0x1006ff8(R2)\n\tMOVB R1, 4096(R2)\n"+
"\tSTP (R3, R4), 11(R0)\n\tSTP (R3, R4), 65536(R2)\n\tLDP -31(R0), (R1, R2)\n")
want := []uint32{
0x913ff85b, // ADD $4094, R2, R27
0xf9000361, // MOVD R1, (R27)
0xd104b05b, // SUB $300, R2, R27
0xf9000361, // MOVD R1, (R27)
0x917ffc5b, // ADD $(4095<<12), R2, R27
0xf93fff61, // MOVD R1, 32760(R27)
0x9100045b, // ADD $1, R2, R27
0x393fff61, // MOVB R1, 4095(R27)
0x91002c1b, // ADD $11, R0, R27
0xa9001363, // STP (R3, R4), (R27)
0x9100005b, // ADD $0, R2, R27
0x9140437b, // ADD $(16<<12), R27, R27
0xa9001363, // STP (R3, R4), (R27)
0xd1007c1b, // SUB $31, R0, R27
0xa9400b61, // LDP (R27), (R1, R2)
0xd65f03c0,
}
if len(got) != len(want) {
t.Fatalf("word count = %d, want %d", len(got), len(want))
}
for i := range want {
if got[i] != want[i] {
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
}
}
}
// TestArm64BTI pins the landing-pad family against the toolchain words:
// only the uppercase C/J/JC spellings assemble, and bare BTI is a
// diagnostic, never a panic.
+57
View File
@@ -0,0 +1,57 @@
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
// SPDX-License-Identifier: BSD-3-Clause
// Differential kernel for the arm64 out-of-band offsets: the ADD/SUB into
// REGTMP for offsets within ±4095 no single instruction carries, and the
// 24-bit hi/lo split for the wide bands. Every function is byte-compared
// against go tool asm.
#include "textflag.h"
// func bytes()
TEXT ·bytes(SB), NOSPLIT, $0-0
MOVB R1, 4096(R2)
MOVB R1, 8190(R2)
MOVB R1, 65520(R2)
MOVB R1, -4095(R2)
MOVB 4096(R2), R1
RET
// func halves()
TEXT ·halves(SB), NOSPLIT, $0-0
MOVH R1, 16384(R2)
MOVH R1, 16781310(R2)
MOVH 0x2002(R1), R2
MOVH R1, -4095(R2)
RET
// func words()
TEXT ·words(SB), NOSPLIT, $0-0
MOVW R1, 16388(R2)
MOVW 0x4004(R1), R2
MOVD R1, 4094(R2)
MOVD R1, -300(R2)
MOVD R1, 0x1006ff8(R2)
MOVD 0x1006ff8(R1), R2
FMOVS F1, 0x4004(R2)
FMOVD F1, 0x1006ff8(R1)
FMOVQ F1, 0x1003000(R2)
RET
// func pairs()
TEXT ·pairs(SB), NOSPLIT, $0-0
STP (R3, R4), 11(R0)
STP (R3, R4), 1024(R0)
STP (R3, R4), -4(R5)
STP (R3, R4), 65536(R2)
STPW (R3, R4), 1024(R0)
LDP 11(R0), (R1, R2)
LDP 1024(R0), (R1, R2)
LDP -31(R0), (R1, R2)
LDP 65536(R2), (R1, R2)
LDPW -5(R0), (R1, R2)
LDPSW 1024(R0), (R1, R2)
FLDPD 1024(R0), (F1, F2)
FLDPQ 1024(R0), (F1, F2)
FLDPS 1024(R0), (F1, F2)
RET