feat(asm): split wide arm64 load and store offsets into REGTMP
Offsets the single-instruction forms cannot carry lower the way the toolchain lowers them: ADD or SUB moves the whole distance into REGTMP within the ±4095 band, and the 24-bit band above it splits into an ADD of the high half and an access of the low half, with the pair family taking the two-ADD sequence. The split band follows loadStoreClass per width, byte accesses taking the full 24 bits and the Q width the widest, so an offset the toolchain pools is never split instead. Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
2bd52eb7ad
commit
71e8dd550d
3 files changed
+282
-42
No files matched your search
+191
-42
@@ -265,6 +265,12 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
return 4
|
||||
}
|
||||
}
|
||||
// The load/store pair family: one word in range, otherwise the REGTMP
|
||||
// expansion the toolchain lowers to (two words within ±4095, three
|
||||
// beyond, the pool words riding the function end).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair {
|
||||
return arm64PairSize(mnem, ops, fi)
|
||||
}
|
||||
// The funcdata pseudo-statements contribute no bytes, the expanded
|
||||
// FUNCDATA/PCDATA forms included.
|
||||
switch mnem {
|
||||
@@ -1599,6 +1605,38 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
|
||||
return encodeARM64RegMove(mnem, src, dst)
|
||||
}
|
||||
|
||||
// arm64PairSize returns the encoded size of a load/store pair instruction,
|
||||
// mirroring the encoder's branch order exactly: the label offsets of pass 1
|
||||
// must match the bytes pass 2 lays down.
|
||||
func arm64PairSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if len(ops) != 2 {
|
||||
return 4
|
||||
}
|
||||
load := strings.Contains(mnem, "LDP")
|
||||
memOp := ops[0]
|
||||
if !load {
|
||||
memOp = ops[1]
|
||||
}
|
||||
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" {
|
||||
return 12 // ADRP + ADD + pair
|
||||
}
|
||||
_, off := arm64MemWithFrame(memOp, fi)
|
||||
if off < 0 {
|
||||
return 4 // the encoder rejects it in pass 2
|
||||
}
|
||||
scale := arm64PairScale(mnem)
|
||||
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
|
||||
return 4
|
||||
}
|
||||
if off >= -4095 && off <= 4095 {
|
||||
return 8 // ADD/SUB the whole offset into REGTMP + pair
|
||||
}
|
||||
if off >= 0 && off <= 0xffffff {
|
||||
return 12 // the two-ADD split
|
||||
}
|
||||
return 12 // the pool sequence: LDR literal + ADD + pair
|
||||
}
|
||||
|
||||
// arm64MovSize returns the encoded size of a MOV instruction.
|
||||
func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if len(ops) != 2 {
|
||||
@@ -1651,10 +1689,15 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if off >= -256 && off <= 255 {
|
||||
return 4 // unscaled
|
||||
}
|
||||
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
|
||||
return 8 // ADD base, REGTMP + access
|
||||
if off >= -4095 && off <= 4095 {
|
||||
return 8 // ADD/SUB the whole offset into REGTMP
|
||||
}
|
||||
return 12 // literal pool range: encoding reports it as unsupported
|
||||
if arm64OffsetSplitReach(off, lt) {
|
||||
if _, _, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale))); ok {
|
||||
return 8 // ADD base, REGTMP + access
|
||||
}
|
||||
}
|
||||
return 12 // the pool sequence: LDR literal + register-offset access
|
||||
default:
|
||||
return 4 // register move
|
||||
}
|
||||
@@ -1929,40 +1972,72 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
|
||||
if off >= -256 && off <= 255 {
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil
|
||||
}
|
||||
// Offsets within ±4095 that no single-instruction form carries ride the
|
||||
// toolchain's ADD/SUB fallback: the whole offset moves into REGTMP and
|
||||
// the access reads it back at zero (asm7.go cases 30/31).
|
||||
if off >= -4095 && off <= 4095 {
|
||||
op, v := uint32(0), off // ADD
|
||||
if v < 0 {
|
||||
op, v = 1, -v // SUB
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), 0, 27, uint32(reg)),
|
||||
), nil
|
||||
}
|
||||
// Large offset: materialise the base in REGTMP (R27) the way the
|
||||
// toolchain does and access what remains. The ADD offsets from the
|
||||
// operand's own base register, [SP] and [Rn] alike.
|
||||
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
|
||||
// toolchain does and access what remains (asm7.go cases 30/31). The
|
||||
// ADD offsets from the operand's own base register, [SP] and [Rn]
|
||||
// alike; beyond the split band the offset reaches the literal pool.
|
||||
if !arm64OffsetSplitReach(off, lt) {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
hi, lo, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale)))
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 0, 0, addShift, addImm, uint32(rn), 27), // ADD $addImm<<shift, Rn, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
|
||||
arm64AddImmWord(hi, uint32(rn), 27), // ADD $hi[<<12], Rn, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(lo), 27, uint32(reg)),
|
||||
), nil
|
||||
}
|
||||
|
||||
// arm64SplitOffset decomposes an out-of-range offset for a REGTMP base: an
|
||||
// ADD (plain, or shifted left by 12) brings the base near the target and the
|
||||
// access covers what remains. ok is false when no decomposition exists
|
||||
// (negative offsets, or beyond 16 MiB, where the toolchain falls back to a
|
||||
// literal pool).
|
||||
func arm64SplitOffset(off int64, scale int64) (addImm, addShift uint32, access int64, ok bool) {
|
||||
// arm64OffsetSplitReach reports whether an offset stays within the band the
|
||||
// toolchain lowers to ADD plus access instead of pooling (loadStoreClass):
|
||||
// ±4095 for every width, then the width's own 24-bit band, aligned to the
|
||||
// access size or under its small unaligned ceiling. Byte accesses take the
|
||||
// full 24 bits with no alignment clause; the Q width, whose size lives in
|
||||
// opc, carries the widest band.
|
||||
func arm64OffsetSplitReach(off int64, lt a64LSType) bool {
|
||||
if off >= -4095 && off <= 4095 {
|
||||
return true
|
||||
}
|
||||
if off < 0 {
|
||||
return 0, 0, 0, false
|
||||
return false
|
||||
}
|
||||
// Plain ADD: bring the base to within the largest scaled access.
|
||||
l := min(off, 4095*scale)
|
||||
l -= l % scale
|
||||
if a := off - l; a <= 4095 {
|
||||
return uint32(a), 0, l, true
|
||||
if lt.size == 0 && lt.V == 1 { // the Q width
|
||||
return off <= 0xfff000+0xfff<<4 && off&15 == 0 || off <= 0xfff+0xfff<<4
|
||||
}
|
||||
// Shifted ADD: cover everything but the bits the access imm12 carries.
|
||||
rest := off &^ (0xFFF * scale)
|
||||
if rest >= 0 && rest>>12 <= 4095 {
|
||||
return uint32(rest >> 12), 1, off - rest, true
|
||||
switch lt.size {
|
||||
case 0: // byte: the full 24-bit band, alignment trivial
|
||||
return off <= 0xffffff
|
||||
case 1:
|
||||
return off <= 0xfff000+0xfff<<1 && off&1 == 0 || off <= 0xfff+0xfff<<1
|
||||
case 2:
|
||||
return off <= 0xfff000+0xfff<<2 && off&3 == 0 || off <= 0xfff+0xfff<<2
|
||||
default:
|
||||
return off <= 0xfff000+0xfff<<3 && off&7 == 0 || off <= 0xfff+0xfff<<3
|
||||
}
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
|
||||
// arm64AddImmWord encodes ADD $v, Rn, Rd the way the toolchain's oaddi
|
||||
// does: a non-zero multiple of 0x1000 encodes shifted left by twelve.
|
||||
func arm64AddImmWord(v int64, rn, rd uint32) uint32 {
|
||||
sh := arm64AddShift(v)
|
||||
if sh == 1 {
|
||||
v >>= 12
|
||||
}
|
||||
return a64AddSub(1, 0, 0, sh, uint32(v), rn, rd)
|
||||
}
|
||||
|
||||
// ---- static symbol references (ADRP + offset) ----
|
||||
@@ -3146,13 +3221,7 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
scale := int64(8)
|
||||
switch {
|
||||
case strings.HasSuffix(mnem, "W"):
|
||||
scale = 4 // the 32-bit pairs, signed and unsigned
|
||||
case strings.HasSuffix(mnem, "Q"):
|
||||
scale = 16 // the 128-bit FP pairs
|
||||
}
|
||||
scale := arm64PairScale(mnem)
|
||||
load := strings.Contains(mnem, "LDP")
|
||||
memOp, pairOp := ops[0], ops[1]
|
||||
if !load {
|
||||
@@ -3193,19 +3262,99 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra
|
||||
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" {
|
||||
return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem)
|
||||
}
|
||||
if off%scale != 0 || off < -64*scale || off > 63*scale {
|
||||
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
|
||||
imm7 := off / scale
|
||||
// The signed-offset form carries bits 24:23 = 10; post-index drops
|
||||
// bit 24 and pre-index sets both, with the base carrying the opc, V
|
||||
// and L halves.
|
||||
switch wb {
|
||||
case "P":
|
||||
baseOp = baseOp&^(1<<24) | 1<<23
|
||||
case "W":
|
||||
baseOp |= 1 << 23
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
||||
}
|
||||
if wb != "" {
|
||||
return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale)
|
||||
}
|
||||
imm7 := off / scale
|
||||
// The signed-offset form carries bits 24:23 = 10; post-index drops bit 24
|
||||
// and pre-index sets both, with the base carrying the opc, V and L halves.
|
||||
switch wb {
|
||||
case "P":
|
||||
baseOp = baseOp&^(1<<24) | 1<<23
|
||||
case "W":
|
||||
baseOp |= 1 << 23
|
||||
// Offsets within ±4095 the imm7 field cannot carry move the whole
|
||||
// distance into REGTMP first (asm7.go cases 74/76: add/sub + ldp/stp).
|
||||
if off >= -4095 && off <= 4095 {
|
||||
op, v := uint32(0), off // ADD
|
||||
if v < 0 {
|
||||
op, v = 1, -v // SUB
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
|
||||
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
||||
), nil
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
||||
// Positive offsets up to 16 MiB split into two ADDs: the low imm12 bits
|
||||
// from the base register into REGTMP, the high multiple of 0x1000 on top
|
||||
// (asm7.go cases 75/77). Beyond the band the offset reaches the pool.
|
||||
if off < 0 || off > 0xffffff {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
hi, lo, ok := arm64SplitImm24(off, 0)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
return a64WordsLE(
|
||||
arm64AddImmWord(lo, uint32(rn), 27), // ADD $lo, Rn, R27
|
||||
arm64AddImmWord(hi, 27, 27), // ADD $hi[<<12], R27, R27
|
||||
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
||||
), nil
|
||||
}
|
||||
|
||||
// arm64PairScale returns the byte width a pair access's imm7 offset divides
|
||||
// by: four for the 32-bit pairs, sixteen for the 128-bit FP pairs, eight for
|
||||
// everything else.
|
||||
func arm64PairScale(mnem string) int64 {
|
||||
switch {
|
||||
case strings.HasSuffix(mnem, "W"):
|
||||
return 4 // the 32-bit pairs, signed and unsigned
|
||||
case strings.HasSuffix(mnem, "Q"):
|
||||
return 16 // the 128-bit FP pairs
|
||||
}
|
||||
return 8
|
||||
}
|
||||
|
||||
// arm64AddShift returns the ADD-immediate shift bit: a non-zero value whose
|
||||
// low twelve bits are zero encodes shifted left by twelve.
|
||||
func arm64AddShift(v int64) uint32 {
|
||||
if v != 0 && v&0xfff == 0 {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// arm64SplitImm24 decomposes an offset the way the toolchain's
|
||||
// splitImm24uScaled does for the two-ADD sequences: hi either fits imm12
|
||||
// alone or is a multiple of 0x1000 up to 0xfff000, and the low half always
|
||||
// fits imm12 after division by the access scale. ok is false when the value
|
||||
// is negative or beyond the 24-bit reach, where the toolchain pools.
|
||||
func arm64SplitImm24(v int64, shift int) (hi, lo int64, ok bool) {
|
||||
if v < 0 || v > 0xfff000+0xfff<<uint(shift) {
|
||||
return 0, 0, false
|
||||
}
|
||||
h := max(v-(0xfff<<uint(shift)), v&((1<<uint(shift))-1))
|
||||
if h <= 0xfff {
|
||||
l := (v - h) >> uint(shift)
|
||||
if l <= 0xfff {
|
||||
return h, l, true
|
||||
}
|
||||
}
|
||||
l := (v >> uint(shift)) & 0xfff
|
||||
h = v - l<<uint(shift)
|
||||
if h > 0xfff000 {
|
||||
h = 0xfff000
|
||||
l = (v - h) >> uint(shift)
|
||||
}
|
||||
if h&^0xfff000 == 0 && h+l<<uint(shift) == v {
|
||||
return h, l, true
|
||||
}
|
||||
return 0, 0, false
|
||||
}
|
||||
|
||||
// encodeARM64AcqRel encodes the acquire/release loads and stores. Loads
|
||||
|
||||
@@ -719,6 +719,40 @@ func TestArm64QMove(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64OffsetSplit pins the ADD/SUB-into-REGTMP fallbacks against the
|
||||
// toolchain words: the whole-offset form for ±4095 and the 24-bit hi/lo
|
||||
// split for the wide bands, on the MOV and pair families alike.
|
||||
func TestArm64OffsetSplit(t *testing.T) {
|
||||
got := arm64Words(t, "\tMOVD R1, 4094(R2)\n\tMOVD R1, -300(R2)\n\tMOVD R1, 0x1006ff8(R2)\n\tMOVB R1, 4096(R2)\n"+
|
||||
"\tSTP (R3, R4), 11(R0)\n\tSTP (R3, R4), 65536(R2)\n\tLDP -31(R0), (R1, R2)\n")
|
||||
want := []uint32{
|
||||
0x913ff85b, // ADD $4094, R2, R27
|
||||
0xf9000361, // MOVD R1, (R27)
|
||||
0xd104b05b, // SUB $300, R2, R27
|
||||
0xf9000361, // MOVD R1, (R27)
|
||||
0x917ffc5b, // ADD $(4095<<12), R2, R27
|
||||
0xf93fff61, // MOVD R1, 32760(R27)
|
||||
0x9100045b, // ADD $1, R2, R27
|
||||
0x393fff61, // MOVB R1, 4095(R27)
|
||||
0x91002c1b, // ADD $11, R0, R27
|
||||
0xa9001363, // STP (R3, R4), (R27)
|
||||
0x9100005b, // ADD $0, R2, R27
|
||||
0x9140437b, // ADD $(16<<12), R27, R27
|
||||
0xa9001363, // STP (R3, R4), (R27)
|
||||
0xd1007c1b, // SUB $31, R0, R27
|
||||
0xa9400b61, // LDP (R27), (R1, R2)
|
||||
0xd65f03c0,
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("word count = %d, want %d", len(got), len(want))
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("word %d = %08x, want %08x", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestArm64BTI pins the landing-pad family against the toolchain words:
|
||||
// only the uppercase C/J/JC spellings assemble, and bare BTI is a
|
||||
// diagnostic, never a panic.
|
||||
|
||||
Vendored
+57
@@ -0,0 +1,57 @@
|
||||
// Copyright (c) 2026 Petr Balvín <opensource@petrbalvin.org> (https://petrbalvin.org)
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
// Differential kernel for the arm64 out-of-band offsets: the ADD/SUB into
|
||||
// REGTMP for offsets within ±4095 no single instruction carries, and the
|
||||
// 24-bit hi/lo split for the wide bands. Every function is byte-compared
|
||||
// against go tool asm.
|
||||
|
||||
#include "textflag.h"
|
||||
|
||||
// func bytes()
|
||||
TEXT ·bytes(SB), NOSPLIT, $0-0
|
||||
MOVB R1, 4096(R2)
|
||||
MOVB R1, 8190(R2)
|
||||
MOVB R1, 65520(R2)
|
||||
MOVB R1, -4095(R2)
|
||||
MOVB 4096(R2), R1
|
||||
RET
|
||||
|
||||
// func halves()
|
||||
TEXT ·halves(SB), NOSPLIT, $0-0
|
||||
MOVH R1, 16384(R2)
|
||||
MOVH R1, 16781310(R2)
|
||||
MOVH 0x2002(R1), R2
|
||||
MOVH R1, -4095(R2)
|
||||
RET
|
||||
|
||||
// func words()
|
||||
TEXT ·words(SB), NOSPLIT, $0-0
|
||||
MOVW R1, 16388(R2)
|
||||
MOVW 0x4004(R1), R2
|
||||
MOVD R1, 4094(R2)
|
||||
MOVD R1, -300(R2)
|
||||
MOVD R1, 0x1006ff8(R2)
|
||||
MOVD 0x1006ff8(R1), R2
|
||||
FMOVS F1, 0x4004(R2)
|
||||
FMOVD F1, 0x1006ff8(R1)
|
||||
FMOVQ F1, 0x1003000(R2)
|
||||
RET
|
||||
|
||||
// func pairs()
|
||||
TEXT ·pairs(SB), NOSPLIT, $0-0
|
||||
STP (R3, R4), 11(R0)
|
||||
STP (R3, R4), 1024(R0)
|
||||
STP (R3, R4), -4(R5)
|
||||
STP (R3, R4), 65536(R2)
|
||||
STPW (R3, R4), 1024(R0)
|
||||
LDP 11(R0), (R1, R2)
|
||||
LDP 1024(R0), (R1, R2)
|
||||
LDP -31(R0), (R1, R2)
|
||||
LDP 65536(R2), (R1, R2)
|
||||
LDPW -5(R0), (R1, R2)
|
||||
LDPSW 1024(R0), (R1, R2)
|
||||
FLDPD 1024(R0), (F1, F2)
|
||||
FLDPQ 1024(R0), (F1, F2)
|
||||
FLDPS 1024(R0), (F1, F2)
|
||||
RET
|
||||
Reference in new issue
Block a user