feat(asm): split wide arm64 load and store offsets into REGTMP
Offsets the single-instruction forms cannot carry lower the way the toolchain lowers them: ADD or SUB moves the whole distance into REGTMP within the ±4095 band, and the 24-bit band above it splits into an ADD of the high half and an access of the low half, with the pair family taking the two-ADD sequence. The split band follows loadStoreClass per width, byte accesses taking the full 24 bits and the Q width the widest, so an offset the toolchain pools is never split instead. Assisted-by: GLM 5.3 Flash
This commit is contained in:
1 parent
2bd52eb7ad
commit
71e8dd550d
3 files changed
+282
-42
No files matched your search
+191
-42
@@ -265,6 +265,12 @@ func arm64InstrSize(instr *ast.Instr, fi arm64FrameInfo, pos int) int {
|
||||
return 4
|
||||
}
|
||||
}
|
||||
// The load/store pair family: one word in range, otherwise the REGTMP
|
||||
// expansion the toolchain lowers to (two words within ±4095, three
|
||||
// beyond, the pool words riding the function end).
|
||||
if enc, ok := a64InstrTable[mnem]; ok && enc.format == a64FPair {
|
||||
return arm64PairSize(mnem, ops, fi)
|
||||
}
|
||||
// The funcdata pseudo-statements contribute no bytes, the expanded
|
||||
// FUNCDATA/PCDATA forms included.
|
||||
switch mnem {
|
||||
@@ -1599,6 +1605,38 @@ func encodeARM64Mov(instr *ast.Instr, mnem string, wb string, fi arm64FrameInfo,
|
||||
return encodeARM64RegMove(mnem, src, dst)
|
||||
}
|
||||
|
||||
// arm64PairSize returns the encoded size of a load/store pair instruction,
|
||||
// mirroring the encoder's branch order exactly: the label offsets of pass 1
|
||||
// must match the bytes pass 2 lays down.
|
||||
func arm64PairSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if len(ops) != 2 {
|
||||
return 4
|
||||
}
|
||||
load := strings.Contains(mnem, "LDP")
|
||||
memOp := ops[0]
|
||||
if !load {
|
||||
memOp = ops[1]
|
||||
}
|
||||
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo == "SB" {
|
||||
return 12 // ADRP + ADD + pair
|
||||
}
|
||||
_, off := arm64MemWithFrame(memOp, fi)
|
||||
if off < 0 {
|
||||
return 4 // the encoder rejects it in pass 2
|
||||
}
|
||||
scale := arm64PairScale(mnem)
|
||||
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
|
||||
return 4
|
||||
}
|
||||
if off >= -4095 && off <= 4095 {
|
||||
return 8 // ADD/SUB the whole offset into REGTMP + pair
|
||||
}
|
||||
if off >= 0 && off <= 0xffffff {
|
||||
return 12 // the two-ADD split
|
||||
}
|
||||
return 12 // the pool sequence: LDR literal + ADD + pair
|
||||
}
|
||||
|
||||
// arm64MovSize returns the encoded size of a MOV instruction.
|
||||
func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if len(ops) != 2 {
|
||||
@@ -1651,10 +1689,15 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if off >= -256 && off <= 255 {
|
||||
return 4 // unscaled
|
||||
}
|
||||
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
|
||||
return 8 // ADD base, REGTMP + access
|
||||
if off >= -4095 && off <= 4095 {
|
||||
return 8 // ADD/SUB the whole offset into REGTMP
|
||||
}
|
||||
return 12 // literal pool range: encoding reports it as unsupported
|
||||
if arm64OffsetSplitReach(off, lt) {
|
||||
if _, _, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale))); ok {
|
||||
return 8 // ADD base, REGTMP + access
|
||||
}
|
||||
}
|
||||
return 12 // the pool sequence: LDR literal + register-offset access
|
||||
default:
|
||||
return 4 // register move
|
||||
}
|
||||
@@ -1929,40 +1972,72 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
|
||||
if off >= -256 && off <= 255 {
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, int32(off), rn, reg)), nil
|
||||
}
|
||||
// Offsets within ±4095 that no single-instruction form carries ride the
|
||||
// toolchain's ADD/SUB fallback: the whole offset moves into REGTMP and
|
||||
// the access reads it back at zero (asm7.go cases 30/31).
|
||||
if off >= -4095 && off <= 4095 {
|
||||
op, v := uint32(0), off // ADD
|
||||
if v < 0 {
|
||||
op, v = 1, -v // SUB
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), 0, 27, uint32(reg)),
|
||||
), nil
|
||||
}
|
||||
// Large offset: materialise the base in REGTMP (R27) the way the
|
||||
// toolchain does and access what remains. The ADD offsets from the
|
||||
// operand's own base register, [SP] and [Rn] alike.
|
||||
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
|
||||
// toolchain does and access what remains (asm7.go cases 30/31). The
|
||||
// ADD offsets from the operand's own base register, [SP] and [Rn]
|
||||
// alike; beyond the split band the offset reaches the literal pool.
|
||||
if !arm64OffsetSplitReach(off, lt) {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
hi, lo, ok := arm64SplitImm24(off, bits.TrailingZeros64(uint64(scale)))
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 0, 0, addShift, addImm, uint32(rn), 27), // ADD $addImm<<shift, Rn, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
|
||||
arm64AddImmWord(hi, uint32(rn), 27), // ADD $hi[<<12], Rn, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(lo), 27, uint32(reg)),
|
||||
), nil
|
||||
}
|
||||
|
||||
// arm64SplitOffset decomposes an out-of-range offset for a REGTMP base: an
|
||||
// ADD (plain, or shifted left by 12) brings the base near the target and the
|
||||
// access covers what remains. ok is false when no decomposition exists
|
||||
// (negative offsets, or beyond 16 MiB, where the toolchain falls back to a
|
||||
// literal pool).
|
||||
func arm64SplitOffset(off int64, scale int64) (addImm, addShift uint32, access int64, ok bool) {
|
||||
// arm64OffsetSplitReach reports whether an offset stays within the band the
|
||||
// toolchain lowers to ADD plus access instead of pooling (loadStoreClass):
|
||||
// ±4095 for every width, then the width's own 24-bit band, aligned to the
|
||||
// access size or under its small unaligned ceiling. Byte accesses take the
|
||||
// full 24 bits with no alignment clause; the Q width, whose size lives in
|
||||
// opc, carries the widest band.
|
||||
func arm64OffsetSplitReach(off int64, lt a64LSType) bool {
|
||||
if off >= -4095 && off <= 4095 {
|
||||
return true
|
||||
}
|
||||
if off < 0 {
|
||||
return 0, 0, 0, false
|
||||
return false
|
||||
}
|
||||
// Plain ADD: bring the base to within the largest scaled access.
|
||||
l := min(off, 4095*scale)
|
||||
l -= l % scale
|
||||
if a := off - l; a <= 4095 {
|
||||
return uint32(a), 0, l, true
|
||||
if lt.size == 0 && lt.V == 1 { // the Q width
|
||||
return off <= 0xfff000+0xfff<<4 && off&15 == 0 || off <= 0xfff+0xfff<<4
|
||||
}
|
||||
// Shifted ADD: cover everything but the bits the access imm12 carries.
|
||||
rest := off &^ (0xFFF * scale)
|
||||
if rest >= 0 && rest>>12 <= 4095 {
|
||||
return uint32(rest >> 12), 1, off - rest, true
|
||||
switch lt.size {
|
||||
case 0: // byte: the full 24-bit band, alignment trivial
|
||||
return off <= 0xffffff
|
||||
case 1:
|
||||
return off <= 0xfff000+0xfff<<1 && off&1 == 0 || off <= 0xfff+0xfff<<1
|
||||
case 2:
|
||||
return off <= 0xfff000+0xfff<<2 && off&3 == 0 || off <= 0xfff+0xfff<<2
|
||||
default:
|
||||
return off <= 0xfff000+0xfff<<3 && off&7 == 0 || off <= 0xfff+0xfff<<3
|
||||
}
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
|
||||
// arm64AddImmWord encodes ADD $v, Rn, Rd the way the toolchain's oaddi
|
||||
// does: a non-zero multiple of 0x1000 encodes shifted left by twelve.
|
||||
func arm64AddImmWord(v int64, rn, rd uint32) uint32 {
|
||||
sh := arm64AddShift(v)
|
||||
if sh == 1 {
|
||||
v >>= 12
|
||||
}
|
||||
return a64AddSub(1, 0, 0, sh, uint32(v), rn, rd)
|
||||
}
|
||||
|
||||
// ---- static symbol references (ADRP + offset) ----
|
||||
@@ -3146,13 +3221,7 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra
|
||||
if len(ops) != 2 {
|
||||
return nil, fmt.Errorf("%s expects 2 operands, got %d", mnem, len(ops))
|
||||
}
|
||||
scale := int64(8)
|
||||
switch {
|
||||
case strings.HasSuffix(mnem, "W"):
|
||||
scale = 4 // the 32-bit pairs, signed and unsigned
|
||||
case strings.HasSuffix(mnem, "Q"):
|
||||
scale = 16 // the 128-bit FP pairs
|
||||
}
|
||||
scale := arm64PairScale(mnem)
|
||||
load := strings.Contains(mnem, "LDP")
|
||||
memOp, pairOp := ops[0], ops[1]
|
||||
if !load {
|
||||
@@ -3193,19 +3262,99 @@ func encodeARM64Pair(mnem string, baseOp uint32, ops []*ast.Operand, fi arm64Fra
|
||||
if memOp.Addr.Sym != nil && memOp.Addr.Sym.Pseudo != "" && wb != "" {
|
||||
return nil, fmt.Errorf("%s: writeback is not supported on a frame-relative operand", mnem)
|
||||
}
|
||||
if off%scale != 0 || off < -64*scale || off > 63*scale {
|
||||
if off%scale == 0 && off >= -64*scale && off <= 63*scale {
|
||||
imm7 := off / scale
|
||||
// The signed-offset form carries bits 24:23 = 10; post-index drops
|
||||
// bit 24 and pre-index sets both, with the base carrying the opc, V
|
||||
// and L halves.
|
||||
switch wb {
|
||||
case "P":
|
||||
baseOp = baseOp&^(1<<24) | 1<<23
|
||||
case "W":
|
||||
baseOp |= 1 << 23
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
||||
}
|
||||
if wb != "" {
|
||||
return nil, fmt.Errorf("%s: offset %d out of pair range or not a multiple of %d", mnem, off, scale)
|
||||
}
|
||||
imm7 := off / scale
|
||||
// The signed-offset form carries bits 24:23 = 10; post-index drops bit 24
|
||||
// and pre-index sets both, with the base carrying the opc, V and L halves.
|
||||
switch wb {
|
||||
case "P":
|
||||
baseOp = baseOp&^(1<<24) | 1<<23
|
||||
case "W":
|
||||
baseOp |= 1 << 23
|
||||
// Offsets within ±4095 the imm7 field cannot carry move the whole
|
||||
// distance into REGTMP first (asm7.go cases 74/76: add/sub + ldp/stp).
|
||||
if off >= -4095 && off <= 4095 {
|
||||
op, v := uint32(0), off // ADD
|
||||
if v < 0 {
|
||||
op, v = 1, -v // SUB
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, op, 0, 0, uint32(v), uint32(rn), 27), // ADD/SUB $v, Rn, R27
|
||||
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
||||
), nil
|
||||
}
|
||||
return a64wordLE(baseOp | uint32(imm7)&0x7F<<15 | uint32(rt2)<<10 | uint32(rn)<<5 | uint32(rt1)), nil
|
||||
// Positive offsets up to 16 MiB split into two ADDs: the low imm12 bits
|
||||
// from the base register into REGTMP, the high multiple of 0x1000 on top
|
||||
// (asm7.go cases 75/77). Beyond the band the offset reaches the pool.
|
||||
if off < 0 || off > 0xffffff {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
hi, lo, ok := arm64SplitImm24(off, 0)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
return a64WordsLE(
|
||||
arm64AddImmWord(lo, uint32(rn), 27), // ADD $lo, Rn, R27
|
||||
arm64AddImmWord(hi, 27, 27), // ADD $hi[<<12], R27, R27
|
||||
baseOp|uint32(rt2)<<10|27<<5|uint32(rt1), // LDP/STP (R27), (…)
|
||||
), nil
|
||||
}
|
||||
|
||||
// arm64PairScale returns the byte width a pair access's imm7 offset divides
|
||||
// by: four for the 32-bit pairs, sixteen for the 128-bit FP pairs, eight for
|
||||
// everything else.
|
||||
func arm64PairScale(mnem string) int64 {
|
||||
switch {
|
||||
case strings.HasSuffix(mnem, "W"):
|
||||
return 4 // the 32-bit pairs, signed and unsigned
|
||||
case strings.HasSuffix(mnem, "Q"):
|
||||
return 16 // the 128-bit FP pairs
|
||||
}
|
||||
return 8
|
||||
}
|
||||
|
||||
// arm64AddShift returns the ADD-immediate shift bit: a non-zero value whose
|
||||
// low twelve bits are zero encodes shifted left by twelve.
|
||||
func arm64AddShift(v int64) uint32 {
|
||||
if v != 0 && v&0xfff == 0 {
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// arm64SplitImm24 decomposes an offset the way the toolchain's
|
||||
// splitImm24uScaled does for the two-ADD sequences: hi either fits imm12
|
||||
// alone or is a multiple of 0x1000 up to 0xfff000, and the low half always
|
||||
// fits imm12 after division by the access scale. ok is false when the value
|
||||
// is negative or beyond the 24-bit reach, where the toolchain pools.
|
||||
func arm64SplitImm24(v int64, shift int) (hi, lo int64, ok bool) {
|
||||
if v < 0 || v > 0xfff000+0xfff<<uint(shift) {
|
||||
return 0, 0, false
|
||||
}
|
||||
h := max(v-(0xfff<<uint(shift)), v&((1<<uint(shift))-1))
|
||||
if h <= 0xfff {
|
||||
l := (v - h) >> uint(shift)
|
||||
if l <= 0xfff {
|
||||
return h, l, true
|
||||
}
|
||||
}
|
||||
l := (v >> uint(shift)) & 0xfff
|
||||
h = v - l<<uint(shift)
|
||||
if h > 0xfff000 {
|
||||
h = 0xfff000
|
||||
l = (v - h) >> uint(shift)
|
||||
}
|
||||
if h&^0xfff000 == 0 && h+l<<uint(shift) == v {
|
||||
return h, l, true
|
||||
}
|
||||
return 0, 0, false
|
||||
}
|
||||
|
||||
// encodeARM64AcqRel encodes the acquire/release loads and stores. Loads
|
||||
|
||||
Reference in new issue
Block a user