fix(asm): close the oracle parity gaps in frame addressing and calls

This commit is contained in:
2026-09-14 23:25:14 +02:00
parent 70218e84ba
commit 40476546df
18 changed files with 456 additions and 60 deletions
+60 -32
View File
@@ -222,7 +222,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
}
return a64wordLE(uint32(immFromOperand(ops[0]))), nil
case "B":
case "B", "JMP":
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
case "BL", "CALL":
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
@@ -337,8 +337,9 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
}
op := ops[0]
// External symbol reference: BL sym(SB).
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
// relocation (R_CALLARM64 either way).
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
if relocs != nil {
*relocs = append(*relocs, Reloc{
Off: 0,
@@ -348,8 +349,12 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
Kind: RelArm64Branch,
})
}
// Emit BL with zero offset; the linker fills in the target.
return a64wordLE(a64Branch(1, 0)), nil
// Emit B/BL with zero offset; the linker fills in the target.
bop := uint32(0) // B
if link {
bop = 1 // BL
}
return a64wordLE(a64Branch(bop, 0)), nil
}
target := resolve(arm64Label(op))
@@ -593,9 +598,9 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
}
_, off := arm64MemWithFrame(mem, fi)
// Scaled unsigned offset fits if aligned and in range.
lt := a64LoadTable[mnem]
if lt.size == 0 {
lt.size = 3 // default to64-bit for MOV
lt, ok := a64LoadTable[mnem]
if !ok {
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
}
scale := int32(1) << uint(lt.size)
if off >= 0 && off%scale == 0 && off/scale < 4096 {
@@ -604,7 +609,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
if off >= -256 && off <= 255 {
return 4 // unscaled
}
return 12 // materialise offset + LDR/STR
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
return 8 // ADD base, REGTMP + access
}
return 12 // literal pool range: encoding reports it as unsupported
default:
return 4 // register move
}
@@ -824,33 +832,53 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
}
scale := int32(1) << uint(lt.size)
if load {
// Try scaled unsigned offset first.
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil
}
}
// Try unscaled (9-bit signed).
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil
}
// Large offset: materialise in R20 (TMP) and use register-offset.
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
}
// Store: same encoding but opc bits indicate store.
storeOpc := a64StoreOpc(lt)
if off >= 0 && off%scale == 0 {
imm12 := uint32(off / scale)
if imm12 < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil
}
var opc int
if load {
opc = lt.opc
} else {
opc = storeOpc
}
// Scaled unsigned offset first, then the unscaled ±255 form.
if off >= 0 && off%scale == 0 && off/scale < 4096 {
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
}
if off >= -256 && off <= 255 {
return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil
}
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
// Large offset: materialise the base in REGTMP (R27) the way the
// toolchain does and access what remains.
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
if !ok {
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
}
return a64WordsLE(
a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<<shift, SP, R27
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
), nil
}
// arm64SplitOffset decomposes an out-of-range frame offset for a REGTMP
// base: an ADD (plain, or shifted left by 12) brings SP near the target and
// the access covers what remains. ok is false when no decomposition exists
// (offsets at or beyond 16 MiB, where the toolchain falls back to a literal
// pool).
func arm64SplitOffset(off int32, scale int32) (addImm, addShift uint32, access int32, ok bool) {
if off < 0 {
return 0, 0, 0, false
}
// Plain ADD: bring SP to within the largest scaled access.
l := min(off, 4095*scale)
l -= l % scale
if a := off - l; a <= 4095 {
return uint32(a), 0, l, true
}
// Shifted ADD: cover everything but the bits the access imm12 carries.
rest := off &^ (0xFFF * scale)
if rest >= 0 && rest>>12 <= 4095 {
return uint32(rest >> 12), 1, off - rest, true
}
return 0, 0, 0, false
}
// ---- static symbol references (ADRP + offset) ----