fix(asm): close the oracle parity gaps in frame addressing and calls
This commit is contained in:
+60
-32
@@ -222,7 +222,7 @@ func encodeARM64Instr(instr *ast.Instr, pc int, offsets map[string]int, fi arm64
|
||||
return nil, fmt.Errorf("WORD expects 1 operand, got %d", len(ops))
|
||||
}
|
||||
return a64wordLE(uint32(immFromOperand(ops[0]))), nil
|
||||
case "B":
|
||||
case "B", "JMP":
|
||||
return encodeARM64Branch(mnem, ops, pc, offsets, false, relocs, resolve)
|
||||
case "BL", "CALL":
|
||||
return encodeARM64Branch(mnem, ops, pc, offsets, true, relocs, resolve)
|
||||
@@ -337,8 +337,9 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
}
|
||||
op := ops[0]
|
||||
|
||||
// External symbol reference: BL sym(SB).
|
||||
if link && op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||||
// Symbol reference: BL sym(SB), or B sym(SB) for a tail call, against a
|
||||
// relocation (R_CALLARM64 either way).
|
||||
if op.Addr.Sym != nil && op.Addr.Sym.Pseudo == "SB" {
|
||||
if relocs != nil {
|
||||
*relocs = append(*relocs, Reloc{
|
||||
Off: 0,
|
||||
@@ -348,8 +349,12 @@ func encodeARM64Branch(mnem string, ops []*ast.Operand, pc int, offsets map[stri
|
||||
Kind: RelArm64Branch,
|
||||
})
|
||||
}
|
||||
// Emit BL with zero offset; the linker fills in the target.
|
||||
return a64wordLE(a64Branch(1, 0)), nil
|
||||
// Emit B/BL with zero offset; the linker fills in the target.
|
||||
bop := uint32(0) // B
|
||||
if link {
|
||||
bop = 1 // BL
|
||||
}
|
||||
return a64wordLE(a64Branch(bop, 0)), nil
|
||||
}
|
||||
|
||||
target := resolve(arm64Label(op))
|
||||
@@ -593,9 +598,9 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
}
|
||||
_, off := arm64MemWithFrame(mem, fi)
|
||||
// Scaled unsigned offset fits if aligned and in range.
|
||||
lt := a64LoadTable[mnem]
|
||||
if lt.size == 0 {
|
||||
lt.size = 3 // default to64-bit for MOV
|
||||
lt, ok := a64LoadTable[mnem]
|
||||
if !ok {
|
||||
lt = a64LoadTable["MOVD"] // the MOV pseudo is a 64-bit access
|
||||
}
|
||||
scale := int32(1) << uint(lt.size)
|
||||
if off >= 0 && off%scale == 0 && off/scale < 4096 {
|
||||
@@ -604,7 +609,10 @@ func arm64MovSize(mnem string, ops []*ast.Operand, fi arm64FrameInfo) int {
|
||||
if off >= -256 && off <= 255 {
|
||||
return 4 // unscaled
|
||||
}
|
||||
return 12 // materialise offset + LDR/STR
|
||||
if _, _, _, ok := arm64SplitOffset(off, scale); ok {
|
||||
return 8 // ADD base, REGTMP + access
|
||||
}
|
||||
return 12 // literal pool range: encoding reports it as unsupported
|
||||
default:
|
||||
return 4 // register move
|
||||
}
|
||||
@@ -824,33 +832,53 @@ func encodeARM64MemOp(mnem string, mem *ast.Operand, reg int, load bool, fi arm6
|
||||
}
|
||||
|
||||
scale := int32(1) << uint(lt.size)
|
||||
if load {
|
||||
// Try scaled unsigned offset first.
|
||||
if off >= 0 && off%scale == 0 {
|
||||
imm12 := uint32(off / scale)
|
||||
if imm12 < 4096 {
|
||||
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(lt.opc), imm12, uint32(rn), uint32(reg))), nil
|
||||
}
|
||||
}
|
||||
// Try unscaled (9-bit signed).
|
||||
if off >= -256 && off <= 255 {
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, lt.opc, off, rn, reg)), nil
|
||||
}
|
||||
// Large offset: materialise in R20 (TMP) and use register-offset.
|
||||
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
|
||||
}
|
||||
// Store: same encoding but opc bits indicate store.
|
||||
storeOpc := a64StoreOpc(lt)
|
||||
if off >= 0 && off%scale == 0 {
|
||||
imm12 := uint32(off / scale)
|
||||
if imm12 < 4096 {
|
||||
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(storeOpc), imm12, uint32(rn), uint32(reg))), nil
|
||||
}
|
||||
var opc int
|
||||
if load {
|
||||
opc = lt.opc
|
||||
} else {
|
||||
opc = storeOpc
|
||||
}
|
||||
// Scaled unsigned offset first, then the unscaled ±255 form.
|
||||
if off >= 0 && off%scale == 0 && off/scale < 4096 {
|
||||
return a64wordLE(a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(off/scale), uint32(rn), uint32(reg))), nil
|
||||
}
|
||||
if off >= -256 && off <= 255 {
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, storeOpc, off, rn, reg)), nil
|
||||
return a64wordLE(a64LSUnscaled(lt.size, lt.V, opc, off, rn, reg)), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%s: offset %d out of range", mnem, off)
|
||||
// Large offset: materialise the base in REGTMP (R27) the way the
|
||||
// toolchain does and access what remains.
|
||||
addImm, addShift, access, ok := arm64SplitOffset(off, scale)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("%s: offset %d out of range (literal pool not supported)", mnem, off)
|
||||
}
|
||||
return a64WordsLE(
|
||||
a64AddSub(1, 0, 0, addShift, uint32(addImm), 31, 27), // ADD $addImm<<shift, SP, R27
|
||||
a64LSU(uint32(lt.size), uint32(lt.V), uint32(opc), uint32(access/scale), 27, uint32(reg)),
|
||||
), nil
|
||||
}
|
||||
|
||||
// arm64SplitOffset decomposes an out-of-range frame offset for a REGTMP
|
||||
// base: an ADD (plain, or shifted left by 12) brings SP near the target and
|
||||
// the access covers what remains. ok is false when no decomposition exists
|
||||
// (offsets at or beyond 16 MiB, where the toolchain falls back to a literal
|
||||
// pool).
|
||||
func arm64SplitOffset(off int32, scale int32) (addImm, addShift uint32, access int32, ok bool) {
|
||||
if off < 0 {
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
// Plain ADD: bring SP to within the largest scaled access.
|
||||
l := min(off, 4095*scale)
|
||||
l -= l % scale
|
||||
if a := off - l; a <= 4095 {
|
||||
return uint32(a), 0, l, true
|
||||
}
|
||||
// Shifted ADD: cover everything but the bits the access imm12 carries.
|
||||
rest := off &^ (0xFFF * scale)
|
||||
if rest >= 0 && rest>>12 <= 4095 {
|
||||
return uint32(rest >> 12), 1, off - rest, true
|
||||
}
|
||||
return 0, 0, 0, false
|
||||
}
|
||||
|
||||
// ---- static symbol references (ADRP + offset) ----
|
||||
|
||||
Reference in New Issue
Block a user